From 2e5439c54c98ec31e0e16cf4dcb568d9efe0c027 Mon Sep 17 00:00:00 2001 From: JetSquirrel Date: Sat, 29 Aug 2026 16:52:10 +0800 Subject: [PATCH 1/5] Add the FOCUS billing ledger and version the app database MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The DuckDB file was a cache, not a ledger: two per-account tables of display-shaped JSON, plus a cost_data table nothing wrote. Sankey, attribution, anomaly detection and month-end freezing all need a fact table to build on, so this adds one. billing.duckdb is a separate file from the application state, because accounts and caches are user-entered or re-fetchable while the ledger is the record that has to survive. It holds fct_charge, ingest_batch, fct_balance_snapshot and dim_fx_rate, named after FOCUS columns so a later ingest of a real CUR or bill export needs no schema change. Writes are whole-period replacement in one transaction: providers re-issue a bill in full mid-month and correct prior months, so a row-by-row upsert would leave behind rows the provider has since deleted and the total would stop matching theirs. Charge ids are a hash of the row's natural key, which is what makes "run ingest twice, get identical results" checkable; rows whose natural keys collide get an occurrence suffix rather than being folded together, so no money goes missing. Nothing normalizes into the ledger yet — PR4 and PR5 do — so the cache tables stay for now. Dropping them before then would mean paying Cost Explorer for a fetch on every launch. The application database is versioned and rebuilt at v1 along the way: cost_data is gone, provider is now source_id, and the credential columns are dropped after any secret still in them is moved to the OS keyring. It is a rebuild rather than a sequence of ALTERs because DuckDB will not alter a table a foreign key points at. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 15 + docs/roadmap.md | 37 ++- src/config.rs | 11 +- src/db.rs | 611 ++++++++++++++++++++++++++-------------- src/ledger/mod.rs | 656 +++++++++++++++++++++++++++++++++++++++++++ src/ledger/schema.rs | 127 +++++++++ src/main.rs | 6 +- 7 files changed, 1235 insertions(+), 228 deletions(-) create mode 100644 src/ledger/mod.rs create mode 100644 src/ledger/schema.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 7dd30d9..2e9cfc7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Budget tracking ### Added +- **FOCUS billing ledger** (roadmap P0/PR2) + - New `billing.duckdb` with `fct_charge`, `ingest_batch`, + `fct_balance_snapshot` and `dim_fx_rate`, named after + [FOCUS](https://focus.finops.org/) columns + - Transactional whole-period replacement keyed by + (source, account, billing period), with deterministic charge ids so a + repeated ingest of an unchanged bill is a no-op + - Amounts stored in the currency they were billed in; conversion is left + to a view (PR6) - **DeepSeek Integration** - DeepSeek API integration for balance queries - Display account balance instead of cost for DeepSeek accounts @@ -23,6 +32,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Support for multiple currencies (CNY, USD) ### Changed +- Billing source registry replaces the `CloudProvider` enum; an unknown + source id is skipped with a warning instead of being read as AWS +- Application database is versioned and rebuilt at schema v1: the dead + `cost_data` table and the credential columns are gone, `provider` is now + `source_id`, and accounts and budgets are carried across. Credentials + live in the OS keyring only. ### Fixed diff --git a/docs/roadmap.md b/docs/roadmap.md index fa48985..90b462e 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -31,16 +31,18 @@ carry their weight for a personal ledger: ## Where we are -The DuckDB file today is a cache, not a ledger. `cost_data` is dead code; -the live path is two per-account cache tables holding JSON blobs. There is -no fact table, so Sankey, attribution, anomaly detection and month-end -freezing have nothing to build on. - -Three structural problems block everything downstream: - -1. **`CloudProvider` is a compile-time enum** with 48 references across 7 - files. Adding a source means editing five `match` arms. `db.rs` also - silently coerces an unknown provider string to `AWS`. +PR1 and PR2 have landed. A source is a registry row rather than an enum +variant, and `billing.duckdb` now holds `fct_charge`, `ingest_batch`, +`fct_balance_snapshot` and `dim_fx_rate` behind a transactional +whole-period write. Nothing normalizes into it yet, so the dashboard still +reads the two per-account cache tables in `cloudbridge.duckdb` — they go +away in PR5, when the last source writes through the ledger. + +Two of the three structural problems are still open: + +1. ~~**`CloudProvider` is a compile-time enum**~~ — replaced by the source + registry in PR1. An unrecognized source id is now skipped with a + warning instead of being silently read as AWS. 2. **Amounts are summed across currencies.** The dashboard total adds AWS USD to Alibaba Cloud CNY and shows the result as one number. 3. **`fetch` and `normalize` are fused.** `get_cost_summary()` returns a @@ -58,7 +60,7 @@ The one-sentence acceptance test for the whole phase: The six changes are a dependency chain — land them in order. -### PR1 · Source registry +### PR1 · Source registry — landed Replace the `CloudProvider` enum with a `SourceId` plus a descriptor table carrying a `Capabilities` struct. The unknown-provider fallback becomes a @@ -69,12 +71,19 @@ its granularity is `SnapshotOnly`, not because it is called DeepSeek. Pure refactor, no behavior change. It comes first because every later PR would otherwise have to edit the same 48 sites. -### PR2 · New database, `fct_charge`, batch table +### PR2 · New database, `fct_charge`, batch table — landed A fresh `billing.duckdb` with a `schema_version` table; credentials move to their own store. Four tables: `fct_charge`, `ingest_batch`, -`fct_balance_snapshot`, `dim_fx_rate`. The three cache tables are dropped — -the data is re-fetchable, so no migration is written. +`fct_balance_snapshot`, `dim_fx_rate`. + +As landed, the two cache tables stay behind in `cloudbridge.duckdb` rather +than being dropped here: they are the only thing feeding the dashboard +until PR4 and PR5 normalize into `fct_charge`, and dropping them early +would mean paying Cost Explorer for a fetch on every launch in between. +`cost_data` — dead code — is gone, and so are the credential columns: the +application database is versioned and rebuilt at v1, keeping accounts and +budgets, with secrets in the OS keyring only. Writes are transactional whole-period replacement keyed by `(provider, account_id, billing_period)`. Providers re-issue a bill in full diff --git a/src/config.rs b/src/config.rs index fa63769..3289b5d 100644 --- a/src/config.rs +++ b/src/config.rs @@ -62,12 +62,21 @@ pub fn get_config_path() -> Result { Ok(data_dir.join("config.json")) } -/// Get database path +/// Path of the application-state database: accounts, budgets and the +/// response caches the dashboard reads. pub fn get_database_path() -> Result { let data_dir = get_app_data_dir()?; Ok(data_dir.join("cloudbridge.duckdb")) } +/// Path of the billing ledger. A separate file from the application state: +/// the ledger is the durable record, everything in `cloudbridge.duckdb` is +/// either user-entered or re-fetchable. +pub fn get_ledger_database_path() -> Result { + let data_dir = get_app_data_dir()?; + Ok(data_dir.join("billing.duckdb")) +} + /// Load configuration pub fn load_config() -> Result { let config_path = get_config_path()?; diff --git a/src/db.rs b/src/db.rs index c84ba14..76dc286 100644 --- a/src/db.rs +++ b/src/db.rs @@ -1,4 +1,10 @@ -//! Database module - Using DuckDB for data storage +//! Application state: cloud accounts, budgets, and the response caches the +//! dashboard reads. +//! +//! Billing facts are not here — they live in [`crate::ledger`], in their own +//! DuckDB file. The split is deliberate: everything in this file is either +//! user-entered or re-fetchable, so it can be rebuilt at any time, while the +//! ledger is the record that has to survive. use anyhow::Result; use chrono::{DateTime, Duration, Utc}; @@ -6,8 +12,8 @@ use duckdb::{params, Connection}; use std::sync::{Arc, Mutex}; use crate::cloud::{ - BudgetInfo, BudgetStatus, CloudAccount, CostData, CostSummary, CostTrend, DailyCost, - ServiceCost, SourceId, + BudgetInfo, BudgetStatus, CloudAccount, CostSummary, CostTrend, DailyCost, ServiceCost, + SourceId, }; use crate::config::get_database_path; use crate::crypto::get_crypto_manager; @@ -20,106 +26,255 @@ lazy_static::lazy_static! { /// Cache time-to-live (hours) const CACHE_TTL_HOURS: i64 = 6; +/// Schema version of the application-state database. +/// +/// v1 is the first version to be recorded at all: it splits the billing +/// ledger out into its own file (see [`crate::ledger`]), removes the dead +/// `cost_data` table, renames `provider` to `source_id` now that a source +/// is a registry row rather than an enum variant, and drops the credential +/// columns for good — secrets live in the OS keyring. +const APP_SCHEMA_VERSION: i32 = 1; + /// Initialize database pub fn init_database() -> Result<()> { let db_path = get_database_path()?; let conn = Connection::open(&db_path)?; + prepare_schema(&conn)?; - // Create cloud accounts table - conn.execute( - r#" - CREATE TABLE IF NOT EXISTS cloud_accounts ( - id VARCHAR PRIMARY KEY, - name VARCHAR NOT NULL, - -- Holds a registry SourceId. Column name predates the registry; - -- PR2 rebuilds this schema, so it is not worth a migration now. - provider VARCHAR NOT NULL, - access_key_id VARCHAR NOT NULL, - secret_access_key VARCHAR NOT NULL, - region VARCHAR, - created_at VARCHAR NOT NULL, - last_synced_at VARCHAR, - enabled BOOLEAN NOT NULL DEFAULT true - ) - "#, - [], - )?; + let mut db = DB_CONNECTION.lock().unwrap(); + *db = Some(conn); - // Create cost data table - conn.execute( + tracing::info!("Database initialized: {:?}", db_path); + Ok(()) +} + +/// Bring a database file up to [`APP_SCHEMA_VERSION`], creating it from +/// scratch if it is empty. +fn prepare_schema(conn: &Connection) -> Result<()> { + conn.execute_batch( r#" - CREATE TABLE IF NOT EXISTS cost_data ( - id INTEGER PRIMARY KEY, - account_id VARCHAR NOT NULL, - date VARCHAR NOT NULL, - service VARCHAR NOT NULL, - amount DOUBLE NOT NULL, - currency VARCHAR NOT NULL, - created_at VARCHAR, - FOREIGN KEY (account_id) REFERENCES cloud_accounts(id) + CREATE TABLE IF NOT EXISTS schema_version ( + version INTEGER PRIMARY KEY, + applied_at VARCHAR NOT NULL ) "#, - [], )?; - // Create index + if current_schema_version(conn)? < 1 { + migrate_to_v1(conn)?; + } + + create_v1_tables(conn)?; + conn.execute( - "CREATE INDEX IF NOT EXISTS idx_cost_data_account_date ON cost_data(account_id, date)", - [], + "INSERT OR REPLACE INTO schema_version (version, applied_at) VALUES (?, ?)", + params![APP_SCHEMA_VERSION, Utc::now().to_rfc3339()], )?; - // Create cost summary cache table - conn.execute( + Ok(()) +} + +/// The v1 shape. Anything the migration already rebuilt is left alone. +/// +/// Tables are declared without foreign keys: DuckDB will not drop or alter a +/// table another table points at, which is what makes a rebuild like +/// [`rebuild_accounts_v1`] necessary in the first place. `delete_account` +/// cleans up dependants instead. +fn create_v1_tables(conn: &Connection) -> Result<()> { + conn.execute_batch( r#" + CREATE TABLE IF NOT EXISTS cloud_accounts ( + id VARCHAR PRIMARY KEY, + name VARCHAR NOT NULL, + -- A registry SourceId; see cloud::registry. Stored verbatim, so + -- these strings are part of the on-disk format. + source_id VARCHAR NOT NULL, + region VARCHAR, + created_at VARCHAR NOT NULL, + last_synced_at VARCHAR, + enabled BOOLEAN NOT NULL DEFAULT true + ); + + CREATE TABLE IF NOT EXISTS budgets ( + account_id VARCHAR PRIMARY KEY, + monthly_budget DOUBLE NOT NULL, + currency VARCHAR NOT NULL, + alert_threshold DOUBLE NOT NULL DEFAULT 80.0, + created_at VARCHAR NOT NULL, + updated_at VARCHAR NOT NULL + ); + + -- The two cache tables below hold display-shaped API responses, not + -- billing facts. They go away in PR5, once every source normalizes + -- into fct_charge and the dashboard reads through the ledger. CREATE TABLE IF NOT EXISTS cost_summary_cache ( - account_id VARCHAR PRIMARY KEY, - current_month_cost DOUBLE NOT NULL, - last_month_cost DOUBLE NOT NULL, - currency VARCHAR NOT NULL, + account_id VARCHAR PRIMARY KEY, + current_month_cost DOUBLE NOT NULL, + last_month_cost DOUBLE NOT NULL, + currency VARCHAR NOT NULL, month_over_month_change DOUBLE NOT NULL, - current_month_details TEXT, - last_month_details TEXT, - cached_at VARCHAR NOT NULL - ) - "#, - [], - )?; + current_month_details TEXT, + last_month_details TEXT, + cached_at VARCHAR NOT NULL + ); - // Create daily cost trend cache table - conn.execute( - r#" CREATE TABLE IF NOT EXISTS cost_trend_cache ( account_id VARCHAR NOT NULL, - date VARCHAR NOT NULL, - amount DOUBLE NOT NULL, - currency VARCHAR NOT NULL, - cached_at VARCHAR NOT NULL, + date VARCHAR NOT NULL, + amount DOUBLE NOT NULL, + currency VARCHAR NOT NULL, + cached_at VARCHAR NOT NULL, PRIMARY KEY (account_id, date) - ) + ); "#, - [], )?; - // Create budgets table + Ok(()) +} + +/// Highest schema version recorded in this file, or 0 for a database that +/// predates versioning (or has just been created). +fn current_schema_version(conn: &Connection) -> Result { + let version: Option = + conn.query_row("SELECT max(version) FROM schema_version", [], |row| { + row.get(0) + })?; + Ok(version.unwrap_or(0)) +} + +fn column_names(conn: &Connection, table: &str) -> Result> { + let mut stmt = conn.prepare("SELECT column_name FROM duckdb_columns() WHERE table_name = ?")?; + let names = stmt + .query_map(params![table], |row| row.get::<_, String>(0))? + .collect::, _>>()?; + Ok(names) +} + +/// Bring a pre-versioning database up to v1. +/// +/// Driven by which columns are actually present, so it is a no-op on a +/// fresh install and safe to re-enter if it is interrupted before the +/// version row is written. +fn migrate_to_v1(conn: &Connection) -> Result<()> { + let account_columns = column_names(conn, "cloud_accounts")?; + if account_columns.is_empty() { + // Fresh install: nothing to carry over. + return Ok(()); + } + + tracing::info!("Migrating application database to schema v1"); + + // Credentials first, because the rebuild is what actually removes them + // from disk. Any account we cannot recover a secret for is named in the + // log — it has to be re-entered. + if account_columns.iter().any(|c| c == "access_key_id") { + recover_legacy_secrets(conn)?; + } + + rebuild_accounts_v1(conn, &account_columns) +} + +/// Rebuild `cloud_accounts` in its v1 shape, carrying the rows across. +/// +/// A rebuild rather than a sequence of `ALTER`s because DuckDB will not +/// alter or drop a table that a foreign key points at, and both `cost_data` +/// and `budgets` pointed at this one. +fn rebuild_accounts_v1(conn: &Connection, account_columns: &[String]) -> Result<()> { + let source_column = if account_columns.iter().any(|c| c == "source_id") { + "source_id" + } else { + "provider" + }; + + // cost_data is dead code and its contents are re-fetchable; budgets is + // copied across. + conn.execute_batch("DROP TABLE IF EXISTS cost_data")?; + + let has_budgets = !column_names(conn, "budgets")?.is_empty(); + if has_budgets { + conn.execute_batch( + "CREATE OR REPLACE TABLE budgets_v1_backup AS SELECT * FROM budgets; + DROP TABLE budgets;", + )?; + } + conn.execute( - r#" - CREATE TABLE IF NOT EXISTS budgets ( - account_id VARCHAR PRIMARY KEY, - monthly_budget DOUBLE NOT NULL, - currency VARCHAR NOT NULL, - alert_threshold DOUBLE NOT NULL DEFAULT 80.0, - created_at VARCHAR NOT NULL, - updated_at VARCHAR NOT NULL, - FOREIGN KEY (account_id) REFERENCES cloud_accounts(id) - ) - "#, + &format!( + "CREATE OR REPLACE TABLE cloud_accounts_v1 AS + SELECT id, name, {source_column} AS source_id, region, + created_at, last_synced_at, enabled + FROM cloud_accounts" + ), [], )?; + conn.execute_batch( + "DROP TABLE cloud_accounts; + ALTER TABLE cloud_accounts_v1 RENAME TO cloud_accounts;", + )?; - let mut db = DB_CONNECTION.lock().unwrap(); - *db = Some(conn); + if has_budgets { + conn.execute_batch( + "CREATE TABLE budgets AS SELECT * FROM budgets_v1_backup; + DROP TABLE budgets_v1_backup;", + )?; + } + + Ok(()) +} + +/// Move any credentials still stored in the database into the OS keyring, +/// before the columns holding them are dropped. +fn recover_legacy_secrets(conn: &Connection) -> Result<()> { + let crypto = match get_crypto_manager() { + Ok(crypto) => crypto, + Err(e) => { + tracing::warn!( + "Cannot decrypt stored credentials ({}); accounts whose secrets \ + are not already in the keyring will have to be re-entered", + e + ); + return Ok(()); + } + }; + + let mut stmt = conn.prepare( + "SELECT id, name, access_key_id, secret_access_key FROM cloud_accounts + WHERE access_key_id <> '' OR secret_access_key <> ''", + )?; + let rows = stmt + .query_map([], |row| { + Ok(( + row.get::<_, String>(0)?, + row.get::<_, String>(1)?, + row.get::<_, String>(2)?, + row.get::<_, String>(3)?, + )) + })? + .collect::, _>>()?; + + for (id, name, encrypted_ak, encrypted_sk) in rows { + if secret_store::get_account_secrets(&id)?.is_some() { + continue; + } + + let access_key_id = crypto.decrypt(&encrypted_ak).unwrap_or_default(); + let secret_access_key = crypto.decrypt(&encrypted_sk).unwrap_or_default(); + if access_key_id.is_empty() && secret_access_key.is_empty() { + tracing::warn!( + "Could not decrypt the stored credentials for account {} ({}); \ + they will have to be re-entered", + name, + id + ); + continue; + } + + if let Err(e) = secret_store::store_account_secrets(&id, &access_key_id, &secret_access_key) + { + tracing::warn!("Failed to move secrets into the keyring for {}: {}", id, e); + } + } - tracing::info!("Database initialized: {:?}", db_path); Ok(()) } @@ -136,30 +291,27 @@ fn get_connection() -> Result> /// Save cloud account pub fn save_account(account: &CloudAccount) -> Result<()> { - // Store secrets in OS keyring for better protection + // Secrets go to the OS keyring; the database holds only the account's + // identity and settings. secret_store::store_account_secrets( &account.id, &account.access_key_id, &account.secret_access_key, )?; - // Still keep an immutable record in DB, but do not store plaintext or encrypted secrets in DB anymore. - // Use empty strings as placeholders for access_key_id/secret_access_key to avoid leaking secrets. let db = get_connection()?; let conn = db.as_ref().unwrap(); conn.execute( r#" - INSERT OR REPLACE INTO cloud_accounts - (id, name, provider, access_key_id, secret_access_key, region, created_at, last_synced_at, enabled) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + INSERT OR REPLACE INTO cloud_accounts + (id, name, source_id, region, created_at, last_synced_at, enabled) + VALUES (?, ?, ?, ?, ?, ?, ?) "#, params![ account.id, account.name, account.source_id.as_str(), - "", - "", account.region, account.created_at.to_rfc3339(), account.last_synced_at.map(|dt| dt.to_rfc3339()), @@ -172,51 +324,29 @@ pub fn save_account(account: &CloudAccount) -> Result<()> { /// Get all cloud accounts pub fn get_all_accounts() -> Result> { - let crypto = get_crypto_manager()?; let db = get_connection()?; let conn = db.as_ref().unwrap(); let mut stmt = conn.prepare( - "SELECT id, name, provider, access_key_id, secret_access_key, region, created_at, last_synced_at, enabled FROM cloud_accounts" + "SELECT id, name, source_id, region, created_at, last_synced_at, enabled FROM cloud_accounts", )?; - let accounts = stmt + let rows = stmt .query_map([], |row| { - let source_id = SourceId::from(row.get::<_, String>(2)?); - - let encrypted_ak: String = row.get(3)?; - let encrypted_sk: String = row.get(4)?; - - let created_at_str: String = row.get(6)?; - let last_synced_str: Option = row.get(7)?; - Ok(( row.get::<_, String>(0)?, row.get::<_, String>(1)?, - source_id, - encrypted_ak, - encrypted_sk, + SourceId::from(row.get::<_, String>(2)?), + row.get::<_, Option>(3)?, + row.get::<_, String>(4)?, row.get::<_, Option>(5)?, - created_at_str, - last_synced_str, - row.get::<_, bool>(8)?, + row.get::<_, bool>(6)?, )) })? .collect::, _>>()?; let mut result = Vec::new(); - for ( - id, - name, - source_id, - encrypted_ak, - encrypted_sk, - region, - created_at_str, - last_synced_str, - enabled, - ) in accounts - { + for (id, name, source_id, region, created_at_str, last_synced_str, enabled) in rows { // An id with no descriptor comes from a build that knew a source this // one does not. Skip the row rather than guessing: silently reading it // as some other provider would sign requests with the wrong scheme and @@ -231,30 +361,21 @@ pub fn get_all_accounts() -> Result> { continue; } - // Try to load secrets from OS keyring first (migration path). If not present, fall back to - // decrypting existing values from DB and migrate them into keyring. + // Credentials live only in the OS keyring (schema v1 moved the last + // of them out of the database). An account whose secrets are gone is + // still listed, so the user can see it and re-enter them. let (access_key_id, secret_access_key) = match secret_store::get_account_secrets(&id)? { - Some((ak, sk)) => (ak, sk), + Some(secrets) => secrets, None => { - // Attempt to decrypt stored values (may be legacy). If successful, migrate to keyring. - let ak_plain = crypto.decrypt(&encrypted_ak).unwrap_or_default(); - let sk_plain = crypto.decrypt(&encrypted_sk).unwrap_or_default(); - - if !ak_plain.is_empty() || !sk_plain.is_empty() { - if let Err(e) = secret_store::store_account_secrets(&id, &ak_plain, &sk_plain) { - tracing::warn!("Failed to migrate secrets into keyring for {}: {}", id, e); - } else { - // Clear secrets in DB to avoid retention of encrypted blobs - let _ = conn.execute( - "UPDATE cloud_accounts SET access_key_id = ?, secret_access_key = ? WHERE id = ?", - params!["", "", id], - ); - } - } - - (ak_plain, sk_plain) + tracing::warn!( + "No credentials in the keyring for account {} ({}); it needs to be re-entered", + name, + id + ); + (String::new(), String::new()) } }; + let created_at = DateTime::parse_from_rfc3339(&created_at_str) .map(|dt| dt.with_timezone(&Utc)) .unwrap_or_else(|_| Utc::now()); @@ -283,12 +404,20 @@ pub fn delete_account(account_id: &str) -> Result<()> { let db = get_connection()?; let conn = db.as_ref().unwrap(); - // First delete associated cost data + // Dependants first: nothing references cloud_accounts through a foreign + // key any more, so the order is ours to keep. + conn.execute( + "DELETE FROM budgets WHERE account_id = ?", + params![account_id], + )?; conn.execute( - "DELETE FROM cost_data WHERE account_id = ?", + "DELETE FROM cost_summary_cache WHERE account_id = ?", + params![account_id], + )?; + conn.execute( + "DELETE FROM cost_trend_cache WHERE account_id = ?", params![account_id], )?; - // Then delete the account conn.execute( "DELETE FROM cloud_accounts WHERE id = ?", params![account_id], @@ -302,84 +431,6 @@ pub fn delete_account(account_id: &str) -> Result<()> { Ok(()) } -/// Save cost data (reserved interface) -#[allow(dead_code)] -pub fn save_cost_data(costs: &[CostData]) -> Result<()> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - for cost in costs { - conn.execute( - r#" - INSERT INTO cost_data (account_id, date, service, amount, currency) - VALUES (?, ?, ?, ?, ?) - "#, - params![ - cost.account_id, - cost.date, - cost.service, - cost.amount, - cost.currency, - ], - )?; - } - - Ok(()) -} - -/// Get account cost data (reserved interface) -#[allow(dead_code)] -pub fn get_cost_data(account_id: &str, start_date: &str, end_date: &str) -> Result> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - let mut stmt = conn.prepare( - "SELECT account_id, date, service, amount, currency FROM cost_data WHERE account_id = ? AND date >= ? AND date <= ? ORDER BY date" - )?; - - let costs = stmt - .query_map(params![account_id, start_date, end_date], |row| { - Ok(CostData { - account_id: row.get(0)?, - date: row.get(1)?, - service: row.get(2)?, - amount: row.get(3)?, - currency: row.get(4)?, - }) - })? - .collect::, _>>()?; - - Ok(costs) -} - -/// Get cost summaries for all accounts (reserved interface) -#[allow(dead_code)] -pub fn get_all_cost_summaries() -> Result> { - let accounts = get_all_accounts()?; - let mut summaries = Vec::new(); - - for account in accounts { - if !account.enabled { - continue; - } - - // Return basic info only, actual costs need to be fetched from cloud - summaries.push(CostSummary { - account_id: account.id, - account_name: account.name, - source_id: account.source_id, - current_month_cost: 0.0, - last_month_cost: 0.0, - currency: "USD".to_string(), - month_over_month_change: 0.0, - current_month_details: Vec::new(), - last_month_details: Vec::new(), - }); - } - - Ok(summaries) -} - // ==================== Cache Functions ==================== /// Check if cost summary cache is valid @@ -807,3 +858,139 @@ pub fn get_all_budget_statuses() -> Result> { Ok(statuses) } + +#[cfg(test)] +mod tests { + use super::*; + + /// The 0.1 schema, as it was written before versioning existed. + const LEGACY_SCHEMA: &str = r#" + CREATE TABLE cloud_accounts ( + id VARCHAR PRIMARY KEY, + name VARCHAR NOT NULL, + provider VARCHAR NOT NULL, + access_key_id VARCHAR NOT NULL, + secret_access_key VARCHAR NOT NULL, + region VARCHAR, + created_at VARCHAR NOT NULL, + last_synced_at VARCHAR, + enabled BOOLEAN NOT NULL DEFAULT true + ); + CREATE TABLE cost_data ( + id INTEGER PRIMARY KEY, + account_id VARCHAR NOT NULL, + date VARCHAR NOT NULL, + service VARCHAR NOT NULL, + amount DOUBLE NOT NULL, + currency VARCHAR NOT NULL, + created_at VARCHAR, + FOREIGN KEY (account_id) REFERENCES cloud_accounts(id) + ); + CREATE TABLE budgets ( + account_id VARCHAR PRIMARY KEY, + monthly_budget DOUBLE NOT NULL, + currency VARCHAR NOT NULL, + alert_threshold DOUBLE NOT NULL DEFAULT 80.0, + created_at VARCHAR NOT NULL, + updated_at VARCHAR NOT NULL, + FOREIGN KEY (account_id) REFERENCES cloud_accounts(id) + ); + INSERT INTO cloud_accounts VALUES + ('acct-1', 'Prod', 'AWS', '', '', 'us-east-1', '2026-08-01T00:00:00+00:00', NULL, true); + INSERT INTO cost_data VALUES + (1, 'acct-1', '2026-08-01', 'EC2', 12.5, 'USD', '2026-08-02T00:00:00+00:00'); + INSERT INTO budgets VALUES + ('acct-1', 100.0, 'USD', 80.0, '2026-08-01T00:00:00+00:00', '2026-08-01T00:00:00+00:00'); + "#; + + fn legacy_database() -> Connection { + let conn = Connection::open_in_memory().expect("in-memory duckdb"); + conn.execute_batch(LEGACY_SCHEMA).expect("legacy schema"); + conn + } + + fn table_exists(conn: &Connection, table: &str) -> bool { + !column_names(conn, table).unwrap().is_empty() + } + + #[test] + fn a_fresh_database_starts_at_the_current_version() { + let conn = Connection::open_in_memory().unwrap(); + prepare_schema(&conn).unwrap(); + + assert_eq!(current_schema_version(&conn).unwrap(), APP_SCHEMA_VERSION); + assert_eq!( + column_names(&conn, "cloud_accounts").unwrap(), + vec![ + "id", + "name", + "source_id", + "region", + "created_at", + "last_synced_at", + "enabled" + ] + ); + assert!(!table_exists(&conn, "cost_data")); + + // Re-opening an up-to-date database changes nothing. + prepare_schema(&conn).unwrap(); + assert_eq!(current_schema_version(&conn).unwrap(), APP_SCHEMA_VERSION); + } + + #[test] + fn the_v1_rebuild_carries_accounts_and_budgets_across() { + let conn = legacy_database(); + let columns = column_names(&conn, "cloud_accounts").unwrap(); + + rebuild_accounts_v1(&conn, &columns).unwrap(); + create_v1_tables(&conn).unwrap(); + + let (id, source_id, region): (String, String, String) = conn + .query_row( + "SELECT id, source_id, region FROM cloud_accounts", + [], + |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), + ) + .unwrap(); + assert_eq!( + (id.as_str(), source_id.as_str(), region.as_str()), + ("acct-1", "AWS", "us-east-1") + ); + + // The credential columns are gone, not merely emptied. + let columns = column_names(&conn, "cloud_accounts").unwrap(); + assert!(!columns.iter().any(|c| c == "access_key_id")); + assert!(!columns.iter().any(|c| c == "secret_access_key")); + assert!(!columns.iter().any(|c| c == "provider")); + + // Dead table dropped, user-entered data kept. + assert!(!table_exists(&conn, "cost_data")); + assert!(!table_exists(&conn, "budgets_v1_backup")); + let budget: f64 = conn + .query_row("SELECT monthly_budget FROM budgets", [], |row| row.get(0)) + .unwrap(); + assert_eq!(budget, 100.0); + } + + #[test] + fn the_v1_rebuild_leaves_an_already_renamed_column_alone() { + // A database that got as far as source_id before being interrupted. + let conn = legacy_database(); + conn.execute_batch( + "DROP TABLE cost_data; + DROP TABLE budgets; + ALTER TABLE cloud_accounts RENAME COLUMN provider TO source_id;", + ) + .unwrap(); + + let columns = column_names(&conn, "cloud_accounts").unwrap(); + rebuild_accounts_v1(&conn, &columns).unwrap(); + create_v1_tables(&conn).unwrap(); + + let source_id: String = conn + .query_row("SELECT source_id FROM cloud_accounts", [], |row| row.get(0)) + .unwrap(); + assert_eq!(source_id, "AWS"); + } +} diff --git a/src/ledger/mod.rs b/src/ledger/mod.rs new file mode 100644 index 0000000..28b40b0 --- /dev/null +++ b/src/ledger/mod.rs @@ -0,0 +1,656 @@ +//! The billing ledger: `fct_charge` and friends, in their own DuckDB file. +//! +//! This is deliberately separate from [`crate::db`], which holds application +//! state — accounts, budgets and the response caches that still feed the +//! dashboard. Those are re-fetchable or user-entered; the ledger is the +//! thing Sankey, attribution, anomaly detection and month-end freezing will +//! be built on, so it gets its own file and its own `schema_version`. +//! +//! Writes are **whole-period replacement**: everything a provider reports +//! for one `(provider, account, billing period)` lands in one transaction +//! that first deletes what was there. Providers re-issue a bill in full +//! mid-month and retroactively correct prior months, so a row-by-row upsert +//! would leave behind entries the provider has since deleted and the total +//! would stop matching theirs. +//! +//! Nothing writes here yet — PR4 (AWS) and PR5 (Alibaba Cloud, DeepSeek) +//! do, through [`replace_period`] and [`record_balance`]. Until then the +//! module is exercised only by its tests. + +// Written by PR4/PR5 and read by PR6; remove once the AWS normalizer lands. +#![allow(dead_code)] + +pub mod schema; + +use anyhow::{anyhow, Result}; +use chrono::{DateTime, Utc}; +use duckdb::{params, Connection}; +use sha2::{Digest, Sha256}; +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use crate::config::get_ledger_database_path; +use schema::TIMESTAMP_FORMAT; + +lazy_static::lazy_static! { + static ref LEDGER_CONNECTION: Arc>> = Arc::new(Mutex::new(None)); +} + +/// What kind of charge a row is, in FOCUS terms. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ChargeCategory { + Usage, + Purchase, + Credit, + Tax, + Adjustment, +} + +impl ChargeCategory { + pub fn as_str(self) -> &'static str { + match self { + Self::Usage => "Usage", + Self::Purchase => "Purchase", + Self::Credit => "Credit", + Self::Tax => "Tax", + Self::Adjustment => "Adjustment", + } + } +} + +/// How much weight the amount on a row carries. +/// +/// Keeps authoritative bills, unit-price-derived amounts and pure usage +/// records in one table without anyone mistaking a shadow cost for money +/// actually spent. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CostBasis { + /// Straight from the provider's bill. + Authoritative, + /// Computed from usage and a unit price. + Derived, + /// A projection or an allocation. + Estimated, + /// Usage with no amount attached; `billed_cost` is NULL. + Absent, +} + +impl CostBasis { + pub fn as_str(self) -> &'static str { + match self { + Self::Authoritative => "authoritative", + Self::Derived => "derived", + Self::Estimated => "estimated", + Self::Absent => "absent", + } + } +} + +/// The unit of replacement: one account's charges for one billing period. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PeriodKey { + /// Registry `SourceId`, stored verbatim. + pub provider: String, + /// Our `cloud_accounts.id`, not the provider-side account number. + pub account_id: String, + /// `YYYY-MM`. + pub billing_period: String, +} + +impl PeriodKey { + pub fn new( + provider: impl Into, + account_id: impl Into, + billing_period: impl Into, + ) -> Self { + Self { + provider: provider.into(), + account_id: account_id.into(), + billing_period: billing_period.into(), + } + } +} + +/// One row of `fct_charge`, minus the columns that come from the +/// [`PeriodKey`] it is written under. +#[derive(Debug, Clone)] +pub struct Charge { + pub charge_period_start: DateTime, + pub charge_period_end: DateTime, + pub charge_category: ChargeCategory, + pub cost_basis: CostBasis, + pub billing_currency: String, + /// Provider-side account, when it differs from the credential's own + /// (an AWS payer account reports its linked accounts). + pub billing_account_id: Option, + pub charge_description: Option, + pub service_name: Option, + pub service_category: Option, + pub resource_id: Option, + pub resource_name: Option, + pub region_id: Option, + pub billed_cost: Option, + pub effective_cost: Option, + pub list_cost: Option, + pub pricing_quantity: Option, + pub pricing_unit: Option, + /// JSON object text, or `None` when the source reports no tags. + pub tags: Option, +} + +impl Charge { + /// An authoritative usage charge with everything optional left unset. + /// Fill the rest in with struct update syntax. + pub fn new( + charge_period_start: DateTime, + charge_period_end: DateTime, + billing_currency: impl Into, + ) -> Self { + Self { + charge_period_start, + charge_period_end, + charge_category: ChargeCategory::Usage, + cost_basis: CostBasis::Authoritative, + billing_currency: billing_currency.into(), + billing_account_id: None, + charge_description: None, + service_name: None, + service_category: None, + resource_id: None, + resource_name: None, + region_id: None, + billed_cost: None, + effective_cost: None, + list_cost: None, + pricing_quantity: None, + pricing_unit: None, + tags: None, + } + } +} + +/// A point-in-time balance for a source that reports state rather than +/// charges. +#[derive(Debug, Clone)] +pub struct BalanceSnapshot { + pub provider: String, + pub account_id: String, + pub observed_at: DateTime, + pub balance: f64, + pub granted_balance: Option, + pub topped_up_balance: Option, + pub currency: String, +} + +/// Open (creating if needed) the ledger database and apply its schema. +pub fn init_ledger() -> Result<()> { + let path = get_ledger_database_path()?; + let conn = Connection::open(&path)?; + schema::apply(&conn)?; + + let mut ledger = LEDGER_CONNECTION.lock().unwrap(); + *ledger = Some(conn); + + tracing::info!("Ledger initialized: {:?}", path); + Ok(()) +} + +fn with_connection(f: impl FnOnce(&mut Connection) -> Result) -> Result { + let mut guard = LEDGER_CONNECTION + .lock() + .map_err(|e| anyhow!("Failed to lock ledger connection: {}", e))?; + let conn = guard + .as_mut() + .ok_or_else(|| anyhow!("Ledger not initialized"))?; + f(conn) +} + +/// Replace everything stored for `key` with `charges`, in one transaction. +/// +/// Returns the id of the batch the rows were written under. `source_ref` +/// points at the raw payload the rows were normalized from; it stays `None` +/// until PR3 persists Parquet. +pub fn replace_period( + key: &PeriodKey, + charges: &[Charge], + source_ref: Option<&str>, +) -> Result { + with_connection(|conn| write_period(conn, key, charges, source_ref)) +} + +/// Record a balance observation. Re-observing the same instant overwrites, +/// so a repeated ingest of one payload is a no-op. +pub fn record_balance(snapshot: &BalanceSnapshot) -> Result<()> { + with_connection(|conn| write_balance(conn, snapshot)) +} + +fn write_period( + conn: &mut Connection, + key: &PeriodKey, + charges: &[Charge], + source_ref: Option<&str>, +) -> Result { + let batch_id = uuid::Uuid::new_v4().to_string(); + let now = Utc::now().format(TIMESTAMP_FORMAT).to_string(); + let ids = charge_ids(key, charges); + + let tx = conn.transaction()?; + + // Older batches for this period stay as history — freezing (P3) needs + // to know a period was ingested more than once — but only one of them + // has rows in fct_charge. + tx.execute( + "UPDATE ingest_batch SET status = 'superseded' + WHERE provider = ? AND account_id = ? AND billing_period = ? AND status = 'complete'", + params![key.provider, key.account_id, key.billing_period], + )?; + tx.execute( + "DELETE FROM fct_charge WHERE provider = ? AND account_id = ? AND billing_period = ?", + params![key.provider, key.account_id, key.billing_period], + )?; + tx.execute( + "INSERT INTO ingest_batch + (batch_id, provider, account_id, billing_period, started_at, completed_at, + status, row_count, source_ref) + VALUES (?, ?, ?, ?, CAST(? AS TIMESTAMP), CAST(? AS TIMESTAMP), 'complete', ?, ?)", + params![ + batch_id, + key.provider, + key.account_id, + key.billing_period, + now, + now, + charges.len() as i64, + source_ref, + ], + )?; + + { + let mut stmt = tx.prepare( + "INSERT INTO fct_charge + (charge_id, batch_id, provider, account_id, billing_account_id, billing_period, + charge_period_start, charge_period_end, charge_category, charge_description, + service_name, service_category, resource_id, resource_name, region_id, + billed_cost, effective_cost, list_cost, billing_currency, cost_basis, + pricing_quantity, pricing_unit, tags, created_at) + VALUES (?, ?, ?, ?, ?, ?, CAST(? AS TIMESTAMP), CAST(? AS TIMESTAMP), ?, ?, + ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, CAST(? AS TIMESTAMP))", + )?; + + for (charge, charge_id) in charges.iter().zip(&ids) { + stmt.execute(params![ + charge_id, + batch_id, + key.provider, + key.account_id, + charge.billing_account_id, + key.billing_period, + charge + .charge_period_start + .format(TIMESTAMP_FORMAT) + .to_string(), + charge + .charge_period_end + .format(TIMESTAMP_FORMAT) + .to_string(), + charge.charge_category.as_str(), + charge.charge_description, + charge.service_name, + charge.service_category, + charge.resource_id, + charge.resource_name, + charge.region_id, + charge.billed_cost, + charge.effective_cost, + charge.list_cost, + charge.billing_currency, + charge.cost_basis.as_str(), + charge.pricing_quantity, + charge.pricing_unit, + charge.tags, + now, + ])?; + } + } + + tx.commit()?; + + tracing::info!( + "Ledger: wrote {} charges for {}/{} {} (batch {})", + charges.len(), + key.provider, + key.account_id, + key.billing_period, + batch_id + ); + Ok(batch_id) +} + +fn write_balance(conn: &mut Connection, snapshot: &BalanceSnapshot) -> Result<()> { + let now = Utc::now().format(TIMESTAMP_FORMAT).to_string(); + + conn.execute( + "INSERT OR REPLACE INTO fct_balance_snapshot + (provider, account_id, observed_at, balance, granted_balance, topped_up_balance, + currency, created_at) + VALUES (?, ?, CAST(? AS TIMESTAMP), ?, ?, ?, ?, CAST(? AS TIMESTAMP))", + params![ + snapshot.provider, + snapshot.account_id, + snapshot.observed_at.format(TIMESTAMP_FORMAT).to_string(), + snapshot.balance, + snapshot.granted_balance, + snapshot.topped_up_balance, + snapshot.currency, + now, + ], + )?; + + Ok(()) +} + +/// Deterministic ids for a batch of charges. +/// +/// The id is a hash of the row's natural key, so re-ingesting an unchanged +/// bill produces the same `charge_id` for the same charge — that is what +/// makes "run ingest twice, get identical results" checkable, and what lets +/// a later diff say which rows the provider actually changed. +/// +/// Providers do emit rows whose natural keys collide (two charges for the +/// same service, day and category, split by something we do not store). A +/// collision gets an occurrence suffix rather than being folded into one +/// row, so no money goes missing; the suffix follows the provider's own +/// ordering of the payload. +fn charge_ids(key: &PeriodKey, charges: &[Charge]) -> Vec { + let mut seen: HashMap = HashMap::new(); + let mut ids = Vec::with_capacity(charges.len()); + + for charge in charges { + let natural_key = [ + key.provider.as_str(), + key.account_id.as_str(), + key.billing_period.as_str(), + &charge + .charge_period_start + .format(TIMESTAMP_FORMAT) + .to_string(), + &charge + .charge_period_end + .format(TIMESTAMP_FORMAT) + .to_string(), + charge.charge_category.as_str(), + charge.billing_account_id.as_deref().unwrap_or(""), + charge.service_name.as_deref().unwrap_or(""), + charge.resource_id.as_deref().unwrap_or(""), + charge.region_id.as_deref().unwrap_or(""), + charge.charge_description.as_deref().unwrap_or(""), + charge.pricing_unit.as_deref().unwrap_or(""), + charge.billing_currency.as_str(), + ] + .join("\u{1f}"); + + let occurrence = seen.entry(natural_key.clone()).or_insert(0); + let mut hasher = Sha256::new(); + hasher.update(natural_key.as_bytes()); + hasher.update(format!("\u{1f}{}", occurrence).as_bytes()); + *occurrence += 1; + + ids.push(hex::encode(hasher.finalize())[..32].to_string()); + } + + ids +} + +#[cfg(test)] +mod tests { + use super::*; + use chrono::TimeZone; + + fn conn() -> Connection { + let conn = Connection::open_in_memory().expect("in-memory duckdb"); + schema::apply(&conn).expect("schema applies"); + conn + } + + fn at(day: u32) -> DateTime { + Utc.with_ymd_and_hms(2026, 8, day, 0, 0, 0).unwrap() + } + + fn usage(service: &str, amount: f64, day: u32) -> Charge { + Charge { + service_name: Some(service.to_string()), + billed_cost: Some(amount), + effective_cost: Some(amount), + ..Charge::new(at(day), at(day + 1), "USD") + } + } + + fn key() -> PeriodKey { + PeriodKey::new("AWS", "acct-1", "2026-08") + } + + /// (charge_id, service, billed_cost) for a period, in id order. + fn stored(conn: &Connection, key: &PeriodKey) -> Vec<(String, String, f64)> { + let mut stmt = conn + .prepare( + "SELECT charge_id, service_name, billed_cost FROM fct_charge + WHERE provider = ? AND account_id = ? AND billing_period = ? + ORDER BY charge_id", + ) + .unwrap(); + stmt.query_map( + params![key.provider, key.account_id, key.billing_period], + |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), + ) + .unwrap() + .collect::, _>>() + .unwrap() + } + + fn scalar_i64(conn: &Connection, sql: &str) -> i64 { + conn.query_row(sql, [], |row| row.get(0)).unwrap() + } + + #[test] + fn ingesting_the_same_bill_twice_is_a_no_op() { + let mut conn = conn(); + let key = key(); + let charges = vec![usage("EC2", 12.5, 1), usage("S3", 0.75, 1)]; + + write_period(&mut conn, &key, &charges, None).unwrap(); + let first = stored(&conn, &key); + + write_period(&mut conn, &key, &charges, None).unwrap(); + let second = stored(&conn, &key); + + assert_eq!(first, second); + assert_eq!(first.len(), 2); + } + + #[test] + fn replacement_drops_rows_the_provider_no_longer_reports() { + let mut conn = conn(); + let key = key(); + + write_period( + &mut conn, + &key, + &[ + usage("EC2", 12.5, 1), + usage("S3", 0.75, 1), + usage("RDS", 3.0, 1), + ], + None, + ) + .unwrap(); + + // The provider reissues the period without RDS and with a corrected + // EC2 amount. + write_period( + &mut conn, + &key, + &[usage("EC2", 11.0, 1), usage("S3", 0.75, 1)], + None, + ) + .unwrap(); + + let rows = stored(&conn, &key); + assert_eq!(rows.len(), 2); + let services: Vec<&str> = rows.iter().map(|(_, s, _)| s.as_str()).collect(); + assert!(!services.contains(&"RDS")); + let ec2 = rows.iter().find(|(_, s, _)| s == "EC2").unwrap(); + assert_eq!(ec2.2, 11.0); + } + + #[test] + fn every_ingest_is_recorded_but_only_the_last_one_holds_rows() { + let mut conn = conn(); + let key = key(); + + write_period(&mut conn, &key, &[usage("EC2", 12.5, 1)], None).unwrap(); + let batch = write_period(&mut conn, &key, &[usage("EC2", 11.0, 1)], None).unwrap(); + + assert_eq!(scalar_i64(&conn, "SELECT count(*) FROM ingest_batch"), 2); + assert_eq!( + scalar_i64( + &conn, + "SELECT count(*) FROM ingest_batch WHERE status = 'superseded'" + ), + 1 + ); + assert_eq!( + conn.query_row::("SELECT batch_id FROM fct_charge", [], |r| r.get(0)) + .unwrap(), + batch + ); + } + + #[test] + fn replacing_one_period_leaves_the_others_alone() { + let mut conn = conn(); + let august = key(); + let july = PeriodKey::new("AWS", "acct-1", "2026-07"); + let other_account = PeriodKey::new("AWS", "acct-2", "2026-08"); + + write_period(&mut conn, &july, &[usage("EC2", 9.0, 1)], None).unwrap(); + write_period(&mut conn, &other_account, &[usage("EC2", 5.0, 1)], None).unwrap(); + write_period(&mut conn, &august, &[usage("EC2", 12.5, 1)], None).unwrap(); + write_period(&mut conn, &august, &[], None).unwrap(); + + assert!(stored(&conn, &august).is_empty()); + assert_eq!(stored(&conn, &july).len(), 1); + assert_eq!(stored(&conn, &other_account).len(), 1); + } + + #[test] + fn charges_that_differ_only_by_an_unstored_dimension_both_survive() { + let mut conn = conn(); + let key = key(); + let charges = vec![usage("EC2", 12.5, 1), usage("EC2", 4.0, 1)]; + + write_period(&mut conn, &key, &charges, None).unwrap(); + let first = stored(&conn, &key); + assert_eq!(first.len(), 2); + + write_period(&mut conn, &key, &charges, None).unwrap(); + assert_eq!(stored(&conn, &key), first); + } + + #[test] + fn a_charge_without_an_amount_is_storable() { + let mut conn = conn(); + let key = key(); + + write_period( + &mut conn, + &key, + &[Charge { + service_name: Some("Claude Code".to_string()), + cost_basis: CostBasis::Absent, + pricing_quantity: Some(18_000.0), + pricing_unit: Some("Tokens".to_string()), + ..Charge::new(at(1), at(2), "USD") + }], + None, + ) + .unwrap(); + + let (basis, unit, cost) = conn + .query_row( + "SELECT cost_basis, pricing_unit, billed_cost FROM fct_charge", + [], + |row| { + Ok(( + row.get::<_, String>(0)?, + row.get::<_, String>(1)?, + row.get::<_, Option>(2)?, + )) + }, + ) + .unwrap(); + + assert_eq!(basis, "absent"); + assert_eq!(unit, "Tokens"); + assert_eq!(cost, None); + } + + #[test] + fn re_observing_a_balance_at_the_same_instant_overwrites() { + let mut conn = conn(); + let snapshot = BalanceSnapshot { + provider: "DeepSeek".to_string(), + account_id: "acct-3".to_string(), + observed_at: at(1), + balance: 42.0, + granted_balance: Some(10.0), + topped_up_balance: Some(32.0), + currency: "CNY".to_string(), + }; + + write_balance(&mut conn, &snapshot).unwrap(); + write_balance( + &mut conn, + &BalanceSnapshot { + balance: 41.0, + ..snapshot.clone() + }, + ) + .unwrap(); + + assert_eq!( + scalar_i64(&conn, "SELECT count(*) FROM fct_balance_snapshot"), + 1 + ); + let balance: f64 = conn + .query_row("SELECT balance FROM fct_balance_snapshot", [], |r| r.get(0)) + .unwrap(); + assert_eq!(balance, 41.0); + } + + #[test] + fn timestamps_survive_the_round_trip() { + let mut conn = conn(); + let key = key(); + + write_period(&mut conn, &key, &[usage("EC2", 1.0, 3)], None).unwrap(); + + let start: String = conn + .query_row( + "SELECT CAST(charge_period_start AS VARCHAR) FROM fct_charge", + [], + |r| r.get(0), + ) + .unwrap(); + assert!(start.starts_with("2026-08-03 00:00:00"), "got {start}"); + + // PR6's view casts this column to DATE for the ASOF join on rates. + let as_date: String = conn + .query_row( + "SELECT CAST(charge_period_start::DATE AS VARCHAR) FROM fct_charge", + [], + |r| r.get(0), + ) + .unwrap(); + assert_eq!(as_date, "2026-08-03"); + } +} diff --git a/src/ledger/schema.rs b/src/ledger/schema.rs new file mode 100644 index 0000000..5fba665 --- /dev/null +++ b/src/ledger/schema.rs @@ -0,0 +1,127 @@ +//! Physical schema of the billing ledger. +//! +//! Column names follow [FOCUS](https://focus.finops.org/) so that a later +//! ingest of a real CUR or Alibaba Cloud bill export needs no schema change. +//! Only the three concepts that carry their weight for a personal ledger are +//! implemented: `billed_cost`, `effective_cost` and `charge_category`. +//! +//! Timestamps are stored as `TIMESTAMP` in UTC. Values are bound as +//! `'%Y-%m-%d %H:%M:%S'` strings through an explicit `CAST`, and read back +//! through `CAST(col AS VARCHAR)`, so no DuckDB feature flag is needed to +//! move a `DateTime` in or out. + +use anyhow::Result; +use chrono::Utc; +use duckdb::{params, Connection}; + +/// Bumped whenever the statements below change shape. +pub const SCHEMA_VERSION: i32 = 1; + +/// Format used for every `TIMESTAMP` bind and parse in this module. +pub const TIMESTAMP_FORMAT: &str = "%Y-%m-%d %H:%M:%S"; + +/// Create the ledger tables and record the schema version. +/// +/// Idempotent: safe to call on every start. +pub fn apply(conn: &Connection) -> Result<()> { + conn.execute_batch( + r#" + CREATE TABLE IF NOT EXISTS schema_version ( + version INTEGER PRIMARY KEY, + applied_at TIMESTAMP NOT NULL + ); + + -- One ingest of one (provider, account, billing period). Whole-period + -- replacement is keyed on the same triple, so a batch is the unit + -- P3 month-end freezing will pin a period to. + CREATE TABLE IF NOT EXISTS ingest_batch ( + batch_id VARCHAR PRIMARY KEY, + provider VARCHAR NOT NULL, + account_id VARCHAR NOT NULL, + billing_period VARCHAR NOT NULL, -- YYYY-MM + started_at TIMESTAMP NOT NULL, + completed_at TIMESTAMP, + status VARCHAR NOT NULL, -- complete | superseded + row_count BIGINT NOT NULL DEFAULT 0, + -- Path of the raw payload this batch was normalized from. + -- Filled in by PR3, once fetch persists Parquet. + source_ref VARCHAR + ); + + -- The fact table. One row per charge, in the currency the provider + -- billed it in; conversion happens in a view (PR6), never here. + CREATE TABLE IF NOT EXISTS fct_charge ( + charge_id VARCHAR PRIMARY KEY, + batch_id VARCHAR NOT NULL, + provider VARCHAR NOT NULL, + account_id VARCHAR NOT NULL, + billing_account_id VARCHAR, + billing_period VARCHAR NOT NULL, -- YYYY-MM + charge_period_start TIMESTAMP NOT NULL, + charge_period_end TIMESTAMP NOT NULL, + charge_category VARCHAR NOT NULL, -- Usage | Purchase | Credit | Tax | Adjustment + charge_description VARCHAR, + service_name VARCHAR, + service_category VARCHAR, + resource_id VARCHAR, + resource_name VARCHAR, + region_id VARCHAR, + -- Nullable on purpose: a usage record with no authoritative + -- amount is representable, and `cost_basis` says which kind of + -- figure this is so the UI can mark a derived one. + billed_cost DOUBLE, + effective_cost DOUBLE, + list_cost DOUBLE, + billing_currency VARCHAR NOT NULL, + cost_basis VARCHAR NOT NULL, -- authoritative | derived | estimated | absent + pricing_quantity DOUBLE, + -- Not restricted to cloud units: holds GB-Mo and Hrs today, + -- Tokens when model-provider usage lands. + pricing_unit VARCHAR, + tags VARCHAR, -- JSON object text + created_at TIMESTAMP NOT NULL + ); + + CREATE INDEX IF NOT EXISTS idx_charge_period + ON fct_charge (provider, account_id, billing_period); + CREATE INDEX IF NOT EXISTS idx_charge_batch + ON fct_charge (batch_id); + + -- A balance is state, not a charge: sources that only report one + -- (DeepSeek today) land here, and only their top-ups become charges. + CREATE TABLE IF NOT EXISTS fct_balance_snapshot ( + provider VARCHAR NOT NULL, + account_id VARCHAR NOT NULL, + observed_at TIMESTAMP NOT NULL, + balance DOUBLE NOT NULL, + granted_balance DOUBLE, + topped_up_balance DOUBLE, + currency VARCHAR NOT NULL, + created_at TIMESTAMP NOT NULL, + PRIMARY KEY (provider, account_id, observed_at) + ); + + -- Rates are dated because they get corrected, and the reporting + -- currency is the user's to change. PR6 seeds this and reads it + -- through an ASOF join. + CREATE TABLE IF NOT EXISTS dim_fx_rate ( + from_ccy VARCHAR NOT NULL, + to_ccy VARCHAR NOT NULL, + rate_date DATE NOT NULL, + rate DOUBLE NOT NULL, + source VARCHAR NOT NULL, + PRIMARY KEY (from_ccy, to_ccy, rate_date) + ); + "#, + )?; + + conn.execute( + "INSERT OR REPLACE INTO schema_version (version, applied_at) VALUES (?, CAST(? AS TIMESTAMP))", + params![ + SCHEMA_VERSION, + Utc::now().format(TIMESTAMP_FORMAT).to_string() + ], + )?; + + Ok(()) +} diff --git a/src/main.rs b/src/main.rs index c3a45bf..dce7aa9 100644 --- a/src/main.rs +++ b/src/main.rs @@ -3,6 +3,7 @@ mod cloud; mod config; mod crypto; mod db; +mod ledger; mod secret_store; mod ui; @@ -49,10 +50,13 @@ fn main() { ); cx.spawn(async move |cx| { - // Initialize database + // Initialize databases: application state, then the billing ledger. if let Err(e) = db::init_database() { tracing::error!("Database initialization failed: {}", e); } + if let Err(e) = ledger::init_ledger() { + tracing::error!("Ledger initialization failed: {}", e); + } cx.open_window( WindowOptions { From 4a52de83112368e909ce3e284fba70d5e1198a62 Mon Sep 17 00:00:00 2001 From: JetSquirrel Date: Sat, 29 Aug 2026 23:56:53 +0800 Subject: [PATCH 2/5] Split fetch from normalize and store raw payloads as Parquet MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fetching and interpreting were one step: get_cost_summary() called Cost Explorer and returned a display-shaped struct, so the response existed only long enough to be reshaped. Cost Explorer bills per request, which made every mapping change cost another round of paid calls, and there was no way to assert on billing logic without a live account. fetch now returns what the provider sent, unchanged, and nothing else. The payloads are written as Hive-partitioned Parquet under raw/provider=/account=/billing_period=/batch=, the layout a bill export bucket already uses, so P1's S3/OSS channel replaces only the fetch half. normalize is a pure function from a stored batch to FOCUS rows: no clock, no network, no database — the fetch time it needs rides along on the batch. Each source is now tested against a recorded response. The two Cost Explorer methods were near-identical copies differing in one field of the request body; they collapse into one signed call plus a request builder. What the normalizers do not do yet is the mapping detail PR4 and PR5 own. AWS asks only for UnblendedCost and files every row as Usage, because without RECORD_TYPE in the grouping the payload cannot tell a credit from a charge and guessing would put refunds on the wrong side of the total. Alibaba Cloud keeps the discount visible as the gap between billed_cost and list_cost rather than inventing Credit rows. DeepSeek writes balance snapshots and derives no top-ups. ingest_period ties the three steps to one batch id, shared by the raw partition and the ledger batch; renormalize_period replays the newest stored batch without fetching, which is the whole reason the raw store exists. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 10 + docs/roadmap.md | 37 +- src/cloud/aliyun.rs | 155 ++++++- src/cloud/aws.rs | 412 +++++++++++++------ src/cloud/deepseek.rs | 137 +++++- src/cloud/mod.rs | 73 +++- src/cloud/raw.rs | 366 ++++++++++++++++ src/cloud/registry.rs | 4 +- src/cloud/testdata/aliyun_bill_overview.json | 36 ++ src/cloud/testdata/aws_cost_and_usage.json | 66 +++ src/cloud/testdata/deepseek_balance.json | 11 + src/config.rs | 7 + src/ingest.rs | 141 +++++++ src/ledger/mod.rs | 58 ++- src/ledger/schema.rs | 4 +- src/main.rs | 1 + 16 files changed, 1335 insertions(+), 183 deletions(-) create mode 100644 src/cloud/raw.rs create mode 100644 src/cloud/testdata/aliyun_bill_overview.json create mode 100644 src/cloud/testdata/aws_cost_and_usage.json create mode 100644 src/cloud/testdata/deepseek_balance.json create mode 100644 src/ingest.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 2e9cfc7..bc89c7c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 repeated ingest of an unchanged bill is a no-op - Amounts stored in the currency they were billed in; conversion is left to a view (PR6) +- **Raw payload store** (roadmap P0/PR3) + - `fetch` persists provider responses unchanged as Hive-partitioned + Parquet under `raw/provider=…/account=…/billing_period=…/batch=…/`, + the same layout a bill export bucket uses + - `normalize` is a pure function from a stored batch to FOCUS rows, so + billing logic is testable from a recorded response and a mapping fix + replays payloads on disk instead of paying for another fetch - **DeepSeek Integration** - DeepSeek API integration for balance queries - Display account balance instead of cost for DeepSeek accounts @@ -32,6 +39,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Support for multiple currencies (CNY, USD) ### Changed +- `CloudService` is now `BillingSource`, with `fetch` and `normalize` split + apart: the first touches the network and interprets nothing, the second + interprets and touches nothing - Billing source registry replaces the `CloudProvider` enum; an unknown source id is skipped with a warning instead of being read as AWS - Application database is versioned and rebuilt at schema v1: the dead diff --git a/docs/roadmap.md b/docs/roadmap.md index 90b462e..cb2c79f 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -31,24 +31,29 @@ carry their weight for a personal ledger: ## Where we are -PR1 and PR2 have landed. A source is a registry row rather than an enum -variant, and `billing.duckdb` now holds `fct_charge`, `ingest_batch`, +PR1 through PR3 have landed. A source is a registry row rather than an +enum variant; `billing.duckdb` holds `fct_charge`, `ingest_batch`, `fct_balance_snapshot` and `dim_fx_rate` behind a transactional -whole-period write. Nothing normalizes into it yet, so the dashboard still -reads the two per-account cache tables in `cloudbridge.duckdb` — they go -away in PR5, when the last source writes through the ledger. +whole-period write; and every source now fetches raw payloads to Parquet +and normalizes them through a pure function, tested against a recorded +response. -Two of the three structural problems are still open: +What is left in P0 is the mapping detail (PR4, PR5) and the read path +(PR6). The dashboard still reads the two per-account cache tables in +`cloudbridge.duckdb` — they go away in PR5, when the last source writes +through the ledger. + +One of the three structural problems is still open: 1. ~~**`CloudProvider` is a compile-time enum**~~ — replaced by the source registry in PR1. An unrecognized source id is now skipped with a warning instead of being silently read as AWS. 2. **Amounts are summed across currencies.** The dashboard total adds AWS USD to Alibaba Cloud CNY and shows the result as one number. -3. **`fetch` and `normalize` are fused.** `get_cost_summary()` returns a - display-shaped struct straight from the API. Cost Explorer charges per - request, so any schema change means paying to re-fetch, and there is no - way to unit-test the billing logic. +3. ~~**`fetch` and `normalize` are fused.**~~ — split in PR3. `fetch` + persists what the provider returned and interprets nothing; + `normalize` interprets and touches nothing, so a mapping fix replays + payloads already on disk instead of paying Cost Explorer again. ## P0 — FOCUS normalization @@ -107,7 +112,7 @@ deferred: - `pricing_unit` is not restricted to cloud units. Today it holds `GB-Mo` and `Hrs`; tomorrow it holds `Tokens`. -### PR3 · Split fetch from normalize, land raw Parquet +### PR3 · Split fetch from normalize, land raw Parquet — landed `BillingSource` replaces `CloudService`. `fetch` retrieves and persists raw payloads unchanged; `normalize` is a pure function from raw to FOCUS rows. @@ -125,6 +130,16 @@ so P1's S3/OSS export channel only replaces the `fetch` implementation — A pure `normalize` is also the first time billing logic becomes testable: record one API response per provider as a fixture and assert on the rows. +As landed, `CloudService` became `BillingSource` and all three sources +implement both halves, so the pipeline is whole end to end +(`ingest::ingest_period`) and re-normalizing without fetching is a +supported operation (`ingest::renormalize_period`). What the normalizers +do *not* do yet is the mapping detail PR4 and PR5 own: AWS asks only for +`UnblendedCost` and files everything as `Usage`, Alibaba Cloud records the +discount as the gap between `billed_cost` and `list_cost` rather than as +`Credit` rows, and DeepSeek writes balance snapshots without deriving +top-ups. + ### PR4 · AWS to FOCUS Cost Explorer currently requests only `UnblendedCost`, grouped by `SERVICE`. diff --git a/src/cloud/aliyun.rs b/src/cloud/aliyun.rs index 3897c54..be30c63 100644 --- a/src/cloud/aliyun.rs +++ b/src/cloud/aliyun.rs @@ -7,7 +7,12 @@ use serde::Deserialize; use sha1::Sha1; use std::collections::BTreeMap; -use super::{CloudService, CostData, CostSummary, ServiceCost, SourceId}; +use super::raw::RawPart; +use super::{ + BillingPeriod, BillingSource, CostData, CostSummary, Normalized, RawBatch, ServiceCost, + SourceId, +}; +use crate::ledger::Charge; type HmacSha1 = Hmac; @@ -217,7 +222,7 @@ impl AliyunCloudService { } } -impl CloudService for AliyunCloudService { +impl BillingSource for AliyunCloudService { fn validate_credentials(&self) -> Result { // Try calling a simple API to validate credentials let now = Utc::now(); @@ -232,6 +237,21 @@ impl CloudService for AliyunCloudService { } } + fn fetch(&self, period: &BillingPeriod) -> Result> { + let billing_cycle = period.label(); + let body = self.call_bss_api("QueryBillOverview", &[("BillingCycle", &billing_cycle)])?; + + Ok(vec![RawPart::new( + PART_BILL_OVERVIEW, + format!("QueryBillOverview BillingCycle={}", billing_cycle), + body, + )]) + } + + fn normalize(&self, batch: &RawBatch) -> Result { + normalize(batch) + } + fn get_cost_data(&self, start_date: &str, end_date: &str) -> Result> { // Alibaba Cloud queries by month, extract year-month let billing_cycle = &start_date[..7]; // YYYY-MM @@ -389,6 +409,76 @@ fn parse_bill_overview(response: &BillOverviewResponse) -> (f64, Vec Result { + let part = batch + .part(PART_BILL_OVERVIEW) + .ok_or_else(|| anyhow!("Raw batch has no '{}' payload", PART_BILL_OVERVIEW))?; + let response: BillOverviewResponse = serde_json::from_str(&part.body) + .map_err(|e| anyhow!("Failed to parse bill overview: {}", e))?; + + let start = batch + .period + .start() + .and_hms_opt(0, 0, 0) + .expect("midnight exists") + .and_utc(); + let end = batch + .period + .end_exclusive() + .and_hms_opt(0, 0, 0) + .expect("midnight exists") + .and_utc(); + + let items = response + .data + .and_then(|data| data.items) + .and_then(|items| items.item) + .unwrap_or_default(); + + let mut charges = Vec::new(); + for item in items { + let billed = item.pretax_amount.unwrap_or(0.0); + let list = item.pretax_gross_amount.unwrap_or(0.0); + // A product with nothing on either side of the discount was not + // used this month. + if billed == 0.0 && list == 0.0 { + continue; + } + + charges.push(Charge { + service_name: item.product_name, + // ProductCode is stable across locales; ProductName is not. + service_category: item.product_code, + billed_cost: Some(billed), + list_cost: Some(list), + ..Charge::new( + start, + end, + item.currency.unwrap_or_else(|| "CNY".to_string()), + ) + }); + } + + Ok(Normalized { + charges, + balances: Vec::new(), + }) +} + // ==================== Response Structs ==================== // Note: These fields are used for serde deserialization of Alibaba Cloud API responses. // Some fields may not be directly read in the code, but are needed for correct JSON parsing. @@ -478,3 +568,64 @@ struct InstanceBillItem { pretax_gross_amount: Option, currency: Option, } + +#[cfg(test)] +mod tests { + use super::*; + use crate::ledger::{ChargeCategory, CostBasis}; + + /// One recorded QueryBillOverview response. + const BILL_OVERVIEW: &str = include_str!("testdata/aliyun_bill_overview.json"); + + fn recorded_batch(body: &str) -> RawBatch { + RawBatch { + provider: "Aliyun".to_string(), + account_id: "acct-2".to_string(), + period: BillingPeriod::new(2026, 8), + batch_id: "b-1".to_string(), + fetched_at: "2026-09-01T02:00:00Z".parse().unwrap(), + parts: vec![RawPart::new(PART_BILL_OVERVIEW, "", body)], + } + } + + #[test] + fn a_recorded_overview_normalizes_to_one_charge_per_product() { + let normalized = normalize(&recorded_batch(BILL_OVERVIEW)).unwrap(); + + // The unused CDN product is dropped. + assert_eq!(normalized.charges.len(), 2); + + let ecs = &normalized.charges[0]; + assert_eq!(ecs.service_name.as_deref(), Some("云服务器 ECS")); + assert_eq!(ecs.service_category.as_deref(), Some("ecs")); + assert_eq!(ecs.billed_cost, Some(288.45)); + assert_eq!(ecs.list_cost, Some(320.5)); + assert_eq!(ecs.billing_currency, "CNY"); + assert_eq!(ecs.charge_category, ChargeCategory::Usage); + assert_eq!(ecs.cost_basis, CostBasis::Authoritative); + } + + #[test] + fn a_monthly_overview_row_covers_the_whole_period() { + let normalized = normalize(&recorded_batch(BILL_OVERVIEW)).unwrap(); + let charge = &normalized.charges[0]; + + assert_eq!( + charge.charge_period_start.to_rfc3339(), + "2026-08-01T00:00:00+00:00" + ); + assert_eq!( + charge.charge_period_end.to_rfc3339(), + "2026-09-01T00:00:00+00:00" + ); + } + + #[test] + fn an_empty_bill_normalizes_to_nothing() { + let normalized = normalize(&recorded_batch( + r#"{"Code":"Success","Data":{"BillingCycle":"2026-08"}}"#, + )) + .unwrap(); + assert!(normalized.charges.is_empty()); + } +} diff --git a/src/cloud/aws.rs b/src/cloud/aws.rs index 59cff4f..653f00e 100644 --- a/src/cloud/aws.rs +++ b/src/cloud/aws.rs @@ -6,7 +6,9 @@ use hmac::{Hmac, Mac}; use serde::Deserialize; use sha2::{Digest, Sha256}; -use super::{CloudService, CostData, CostSummary, SourceId}; +use super::raw::RawPart; +use super::{BillingPeriod, BillingSource, CostData, CostSummary, Normalized, RawBatch, SourceId}; +use crate::ledger::Charge; type HmacSha256 = Hmac; @@ -155,35 +157,24 @@ impl AwsCloudService { parse_sts_response(&body) } - /// Call Cost Explorer API - /// Note: Cost Explorer API is only available in us-east-1 region - fn call_cost_explorer(&self, start_date: &str, end_date: &str) -> Result> { + /// Ask Cost Explorer for one time range and return the response body + /// unchanged. + /// + /// The only place in this file that talks to Cost Explorer. Each call + /// is billed, so callers ask for everything they need in one request. + /// + /// Note: the Cost Explorer endpoint only exists in us-east-1. + fn cost_and_usage_raw(&self, request: &serde_json::Value) -> Result { let timestamp = Utc::now(); let service = "ce"; - // Cost Explorer API is only available in us-east-1 let ce_region = "us-east-1"; let host = format!("ce.{}.amazonaws.com", ce_region); let uri = "/"; let amz_date = timestamp.format("%Y%m%dT%H%M%SZ").to_string(); - - // Build request body - let request_body = serde_json::json!({ - "TimePeriod": { - "Start": start_date, - "End": end_date - }, - "Granularity": "DAILY", - "Metrics": ["UnblendedCost"], - "GroupBy": [{ - "Type": "DIMENSION", - "Key": "SERVICE" - }] - }); - let payload = serde_json::to_string(&request_body)?; + let payload = serde_json::to_string(request)?; let payload_hash = Self::sha256_hash(payload.as_bytes()); - // Add required headers let headers = vec![ ( "content-type".to_string(), @@ -195,14 +186,14 @@ impl AwsCloudService { ), ]; - // Sign with us-east-1 region let authorization = self.sign_request_with_region( "POST", service, ce_region, &host, uri, "", &headers, &payload, timestamp, )?; let url = format!("https://{}{}", host, uri); - // Use Agent and disable status code as error, so we can read 4xx/5xx response body + // Do not treat a 4xx/5xx as a transport error, so the response body + // makes it into the log — Cost Explorer explains itself there. let agent = ureq::Agent::config_builder() .http_status_as_error(false) .timeout_global(Some(std::time::Duration::from_secs(30))) @@ -211,7 +202,7 @@ impl AwsCloudService { tracing::debug!("Sending Cost Explorer request: {}", url); - let result = agent + let response = agent .post(&url) .header("Authorization", &authorization) .header("X-Amz-Date", &amz_date) @@ -219,121 +210,40 @@ impl AwsCloudService { .header("Host", &host) .header("Content-Type", "application/x-amz-json-1.1") .header("X-Amz-Target", "AWSInsightsIndexService.GetCostAndUsage") - .send(&payload); - - match result { - Ok(response) => { - let status = response.status().as_u16(); - let body = response - .into_body() - .read_to_string() - .map_err(|e| anyhow!("Failed to read response: {}", e))?; - - if status >= 400 { - tracing::error!("Cost Explorer error response (HTTP {}): {}", status, body); - return Err(anyhow!( - "Cost Explorer request failed: HTTP {} - {}", - status, - body - )); - } + .send(&payload) + .map_err(|e| { + tracing::error!("Cost Explorer request error details: {:?}", e); + anyhow!("Cost Explorer request failed: {}", e) + })?; - parse_cost_explorer_response(&body, &self.account_id, &self.account_name) - } - Err(e) => { - // Network or other errors - let error_msg = format!("{:?}", e); - tracing::error!("Cost Explorer request error details: {}", error_msg); - Err(anyhow!("Cost Explorer request failed: {}", e)) - } + let status = response.status().as_u16(); + let body = response + .into_body() + .read_to_string() + .map_err(|e| anyhow!("Failed to read response: {}", e))?; + + if status >= 400 { + tracing::error!("Cost Explorer error response (HTTP {}): {}", status, body); + return Err(anyhow!( + "Cost Explorer request failed: HTTP {} - {}", + status, + body + )); } + + Ok(body) + } + + /// Call Cost Explorer API + fn call_cost_explorer(&self, start_date: &str, end_date: &str) -> Result> { + let body = self.cost_and_usage_raw(&cost_and_usage_request(start_date, end_date, true))?; + parse_cost_explorer_response(&body, &self.account_id, &self.account_name) } /// Call Cost Explorer API to get daily costs (not grouped by service, for trend charts) fn call_cost_explorer_daily(&self, start_date: &str, end_date: &str) -> Result> { - let timestamp = Utc::now(); - let service = "ce"; - let ce_region = "us-east-1"; - let host = format!("ce.{}.amazonaws.com", ce_region); - let uri = "/"; - - let amz_date = timestamp.format("%Y%m%dT%H%M%SZ").to_string(); - - // Build request body - not grouped by service, get daily total cost directly - let request_body = serde_json::json!({ - "TimePeriod": { - "Start": start_date, - "End": end_date - }, - "Granularity": "DAILY", - "Metrics": ["UnblendedCost"] - }); - let payload = serde_json::to_string(&request_body)?; - let payload_hash = Self::sha256_hash(payload.as_bytes()); - - let headers = vec![ - ( - "content-type".to_string(), - "application/x-amz-json-1.1".to_string(), - ), - ( - "x-amz-target".to_string(), - "AWSInsightsIndexService.GetCostAndUsage".to_string(), - ), - ]; - - let authorization = self.sign_request_with_region( - "POST", service, ce_region, &host, uri, "", &headers, &payload, timestamp, - )?; - - let url = format!("https://{}{}", host, uri); - - let agent = ureq::Agent::config_builder() - .http_status_as_error(false) - .timeout_global(Some(std::time::Duration::from_secs(30))) - .build() - .new_agent(); - - tracing::debug!("Sending Cost Explorer daily cost request: {}", url); - - let result = agent - .post(&url) - .header("Authorization", &authorization) - .header("X-Amz-Date", &amz_date) - .header("X-Amz-Content-Sha256", &payload_hash) - .header("Host", &host) - .header("Content-Type", "application/x-amz-json-1.1") - .header("X-Amz-Target", "AWSInsightsIndexService.GetCostAndUsage") - .send(&payload); - - match result { - Ok(response) => { - let status = response.status().as_u16(); - let body = response - .into_body() - .read_to_string() - .map_err(|e| anyhow!("Failed to read response: {}", e))?; - - if status >= 400 { - tracing::error!( - "Cost Explorer daily cost request error (HTTP {}): {}", - status, - body - ); - return Err(anyhow!( - "Cost Explorer request failed: HTTP {} - {}", - status, - body - )); - } - - parse_daily_cost_response(&body, &self.account_id) - } - Err(e) => { - tracing::error!("Cost Explorer daily cost request error: {:?}", e); - Err(anyhow!("Cost Explorer request failed: {}", e)) - } - } + let body = self.cost_and_usage_raw(&cost_and_usage_request(start_date, end_date, false))?; + parse_daily_cost_response(&body, &self.account_id) } /// Sign with specified region (for services like Cost Explorer that are only available in specific regions) @@ -410,6 +320,131 @@ impl AwsCloudService { } } +/// Name the Cost Explorer payload is stored under in a raw batch. +const PART_COST_AND_USAGE: &str = "cost_and_usage"; + +/// The GetCostAndUsage request body. +/// +/// `group_by_service` off is the trend query: one total per day, which is +/// a cheaper response than summing the grouped one client-side. +fn cost_and_usage_request( + start_date: &str, + end_date: &str, + group_by_service: bool, +) -> serde_json::Value { + let mut request = serde_json::json!({ + "TimePeriod": { + "Start": start_date, + "End": end_date + }, + "Granularity": "DAILY", + "Metrics": ["UnblendedCost"] + }); + + if group_by_service { + request["GroupBy"] = serde_json::json!([{ + "Type": "DIMENSION", + "Key": "SERVICE" + }]); + } + + request +} + +/// Turn a fetched Cost Explorer payload into ledger rows. +/// +/// Pure — every input is in `batch`. Cost Explorer reports `UnblendedCost`, +/// which is what was actually charged, so `cost_basis` is `authoritative` +/// and `effective_cost` stays empty until PR4 also requests +/// `AmortizedCost`. Every row is `Usage` for the same reason: without +/// `RECORD_TYPE` in the grouping the payload cannot tell a credit from a +/// charge, and guessing would put refunds on the wrong side of the total. +pub fn normalize(batch: &RawBatch) -> Result { + #[derive(Deserialize)] + struct CeResponse { + #[serde(rename = "ResultsByTime")] + results_by_time: Option>, + } + + #[derive(Deserialize)] + struct TimeResult { + #[serde(rename = "TimePeriod")] + time_period: TimePeriod, + #[serde(rename = "Groups")] + groups: Option>, + } + + #[derive(Deserialize)] + struct TimePeriod { + #[serde(rename = "Start")] + start: String, + #[serde(rename = "End")] + end: String, + } + + #[derive(Deserialize)] + struct CostGroup { + #[serde(rename = "Keys")] + keys: Vec, + #[serde(rename = "Metrics")] + metrics: CostMetrics, + } + + #[derive(Deserialize)] + struct CostMetrics { + #[serde(rename = "UnblendedCost")] + unblended_cost: CostAmount, + } + + #[derive(Deserialize)] + struct CostAmount { + #[serde(rename = "Amount")] + amount: String, + #[serde(rename = "Unit")] + unit: String, + } + + let part = batch + .part(PART_COST_AND_USAGE) + .ok_or_else(|| anyhow!("Raw batch has no '{}' payload", PART_COST_AND_USAGE))?; + let response: CeResponse = serde_json::from_str(&part.body) + .map_err(|e| anyhow!("Failed to parse Cost Explorer payload: {}", e))?; + + let mut charges = Vec::new(); + for result in response.results_by_time.unwrap_or_default() { + let start = parse_day(&result.time_period.start)?; + let end = parse_day(&result.time_period.end)?; + + for group in result.groups.unwrap_or_default() { + let amount: f64 = group.metrics.unblended_cost.amount.parse().unwrap_or(0.0); + // Cost Explorer returns a row for every service in the account, + // most of them zero. They carry no information and would bloat + // the fact table by an order of magnitude. + if amount == 0.0 { + continue; + } + + charges.push(Charge { + service_name: group.keys.first().cloned(), + billed_cost: Some(amount), + ..Charge::new(start, end, group.metrics.unblended_cost.unit) + }); + } + } + + Ok(Normalized { + charges, + balances: Vec::new(), + }) +} + +/// Parse a Cost Explorer `YYYY-MM-DD` into an instant at UTC midnight. +fn parse_day(date: &str) -> Result> { + let day = chrono::NaiveDate::parse_from_str(date, "%Y-%m-%d") + .map_err(|e| anyhow!("Unexpected Cost Explorer date {:?}: {}", date, e))?; + Ok(day.and_hms_opt(0, 0, 0).expect("midnight exists").and_utc()) +} + /// STS Caller Identity #[derive(Debug)] struct StsCallerIdentity { @@ -590,7 +625,7 @@ fn parse_daily_cost_response(json: &str, account_id: &str) -> Result Result { match self.call_sts_get_caller_identity() { Ok(identity) => { @@ -608,6 +643,25 @@ impl CloudService for AwsCloudService { } } + fn fetch(&self, period: &BillingPeriod) -> Result> { + let request = cost_and_usage_request( + &period.start().to_string(), + &period.end_exclusive().to_string(), + true, + ); + let body = self.cost_and_usage_raw(&request)?; + + Ok(vec![RawPart::new( + PART_COST_AND_USAGE, + serde_json::to_string(&request)?, + body, + )]) + } + + fn normalize(&self, batch: &RawBatch) -> Result { + normalize(batch) + } + fn get_cost_data(&self, start_date: &str, end_date: &str) -> Result> { self.call_cost_explorer(start_date, end_date) } @@ -748,6 +802,21 @@ fn aggregate_daily_costs(costs: &[CostData]) -> (Vec, String) #[cfg(test)] mod tests { use super::*; + use crate::ledger::{ChargeCategory, CostBasis}; + + /// One recorded GetCostAndUsage response, DAILY and grouped by SERVICE. + const COST_AND_USAGE: &str = include_str!("testdata/aws_cost_and_usage.json"); + + fn recorded_batch(body: &str) -> RawBatch { + RawBatch { + provider: "AWS".to_string(), + account_id: "acct-1".to_string(), + period: BillingPeriod::new(2026, 8), + batch_id: "b-1".to_string(), + fetched_at: "2026-08-03T04:00:00Z".parse().unwrap(), + parts: vec![RawPart::new(PART_COST_AND_USAGE, "{}", body)], + } + } #[test] fn test_sha256_hash() { @@ -755,4 +824,79 @@ mod tests { assert!(!hash.is_empty()); assert_eq!(hash.len(), 64); // SHA256 produces 32 bytes = 64 hex characters } + + #[test] + fn a_recorded_response_normalizes_to_one_charge_per_service_day() { + let normalized = normalize(&recorded_batch(COST_AND_USAGE)).unwrap(); + + // Three non-zero groups across two days; the zero-cost KMS row is + // dropped. + assert_eq!(normalized.charges.len(), 3); + assert!(normalized.balances.is_empty()); + + let first = &normalized.charges[0]; + assert_eq!( + first.service_name.as_deref(), + Some("Amazon Elastic Compute Cloud - Compute") + ); + assert_eq!(first.billed_cost, Some(12.45)); + assert_eq!(first.billing_currency, "USD"); + assert_eq!(first.charge_category, ChargeCategory::Usage); + assert_eq!(first.cost_basis, CostBasis::Authoritative); + assert_eq!( + first.charge_period_start.to_rfc3339(), + "2026-08-01T00:00:00+00:00" + ); + assert_eq!( + first.charge_period_end.to_rfc3339(), + "2026-08-02T00:00:00+00:00" + ); + + // Amortization needs AmortizedCost, which this request does not ask + // for: the column stays empty rather than being filled with the + // unblended figure. + assert_eq!(first.effective_cost, None); + + let total: f64 = normalized + .charges + .iter() + .filter_map(|charge| charge.billed_cost) + .sum(); + assert_eq!(total, 25.1); + } + + #[test] + fn normalizing_is_not_affected_by_when_it_runs() { + let batch = recorded_batch(COST_AND_USAGE); + let mut later = batch.clone(); + later.fetched_at = "2027-01-01T00:00:00Z".parse().unwrap(); + later.batch_id = "b-2".to_string(); + + let first = normalize(&batch).unwrap(); + let second = normalize(&later).unwrap(); + + assert_eq!(first.charges.len(), second.charges.len()); + for (a, b) in first.charges.iter().zip(&second.charges) { + assert_eq!(a.billed_cost, b.billed_cost); + assert_eq!(a.charge_period_start, b.charge_period_start); + assert_eq!(a.service_name, b.service_name); + } + } + + #[test] + fn a_batch_without_the_expected_payload_is_an_error() { + let mut batch = recorded_batch(COST_AND_USAGE); + batch.parts.clear(); + assert!(normalize(&batch).is_err()); + } + + #[test] + fn the_trend_request_asks_for_totals_and_the_summary_request_for_services() { + let grouped = cost_and_usage_request("2026-08-01", "2026-09-01", true); + assert_eq!(grouped["GroupBy"][0]["Key"], "SERVICE"); + assert_eq!(grouped["Granularity"], "DAILY"); + + let totals = cost_and_usage_request("2026-08-01", "2026-09-01", false); + assert!(totals.get("GroupBy").is_none()); + } } diff --git a/src/cloud/deepseek.rs b/src/cloud/deepseek.rs index ce1965e..c942ffb 100644 --- a/src/cloud/deepseek.rs +++ b/src/cloud/deepseek.rs @@ -3,7 +3,12 @@ use anyhow::{anyhow, Result}; use serde::Deserialize; -use super::{CloudService, CostData, CostSummary, CostTrend, ServiceCost, SourceId}; +use super::raw::RawPart; +use super::{ + BillingPeriod, BillingSource, CostData, CostSummary, CostTrend, Normalized, RawBatch, + ServiceCost, SourceId, +}; +use crate::ledger::BalanceSnapshot; /// DeepSeek balance info #[derive(Debug, Deserialize)] @@ -28,6 +33,11 @@ pub struct BalanceResponse { pub balance_infos: Vec, } +const BALANCE_URL: &str = "https://api.deepseek.com/user/balance"; + +/// Name the balance payload is stored under in a raw batch. +const PART_BALANCE: &str = "balance"; + /// DeepSeek service pub struct DeepSeekService { account_id: String, @@ -50,27 +60,65 @@ impl DeepSeekService { } } - /// Get user balance from DeepSeek API - pub fn get_balance(&self) -> Result { - let response = ureq::get("https://api.deepseek.com/user/balance") + /// Ask the balance endpoint and return the response body unchanged. + fn balance_raw(&self) -> Result { + let response = ureq::get(BALANCE_URL) .header("Accept", "application/json") .header("Authorization", &format!("Bearer {}", self.api_key)) .call() .map_err(|e| anyhow!("Failed to call DeepSeek API: {}", e))?; - let body = response + response .into_body() .read_to_string() - .map_err(|e| anyhow!("Failed to read response: {}", e))?; + .map_err(|e| anyhow!("Failed to read response: {}", e)) + } - let balance: BalanceResponse = serde_json::from_str(&body) - .map_err(|e| anyhow!("Failed to parse DeepSeek response: {}", e))?; + /// Get user balance from DeepSeek API + pub fn get_balance(&self) -> Result { + let body = self.balance_raw()?; - Ok(balance) + serde_json::from_str(&body).map_err(|e| anyhow!("Failed to parse DeepSeek response: {}", e)) } } -impl CloudService for DeepSeekService { +/// Turn a fetched balance payload into ledger rows. +/// +/// Pure — the observation time comes from the batch, not the clock. +/// +/// A balance is state, not a charge: it says what is left, not what was +/// spent, so it lands in `fct_balance_snapshot` and contributes nothing to +/// a cost total. Top-ups are the part that is a charge, and DeepSeek does +/// not report them here; PR5 derives them from the movement between two +/// snapshots. +pub fn normalize(batch: &RawBatch) -> Result { + let part = batch + .part(PART_BALANCE) + .ok_or_else(|| anyhow!("Raw batch has no '{}' payload", PART_BALANCE))?; + let response: BalanceResponse = serde_json::from_str(&part.body) + .map_err(|e| anyhow!("Failed to parse DeepSeek response: {}", e))?; + + let balances = response + .balance_infos + .into_iter() + .map(|info| BalanceSnapshot { + provider: batch.provider.clone(), + account_id: batch.account_id.clone(), + observed_at: batch.fetched_at, + balance: info.total_balance.parse().unwrap_or(0.0), + granted_balance: info.granted_balance.parse().ok(), + topped_up_balance: info.topped_up_balance.parse().ok(), + currency: info.currency, + }) + .collect(); + + Ok(Normalized { + charges: Vec::new(), + balances, + }) +} + +impl BillingSource for DeepSeekService { fn validate_credentials(&self) -> Result { match self.get_balance() { Ok(_) => Ok(true), @@ -81,6 +129,20 @@ impl CloudService for DeepSeekService { } } + fn fetch(&self, _period: &BillingPeriod) -> Result> { + // A balance is the same value whichever period is being ingested: + // the endpoint reports the account as it stands right now. + Ok(vec![RawPart::new( + PART_BALANCE, + format!("GET {}", BALANCE_URL), + self.balance_raw()?, + )]) + } + + fn normalize(&self, batch: &RawBatch) -> Result { + normalize(batch) + } + fn get_cost_data(&self, _start_date: &str, _end_date: &str) -> Result> { // DeepSeek doesn't provide detailed cost history, return empty Ok(vec![]) @@ -145,3 +207,58 @@ impl CloudService for DeepSeekService { }) } } + +#[cfg(test)] +mod tests { + use super::*; + + /// One recorded /user/balance response. + const BALANCE: &str = include_str!("testdata/deepseek_balance.json"); + + fn recorded_batch(body: &str) -> RawBatch { + RawBatch { + provider: "DeepSeek".to_string(), + account_id: "acct-3".to_string(), + period: BillingPeriod::new(2026, 8), + batch_id: "b-1".to_string(), + fetched_at: "2026-08-29T09:30:00Z".parse().unwrap(), + parts: vec![RawPart::new(PART_BALANCE, "", body)], + } + } + + #[test] + fn a_balance_becomes_a_snapshot_and_never_a_charge() { + let normalized = normalize(&recorded_batch(BALANCE)).unwrap(); + + assert!(normalized.charges.is_empty()); + assert_eq!(normalized.balances.len(), 1); + + let snapshot = &normalized.balances[0]; + assert_eq!(snapshot.balance, 42.75); + assert_eq!(snapshot.granted_balance, Some(10.0)); + assert_eq!(snapshot.topped_up_balance, Some(32.75)); + assert_eq!(snapshot.currency, "CNY"); + assert_eq!( + snapshot.observed_at.to_rfc3339(), + "2026-08-29T09:30:00+00:00" + ); + assert_eq!(snapshot.account_id, "acct-3"); + } + + #[test] + fn an_account_holding_two_currencies_gets_a_snapshot_each() { + let normalized = normalize(&recorded_batch( + r#"{"is_available":true,"balance_infos":[ + {"currency":"CNY","total_balance":"42.75","granted_balance":"10.00","topped_up_balance":"32.75"}, + {"currency":"USD","total_balance":"6.00","granted_balance":"0.00","topped_up_balance":"6.00"}]}"#, + )) + .unwrap(); + + let currencies: Vec<&str> = normalized + .balances + .iter() + .map(|b| b.currency.as_str()) + .collect(); + assert_eq!(currencies, vec!["CNY", "USD"]); + } +} diff --git a/src/cloud/mod.rs b/src/cloud/mod.rs index a31fa38..f2a0994 100644 --- a/src/cloud/mod.rs +++ b/src/cloud/mod.rs @@ -3,12 +3,15 @@ pub mod aliyun; pub mod aws; pub mod deepseek; +pub mod raw; pub mod registry; use anyhow::Result; -use chrono::{DateTime, Utc}; +use chrono::{DateTime, Datelike, NaiveDate, Utc}; use serde::{Deserialize, Serialize}; +use crate::ledger::{BalanceSnapshot, Charge}; +pub use raw::{RawBatch, RawPart}; pub use registry::{SourceDescriptor, SourceId}; /// Shown in place of a source's name when its id is not in the registry. @@ -204,11 +207,75 @@ pub struct BudgetStatus { pub alert_triggered: bool, } -/// Cloud service provider trait (sync version, using ureq) -pub trait CloudService: Send + Sync { +/// A calendar month of billing, the unit providers issue a bill in and the +/// unit the ledger replaces as a whole. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct BillingPeriod { + pub year: i32, + pub month: u32, +} + +impl BillingPeriod { + pub fn new(year: i32, month: u32) -> Self { + Self { year, month } + } + + /// The period the given instant falls in. + pub fn containing(instant: DateTime) -> Self { + Self::new(instant.year(), instant.month()) + } + + /// `YYYY-MM`, as stored in `billing_period` and in the raw path. + pub fn label(&self) -> String { + format!("{:04}-{:02}", self.year, self.month) + } + + /// First day of the period. + pub fn start(&self) -> NaiveDate { + NaiveDate::from_ymd_opt(self.year, self.month, 1).expect("a valid billing period") + } + + /// First day of the following period. Cost Explorer and the BSS API + /// both take an exclusive end. + pub fn end_exclusive(&self) -> NaiveDate { + let (year, month) = if self.month == 12 { + (self.year + 1, 1) + } else { + (self.year, self.month + 1) + }; + NaiveDate::from_ymd_opt(year, month, 1).expect("a valid billing period") + } +} + +/// What a normalizer produces: FOCUS rows ready for the ledger. +/// +/// Charges and balances are separate because a balance is state, not a +/// charge — see `fct_balance_snapshot`. +#[derive(Debug, Default)] +pub struct Normalized { + pub charges: Vec, + pub balances: Vec, +} + +/// A source of billing data (sync, using ureq). +/// +/// [`Self::fetch`] and [`Self::normalize`] are deliberately split. `fetch` +/// touches the network and interprets nothing; `normalize` interprets and +/// touches nothing. That is what makes the billing logic testable from a +/// recorded payload, and what keeps a mapping fix from costing another +/// round of paid API calls. +pub trait BillingSource: Send + Sync { /// Validate credentials fn validate_credentials(&self) -> Result; + /// Retrieve everything the provider reports for one billing period, + /// unchanged. The only method here that talks to the network. + fn fetch(&self, period: &BillingPeriod) -> Result>; + + /// Turn a fetched batch into ledger rows. Pure: no clock, no network, + /// no database — everything it needs is in the batch. + fn normalize(&self, batch: &RawBatch) -> Result; + /// Get cost data fn get_cost_data(&self, start_date: &str, end_date: &str) -> Result>; diff --git a/src/cloud/raw.rs b/src/cloud/raw.rs new file mode 100644 index 0000000..1b464af --- /dev/null +++ b/src/cloud/raw.rs @@ -0,0 +1,366 @@ +//! Raw payload store. +//! +//! `fetch` writes what a provider actually returned, byte for byte, before +//! anything interprets it. `normalize` then reads from here rather than +//! from the network. Cost Explorer bills per request, so this is what makes +//! a schema change cheap: re-normalizing a corrected mapping over payloads +//! already on disk costs nothing. +//! +//! The layout is Hive-partitioned: +//! +//! ```text +//! raw/provider=

/account=/billing_period=/batch=/part-0.parquet +//! ``` +//! +//! The same path semantics work for a local directory and for an object +//! store, so P1's S3/OSS export channel replaces the `fetch` implementation +//! and leaves everything downstream alone. + +use anyhow::{anyhow, Result}; +use chrono::{DateTime, Utc}; +use duckdb::{params, Connection}; +use std::path::{Path, PathBuf}; + +use super::BillingPeriod; +use crate::ledger::schema::TIMESTAMP_FORMAT; + +/// One response, exactly as the provider sent it. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RawPart { + /// Logical name of the call, unique within a batch — this is what + /// `normalize` looks the payload up by. + pub name: String, + /// The request that produced it, for reproducing the call later. + pub request: String, + /// The response body, unchanged. Never parsed on the way in. + pub body: String, +} + +impl RawPart { + pub fn new( + name: impl Into, + request: impl Into, + body: impl Into, + ) -> Self { + Self { + name: name.into(), + request: request.into(), + body: body.into(), + } + } +} + +/// Everything one fetch of one account's billing period returned. +/// +/// This is the whole input to [`super::BillingSource::normalize`]: keeping +/// `fetched_at` on the batch rather than reading the clock inside a +/// normalizer is what lets a normalizer stay a pure function. +#[derive(Debug, Clone)] +pub struct RawBatch { + pub provider: String, + pub account_id: String, + pub period: BillingPeriod, + pub batch_id: String, + pub fetched_at: DateTime, + pub parts: Vec, +} + +impl RawBatch { + /// The payload stored under `name`. + pub fn part(&self, name: &str) -> Option<&RawPart> { + self.parts.iter().find(|part| part.name == name) + } + + /// Directory this batch is stored under, below `root`. + pub fn directory(&self, root: &Path) -> PathBuf { + batch_directory( + root, + &self.provider, + &self.account_id, + &self.period, + &self.batch_id, + ) + } +} + +/// `/provider=

/account=/billing_period=/batch=` +pub fn batch_directory( + root: &Path, + provider: &str, + account_id: &str, + period: &BillingPeriod, + batch_id: &str, +) -> PathBuf { + root.join(format!("provider={}", provider)) + .join(format!("account={}", account_id)) + .join(format!("billing_period={}", period.label())) + .join(format!("batch={}", batch_id)) +} + +/// Write a batch as a single Parquet part. Returns the file's path, which +/// is what the ledger records as the batch's `source_ref`. +pub fn write(root: &Path, batch: &RawBatch) -> Result { + let directory = batch.directory(root); + std::fs::create_dir_all(&directory)?; + let file = directory.join("part-0.parquet"); + + let conn = Connection::open_in_memory()?; + conn.execute_batch( + "CREATE TABLE part ( + part_name VARCHAR NOT NULL, + request VARCHAR NOT NULL, + body VARCHAR NOT NULL, + fetched_at TIMESTAMP NOT NULL + )", + )?; + + let fetched_at = batch.fetched_at.format(TIMESTAMP_FORMAT).to_string(); + for part in &batch.parts { + conn.execute( + "INSERT INTO part VALUES (?, ?, ?, CAST(? AS TIMESTAMP))", + params![part.name, part.request, part.body, fetched_at], + )?; + } + + conn.execute_batch(&format!( + "COPY part TO '{}' (FORMAT PARQUET)", + sql_literal(&file.to_string_lossy()) + ))?; + + tracing::info!("Raw: wrote {} payload(s) to {:?}", batch.parts.len(), file); + Ok(file) +} + +/// Read the payloads of a batch back, in the order they were written, +/// with the instant they were fetched at. +pub fn read(file: &Path) -> Result<(Vec, DateTime)> { + let conn = Connection::open_in_memory()?; + let mut stmt = conn.prepare(&format!( + "SELECT part_name, request, body, CAST(fetched_at AS VARCHAR) FROM read_parquet('{}')", + sql_literal(&file.to_string_lossy()) + ))?; + + let rows = stmt + .query_map([], |row| { + Ok(( + RawPart { + name: row.get(0)?, + request: row.get(1)?, + body: row.get(2)?, + }, + row.get::<_, String>(3)?, + )) + })? + .collect::, _>>()?; + + let fetched_at = rows + .first() + .map(|(_, stamp)| parse_timestamp(stamp)) + .transpose()? + .unwrap_or_else(Utc::now); + + Ok((rows.into_iter().map(|(part, _)| part).collect(), fetched_at)) +} + +/// Read a stored batch back in full, so it can be normalized again without +/// paying for another fetch. +pub fn read_batch( + root: &Path, + provider: &str, + account_id: &str, + period: &BillingPeriod, + batch_id: &str, +) -> Result { + let file = batch_directory(root, provider, account_id, period, batch_id).join("part-0.parquet"); + let (parts, fetched_at) = read(&file)?; + + Ok(RawBatch { + provider: provider.to_string(), + account_id: account_id.to_string(), + period: *period, + batch_id: batch_id.to_string(), + fetched_at, + parts, + }) +} + +/// Parse a timestamp as DuckDB renders it, `YYYY-MM-DD HH:MM:SS` in UTC. +fn parse_timestamp(value: &str) -> Result> { + let stamp = value.split('.').next().unwrap_or(value); + Ok( + chrono::NaiveDateTime::parse_from_str(stamp, TIMESTAMP_FORMAT) + .map_err(|e| anyhow!("Unexpected timestamp {:?} in a raw batch: {}", value, e))? + .and_utc(), + ) +} + +/// Batch ids stored for one account and period, oldest first. +/// +/// Re-normalizing reads the newest of these instead of paying for another +/// fetch. +pub fn batches( + root: &Path, + provider: &str, + account_id: &str, + period: &BillingPeriod, +) -> Result> { + let period_directory = root + .join(format!("provider={}", provider)) + .join(format!("account={}", account_id)) + .join(format!("billing_period={}", period.label())); + + if !period_directory.exists() { + return Ok(Vec::new()); + } + + let mut ids = Vec::new(); + for entry in std::fs::read_dir(&period_directory)? { + let name = entry?.file_name().to_string_lossy().into_owned(); + if let Some(id) = name.strip_prefix("batch=") { + ids.push(id.to_string()); + } + } + ids.sort(); + Ok(ids) +} + +/// Escape a value for interpolation into a SQL string literal. +/// +/// Paths cannot be bound as parameters in `COPY ... TO` or `read_parquet`, +/// so they are interpolated; a path containing a quote must not end the +/// literal. +fn sql_literal(value: &str) -> String { + value.replace('\'', "''") +} + +/// Reject a value that would break out of its path segment. +/// +/// Provider ids come from the registry, but account ids and period labels +/// reach here from the database, so the layout is checked rather than +/// assumed. +pub fn check_path_segment(value: &str, what: &str) -> Result<()> { + if value.is_empty() || value.contains(['/', '\\', '\0']) || value == "." || value == ".." { + return Err(anyhow!( + "{} is not usable as a path segment: {:?}", + what, + value + )); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + struct TempDir(PathBuf); + + impl TempDir { + fn new() -> Self { + let path = + std::env::temp_dir().join(format!("cloudbridge-raw-{}", uuid::Uuid::new_v4())); + std::fs::create_dir_all(&path).unwrap(); + Self(path) + } + } + + impl Drop for TempDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } + } + + fn batch(batch_id: &str, parts: Vec) -> RawBatch { + RawBatch { + provider: "AWS".to_string(), + account_id: "acct-1".to_string(), + period: BillingPeriod::new(2026, 8), + batch_id: batch_id.to_string(), + fetched_at: Utc::now(), + parts, + } + } + + #[test] + fn the_partition_layout_is_the_one_a_bucket_would_use() { + let path = batch_directory( + Path::new("/data/raw"), + "AWS", + "acct-1", + &BillingPeriod::new(2026, 8), + "b-1", + ); + assert!(path.ends_with("provider=AWS/account=acct-1/billing_period=2026-08/batch=b-1")); + } + + #[test] + fn payloads_come_back_unchanged() { + let dir = TempDir::new(); + // A body with quotes and newlines: nothing here is escaped or + // reformatted on the way through Parquet. + let body = "{\"Results\": [{\"Amount\": \"1.5\"}],\n \"note\": \"it's fine\"}"; + let written = batch( + "b-1", + vec![ + RawPart::new("cost_and_usage", "{\"Granularity\":\"DAILY\"}", body), + RawPart::new("second", "{}", ""), + ], + ); + + let file = write(&dir.0, &written).unwrap(); + assert!(file.ends_with("part-0.parquet")); + + let (parts, _) = read(&file).unwrap(); + assert_eq!(parts, written.parts); + } + + #[test] + fn a_stored_batch_can_be_normalized_again_without_refetching() { + let dir = TempDir::new(); + let mut written = batch( + "b-1", + vec![RawPart::new("cost_and_usage", "{}", "{\"a\":1}")], + ); + written.fetched_at = "2026-08-29T09:30:00Z".parse().unwrap(); + + write(&dir.0, &written).unwrap(); + let reread = + read_batch(&dir.0, "AWS", "acct-1", &BillingPeriod::new(2026, 8), "b-1").unwrap(); + + assert_eq!(reread.parts, written.parts); + assert_eq!(reread.fetched_at, written.fetched_at); + assert_eq!(reread.batch_id, "b-1"); + } + + #[test] + fn batches_are_listed_oldest_first_per_period() { + let dir = TempDir::new(); + let period = BillingPeriod::new(2026, 8); + + for id in ["b-2", "b-1"] { + write(&dir.0, &batch(id, vec![RawPart::new("p", "", "{}")])).unwrap(); + } + // A different period must not show up in the listing. + let mut other = batch("b-3", vec![RawPart::new("p", "", "{}")]); + other.period = BillingPeriod::new(2026, 7); + write(&dir.0, &other).unwrap(); + + let ids = batches(&dir.0, "AWS", "acct-1", &period).unwrap(); + assert_eq!(ids, vec!["b-1".to_string(), "b-2".to_string()]); + } + + #[test] + fn an_unfetched_period_lists_nothing() { + let dir = TempDir::new(); + let ids = batches(&dir.0, "AWS", "acct-1", &BillingPeriod::new(2026, 8)).unwrap(); + assert!(ids.is_empty()); + } + + #[test] + fn a_path_segment_cannot_escape_its_directory() { + assert!(check_path_segment("acct-1", "account id").is_ok()); + assert!(check_path_segment("..", "account id").is_err()); + assert!(check_path_segment("a/b", "account id").is_err()); + assert!(check_path_segment("", "account id").is_err()); + } +} diff --git a/src/cloud/registry.rs b/src/cloud/registry.rs index e6c7449..7760bbb 100644 --- a/src/cloud/registry.rs +++ b/src/cloud/registry.rs @@ -13,7 +13,7 @@ use serde::{Deserialize, Serialize}; use super::{aliyun::AliyunCloudService, aws::AwsCloudService, deepseek::DeepSeekService}; -use super::{CloudService, SourceContext}; +use super::{BillingSource, SourceContext}; /// What a source reports, and therefore how it can be displayed. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -77,7 +77,7 @@ pub struct SourceDescriptor { pub reporting: Reporting, /// Builds the client. A function pointer keeps construction in this /// table instead of a `match` in every caller. - pub build: fn(SourceContext) -> Box, + pub build: fn(SourceContext) -> Box, } impl SourceDescriptor { diff --git a/src/cloud/testdata/aliyun_bill_overview.json b/src/cloud/testdata/aliyun_bill_overview.json new file mode 100644 index 0000000..428d04d --- /dev/null +++ b/src/cloud/testdata/aliyun_bill_overview.json @@ -0,0 +1,36 @@ +{ + "RequestId": "8A1E1E1B-0000-4E4E-9C9C-3F3F3F3F3F3F", + "Success": true, + "Code": "Success", + "Message": "Successful!", + "Data": { + "BillingCycle": "2026-08", + "AccountID": "1234567890123456", + "AccountName": "example@aliyun.com", + "Items": { + "Item": [ + { + "ProductCode": "ecs", + "ProductName": "云服务器 ECS", + "PretaxGrossAmount": 320.5, + "PretaxAmount": 288.45, + "Currency": "CNY" + }, + { + "ProductCode": "oss", + "ProductName": "对象存储 OSS", + "PretaxGrossAmount": 42.0, + "PretaxAmount": 42.0, + "Currency": "CNY" + }, + { + "ProductCode": "cdn", + "ProductName": "CDN", + "PretaxGrossAmount": 0, + "PretaxAmount": 0, + "Currency": "CNY" + } + ] + } + } +} diff --git a/src/cloud/testdata/aws_cost_and_usage.json b/src/cloud/testdata/aws_cost_and_usage.json new file mode 100644 index 0000000..3086e99 --- /dev/null +++ b/src/cloud/testdata/aws_cost_and_usage.json @@ -0,0 +1,66 @@ +{ + "GroupDefinitions": [ + { + "Type": "DIMENSION", + "Key": "SERVICE" + } + ], + "ResultsByTime": [ + { + "TimePeriod": { + "Start": "2026-08-01", + "End": "2026-08-02" + }, + "Total": {}, + "Groups": [ + { + "Keys": ["Amazon Elastic Compute Cloud - Compute"], + "Metrics": { + "UnblendedCost": { + "Amount": "12.4500000000", + "Unit": "USD" + } + } + }, + { + "Keys": ["Amazon Simple Storage Service"], + "Metrics": { + "UnblendedCost": { + "Amount": "0.7500000000", + "Unit": "USD" + } + } + }, + { + "Keys": ["AWS Key Management Service"], + "Metrics": { + "UnblendedCost": { + "Amount": "0", + "Unit": "USD" + } + } + } + ], + "Estimated": false + }, + { + "TimePeriod": { + "Start": "2026-08-02", + "End": "2026-08-03" + }, + "Total": {}, + "Groups": [ + { + "Keys": ["Amazon Elastic Compute Cloud - Compute"], + "Metrics": { + "UnblendedCost": { + "Amount": "11.9000000000", + "Unit": "USD" + } + } + } + ], + "Estimated": true + } + ] +} diff --git a/src/cloud/testdata/deepseek_balance.json b/src/cloud/testdata/deepseek_balance.json new file mode 100644 index 0000000..7565df3 --- /dev/null +++ b/src/cloud/testdata/deepseek_balance.json @@ -0,0 +1,11 @@ +{ + "is_available": true, + "balance_infos": [ + { + "currency": "CNY", + "total_balance": "42.75", + "granted_balance": "10.00", + "topped_up_balance": "32.75" + } + ] +} diff --git a/src/config.rs b/src/config.rs index 3289b5d..bc1a00a 100644 --- a/src/config.rs +++ b/src/config.rs @@ -77,6 +77,13 @@ pub fn get_ledger_database_path() -> Result { Ok(data_dir.join("billing.duckdb")) } +/// Root of the raw payload store. Laid out so the same path semantics +/// work for a local directory and for an object store; see [`crate::cloud::raw`]. +pub fn get_raw_data_dir() -> Result { + let data_dir = get_app_data_dir()?; + Ok(data_dir.join("raw")) +} + /// Load configuration pub fn load_config() -> Result { let config_path = get_config_path()?; diff --git a/src/ingest.rs b/src/ingest.rs new file mode 100644 index 0000000..ee8f03f --- /dev/null +++ b/src/ingest.rs @@ -0,0 +1,141 @@ +//! Ingest: fetch → persist raw → normalize → ledger. +//! +//! The order matters. Raw payloads are written *before* anything is +//! normalized, so a mapping bug costs a re-run of [`renormalize_period`] +//! rather than another round of paid API calls. Cost Explorer bills per +//! request; the payloads on disk do not. +//! +//! One run of [`ingest_period`] is one `ingest_batch` row, one raw +//! partition, and one whole-period replacement in `fct_charge` — all under +//! the same batch id. + +// The dashboard still reads through the response caches; it starts calling +// this in PR6, when the ledger becomes the read path. +#![allow(dead_code)] + +use anyhow::{anyhow, Result}; +use chrono::Utc; +use std::path::{Path, PathBuf}; + +use crate::cloud::raw::{self, RawBatch}; +use crate::cloud::{BillingPeriod, CloudAccount, Normalized}; +use crate::config::get_raw_data_dir; +use crate::ledger::{self, PeriodKey}; + +/// What one ingest did, for logging and for the UI to report. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct IngestOutcome { + pub batch_id: String, + pub charges: usize, + pub balances: usize, + /// Where the raw payloads were written. + pub raw_path: PathBuf, +} + +/// Ingest the period that is currently accruing — the UI's "refresh now". +pub fn ingest_current_period(account: &CloudAccount) -> Result { + ingest_period(account, &BillingPeriod::containing(Utc::now())) +} + +/// Fetch one account's billing period and land it in the ledger. +pub fn ingest_period(account: &CloudAccount, period: &BillingPeriod) -> Result { + let descriptor = account.descriptor().ok_or_else(|| { + anyhow!( + "No billing source registered under '{}'", + account.source_id.as_str() + ) + })?; + + let source = (descriptor.build)(account.context(descriptor)); + let parts = source.fetch(period)?; + + let batch = RawBatch { + provider: descriptor.id.to_string(), + account_id: account.id.clone(), + period: *period, + batch_id: ledger::new_batch_id(), + fetched_at: Utc::now(), + parts, + }; + + let raw_path = persist(&batch)?; + let normalized = source.normalize(&batch)?; + record(&batch, &normalized, &raw_path) +} + +/// Normalize a period again from payloads already on disk. +/// +/// This is the whole point of storing raw: correcting a mapping, or adding +/// a column, replays the newest batch at no cost. It fails rather than +/// silently fetching if nothing has been stored for the period yet. +pub fn renormalize_period(account: &CloudAccount, period: &BillingPeriod) -> Result { + let descriptor = account.descriptor().ok_or_else(|| { + anyhow!( + "No billing source registered under '{}'", + account.source_id.as_str() + ) + })?; + + let root = get_raw_data_dir()?; + let batch_id = raw::batches(&root, descriptor.id, &account.id, period)? + .pop() + .ok_or_else(|| { + anyhow!( + "No raw payload stored for {} {} — fetch it first", + account.name, + period.label() + ) + })?; + + let batch = raw::read_batch(&root, descriptor.id, &account.id, period, &batch_id)?; + let raw_path = batch.directory(&root).join("part-0.parquet"); + + let source = (descriptor.build)(account.context(descriptor)); + let normalized = source.normalize(&batch)?; + record(&batch, &normalized, &raw_path) +} + +/// Write the payloads under `raw/`, checking first that nothing in the path +/// came out of the database with a separator in it. +fn persist(batch: &RawBatch) -> Result { + raw::check_path_segment(&batch.provider, "source id")?; + raw::check_path_segment(&batch.account_id, "account id")?; + raw::check_path_segment(&batch.batch_id, "batch id")?; + + raw::write(&get_raw_data_dir()?, batch) +} + +/// Replace the period in the ledger with what the normalizer produced. +fn record(batch: &RawBatch, normalized: &Normalized, raw_path: &Path) -> Result { + let key = PeriodKey::new( + batch.provider.clone(), + batch.account_id.clone(), + batch.period.label(), + ); + + ledger::replace_period( + &key, + &batch.batch_id, + &normalized.charges, + Some(&raw_path.to_string_lossy()), + )?; + + for balance in &normalized.balances { + ledger::record_balance(balance)?; + } + + tracing::info!( + "Ingested {} {}: {} charge(s), {} balance(s)", + batch.provider, + batch.period.label(), + normalized.charges.len(), + normalized.balances.len() + ); + + Ok(IngestOutcome { + batch_id: batch.batch_id.clone(), + charges: normalized.charges.len(), + balances: normalized.balances.len(), + raw_path: raw_path.to_path_buf(), + }) +} diff --git a/src/ledger/mod.rs b/src/ledger/mod.rs index 28b40b0..f3e6c50 100644 --- a/src/ledger/mod.rs +++ b/src/ledger/mod.rs @@ -205,17 +205,25 @@ fn with_connection(f: impl FnOnce(&mut Connection) -> Result) -> Result f(conn) } +/// Identifier for one ingest. +/// +/// Minted by the caller rather than in here, because the raw payloads are +/// stored under the same id: `raw/.../batch=/` is what `source_ref` +/// points at, and the two have to agree. +pub fn new_batch_id() -> String { + uuid::Uuid::new_v4().to_string() +} + /// Replace everything stored for `key` with `charges`, in one transaction. /// -/// Returns the id of the batch the rows were written under. `source_ref` -/// points at the raw payload the rows were normalized from; it stays `None` -/// until PR3 persists Parquet. +/// `source_ref` points at the raw payload the rows were normalized from. pub fn replace_period( key: &PeriodKey, + batch_id: &str, charges: &[Charge], source_ref: Option<&str>, -) -> Result { - with_connection(|conn| write_period(conn, key, charges, source_ref)) +) -> Result<()> { + with_connection(|conn| write_period(conn, key, batch_id, charges, source_ref)) } /// Record a balance observation. Re-observing the same instant overwrites, @@ -227,10 +235,10 @@ pub fn record_balance(snapshot: &BalanceSnapshot) -> Result<()> { fn write_period( conn: &mut Connection, key: &PeriodKey, + batch_id: &str, charges: &[Charge], source_ref: Option<&str>, -) -> Result { - let batch_id = uuid::Uuid::new_v4().to_string(); +) -> Result<()> { let now = Utc::now().format(TIMESTAMP_FORMAT).to_string(); let ids = charge_ids(key, charges); @@ -323,7 +331,7 @@ fn write_period( key.billing_period, batch_id ); - Ok(batch_id) + Ok(()) } fn write_balance(conn: &mut Connection, snapshot: &BalanceSnapshot) -> Result<()> { @@ -457,10 +465,10 @@ mod tests { let key = key(); let charges = vec![usage("EC2", 12.5, 1), usage("S3", 0.75, 1)]; - write_period(&mut conn, &key, &charges, None).unwrap(); + write_period(&mut conn, &key, "b-1", &charges, None).unwrap(); let first = stored(&conn, &key); - write_period(&mut conn, &key, &charges, None).unwrap(); + write_period(&mut conn, &key, "b-2", &charges, None).unwrap(); let second = stored(&conn, &key); assert_eq!(first, second); @@ -475,6 +483,7 @@ mod tests { write_period( &mut conn, &key, + "b-1", &[ usage("EC2", 12.5, 1), usage("S3", 0.75, 1), @@ -489,6 +498,7 @@ mod tests { write_period( &mut conn, &key, + "b-2", &[usage("EC2", 11.0, 1), usage("S3", 0.75, 1)], None, ) @@ -507,8 +517,8 @@ mod tests { let mut conn = conn(); let key = key(); - write_period(&mut conn, &key, &[usage("EC2", 12.5, 1)], None).unwrap(); - let batch = write_period(&mut conn, &key, &[usage("EC2", 11.0, 1)], None).unwrap(); + write_period(&mut conn, &key, "b-1", &[usage("EC2", 12.5, 1)], None).unwrap(); + write_period(&mut conn, &key, "b-2", &[usage("EC2", 11.0, 1)], None).unwrap(); assert_eq!(scalar_i64(&conn, "SELECT count(*) FROM ingest_batch"), 2); assert_eq!( @@ -521,7 +531,7 @@ mod tests { assert_eq!( conn.query_row::("SELECT batch_id FROM fct_charge", [], |r| r.get(0)) .unwrap(), - batch + "b-2" ); } @@ -532,10 +542,17 @@ mod tests { let july = PeriodKey::new("AWS", "acct-1", "2026-07"); let other_account = PeriodKey::new("AWS", "acct-2", "2026-08"); - write_period(&mut conn, &july, &[usage("EC2", 9.0, 1)], None).unwrap(); - write_period(&mut conn, &other_account, &[usage("EC2", 5.0, 1)], None).unwrap(); - write_period(&mut conn, &august, &[usage("EC2", 12.5, 1)], None).unwrap(); - write_period(&mut conn, &august, &[], None).unwrap(); + write_period(&mut conn, &july, "b-1", &[usage("EC2", 9.0, 1)], None).unwrap(); + write_period( + &mut conn, + &other_account, + "b-2", + &[usage("EC2", 5.0, 1)], + None, + ) + .unwrap(); + write_period(&mut conn, &august, "b-3", &[usage("EC2", 12.5, 1)], None).unwrap(); + write_period(&mut conn, &august, "b-4", &[], None).unwrap(); assert!(stored(&conn, &august).is_empty()); assert_eq!(stored(&conn, &july).len(), 1); @@ -548,11 +565,11 @@ mod tests { let key = key(); let charges = vec![usage("EC2", 12.5, 1), usage("EC2", 4.0, 1)]; - write_period(&mut conn, &key, &charges, None).unwrap(); + write_period(&mut conn, &key, "b-1", &charges, None).unwrap(); let first = stored(&conn, &key); assert_eq!(first.len(), 2); - write_period(&mut conn, &key, &charges, None).unwrap(); + write_period(&mut conn, &key, "b-2", &charges, None).unwrap(); assert_eq!(stored(&conn, &key), first); } @@ -564,6 +581,7 @@ mod tests { write_period( &mut conn, &key, + "b-1", &[Charge { service_name: Some("Claude Code".to_string()), cost_basis: CostBasis::Absent, @@ -632,7 +650,7 @@ mod tests { let mut conn = conn(); let key = key(); - write_period(&mut conn, &key, &[usage("EC2", 1.0, 3)], None).unwrap(); + write_period(&mut conn, &key, "b-1", &[usage("EC2", 1.0, 3)], None).unwrap(); let start: String = conn .query_row( diff --git a/src/ledger/schema.rs b/src/ledger/schema.rs index 5fba665..272e0f2 100644 --- a/src/ledger/schema.rs +++ b/src/ledger/schema.rs @@ -98,7 +98,9 @@ pub fn apply(conn: &Connection) -> Result<()> { topped_up_balance DOUBLE, currency VARCHAR NOT NULL, created_at TIMESTAMP NOT NULL, - PRIMARY KEY (provider, account_id, observed_at) + -- Currency is part of the key: an account can hold a balance in + -- more than one, and they are observed at the same instant. + PRIMARY KEY (provider, account_id, observed_at, currency) ); -- Rates are dated because they get corrected, and the reporting diff --git a/src/main.rs b/src/main.rs index dce7aa9..cb7edc1 100644 --- a/src/main.rs +++ b/src/main.rs @@ -3,6 +3,7 @@ mod cloud; mod config; mod crypto; mod db; +mod ingest; mod ledger; mod secret_store; mod ui; From 50db24222aac96ad2489d4cc015e0107e872eef5 Mon Sep 17 00:00:00 2001 From: JetSquirrel Date: Sun, 30 Aug 2026 01:04:20 +0800 Subject: [PATCH 3/5] Map AWS charges to FOCUS categories MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cost Explorer was asked for UnblendedCost grouped by service, which is one unlabelled amount per service per day. A credit, a refund, a tax line and a support fee all arrived looking like usage, and there was no amortized figure at all, so a reserved instance paid for up front showed as a spike in the month it was bought and as free capacity for a year afterwards. The ingest request now carries UnblendedCost, AmortizedCost and UsageQuantity in one call — Cost Explorer bills per request, not per metric — and groups by RECORD_TYPE as well as SERVICE. The record type maps to charge_category, so the ledger can tell a credit from a charge. Amounts keep the sign Cost Explorer gives them, so a period total is a plain sum. An unrecognized record type becomes an Adjustment with a warning naming it: money moved, and filing it as Usage would quietly inflate what reads as consumption. A row is dropped only when both cost metrics are zero. Usage covered by a savings plan is zero unblended and non-zero amortized; dropping it would lose what the commitment bought. Grouping by service also mixes usage types, which Cost Explorer signals by returning the unit N/A — a quantity like that is not stored, because it cannot be added to anything. Which key is which is read from the response's GroupDefinitions rather than assumed from the request this build would have sent, so payloads already in the raw store still normalize; they normalize as Usage, which is what they always were. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 6 + docs/roadmap.md | 27 +- src/cloud/aws.rs | 327 +++++++++++++++--- .../aws_cost_and_usage_record_type.json | 106 ++++++ 4 files changed, 415 insertions(+), 51 deletions(-) create mode 100644 src/cloud/testdata/aws_cost_and_usage_record_type.json diff --git a/CHANGELOG.md b/CHANGELOG.md index bc89c7c..1891288 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,6 +32,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `normalize` is a pure function from a stored batch to FOCUS rows, so billing logic is testable from a recorded response and a mapping fix replays payloads on disk instead of paying for another fetch +- **AWS charges land as FOCUS rows** (roadmap P0/PR4) + - One Cost Explorer call now carries `UnblendedCost`, `AmortizedCost` and + `UsageQuantity`, grouped by service and record type + - `charge_category` comes from the record type, so credits, refunds, + taxes and support fees are each labelled as themselves instead of all + reading as usage; amounts keep their sign - **DeepSeek Integration** - DeepSeek API integration for balance queries - Display account balance instead of cost for DeepSeek accounts diff --git a/docs/roadmap.md b/docs/roadmap.md index cb2c79f..aa8a0f3 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -31,14 +31,15 @@ carry their weight for a personal ledger: ## Where we are -PR1 through PR3 have landed. A source is a registry row rather than an +PR1 through PR4 have landed. A source is a registry row rather than an enum variant; `billing.duckdb` holds `fct_charge`, `ingest_batch`, `fct_balance_snapshot` and `dim_fx_rate` behind a transactional -whole-period write; and every source now fetches raw payloads to Parquet -and normalizes them through a pure function, tested against a recorded -response. +whole-period write; every source fetches raw payloads to Parquet and +normalizes them through a pure function, tested against a recorded +response; and AWS lands as real FOCUS rows, with credits, refunds, taxes +and fees each labelled as themselves. -What is left in P0 is the mapping detail (PR4, PR5) and the read path +What is left in P0 is Alibaba Cloud and DeepSeek (PR5) and the read path (PR6). The dashboard still reads the two per-account cache tables in `cloudbridge.duckdb` — they go away in PR5, when the last source writes through the ledger. @@ -140,13 +141,27 @@ discount as the gap between `billed_cost` and `list_cost` rather than as `Credit` rows, and DeepSeek writes balance snapshots without deriving top-ups. -### PR4 · AWS to FOCUS +### PR4 · AWS to FOCUS — landed Cost Explorer currently requests only `UnblendedCost`, grouped by `SERVICE`. Request `UnblendedCost`, `AmortizedCost` and `UsageQuantity` in a single call — each call is billed, so do not split it — and add `RECORD_TYPE` to the grouping to populate `charge_category`. `cost_basis` is `authoritative`. +Three decisions worth recording: + +- Amounts keep the sign Cost Explorer gives them, so credits and refunds + stay negative and a period total is a plain sum. +- A record type this build does not recognize is an `Adjustment`, with a + warning naming it. Money moved; calling it `Usage` would quietly inflate + what reads as consumption. +- A row is only dropped when it is zero on *both* cost metrics. Usage + covered by a commitment is zero unblended and non-zero amortized, and + dropping it would lose what the commitment actually bought. Grouping by + service also mixes usage types, and Cost Explorer says so by returning + the unit `N/A`: a quantity like that is not stored, because it cannot be + added to anything. + ### PR5 · Alibaba Cloud and DeepSeek Alibaba Cloud `QueryBillOverview`: `PretaxAmount` to `billed_cost`, diff --git a/src/cloud/aws.rs b/src/cloud/aws.rs index 653f00e..9d54050 100644 --- a/src/cloud/aws.rs +++ b/src/cloud/aws.rs @@ -8,7 +8,7 @@ use sha2::{Digest, Sha256}; use super::raw::RawPart; use super::{BillingPeriod, BillingSource, CostData, CostSummary, Normalized, RawBatch, SourceId}; -use crate::ledger::Charge; +use crate::ledger::{Charge, ChargeCategory}; type HmacSha256 = Hmac; @@ -323,7 +323,22 @@ impl AwsCloudService { /// Name the Cost Explorer payload is stored under in a raw batch. const PART_COST_AND_USAGE: &str = "cost_and_usage"; -/// The GetCostAndUsage request body. +/// What was actually charged. +const METRIC_UNBLENDED: &str = "UnblendedCost"; +/// The same spend with commitment fees spread over the term they cover. +const METRIC_AMORTIZED: &str = "AmortizedCost"; +/// How much was consumed, when the grouping leaves one meaningful unit. +const METRIC_USAGE_QUANTITY: &str = "UsageQuantity"; + +/// Cost Explorer returns this unit when a group mixes usage types, which +/// grouping by service usually does. A quantity in mixed units cannot be +/// added to anything, so it is not stored. +const UNIT_NOT_APPLICABLE: &str = "N/A"; + +const DIMENSION_SERVICE: &str = "SERVICE"; +const DIMENSION_RECORD_TYPE: &str = "RECORD_TYPE"; + +/// The GetCostAndUsage request body for the display path. /// /// `group_by_service` off is the trend query: one total per day, which is /// a cheaper response than summing the grouped one client-side. @@ -338,34 +353,101 @@ fn cost_and_usage_request( "End": end_date }, "Granularity": "DAILY", - "Metrics": ["UnblendedCost"] + "Metrics": [METRIC_UNBLENDED] }); if group_by_service { request["GroupBy"] = serde_json::json!([{ "Type": "DIMENSION", - "Key": "SERVICE" + "Key": DIMENSION_SERVICE }]); } request } +/// The GetCostAndUsage request the ledger is built from. +/// +/// All three metrics ride in one request: Cost Explorer bills per request, +/// not per metric, so splitting them would triple the cost of an ingest +/// for nothing. `RECORD_TYPE` is what makes a credit distinguishable from +/// a charge — without it every line arrives as an unlabelled amount. +fn ledger_request(start_date: &str, end_date: &str) -> serde_json::Value { + serde_json::json!({ + "TimePeriod": { + "Start": start_date, + "End": end_date + }, + "Granularity": "DAILY", + "Metrics": [METRIC_UNBLENDED, METRIC_AMORTIZED, METRIC_USAGE_QUANTITY], + "GroupBy": [ + { + "Type": "DIMENSION", + "Key": DIMENSION_SERVICE + }, + { + "Type": "DIMENSION", + "Key": DIMENSION_RECORD_TYPE + } + ] + }) +} + +/// FOCUS category for an AWS `RECORD_TYPE`. +/// +/// Discounts and negations are `Adjustment` rather than `Credit`: they +/// reduce what a charge costs, whereas AWS's own `Credit` record type is a +/// balance applied against the bill. An unrecognized type is also +/// `Adjustment`, and says so in the log — money moved, and filing it as +/// `Usage` would quietly inflate what looks like consumption. +fn charge_category(record_type: &str) -> ChargeCategory { + match record_type { + "Usage" | "DiscountedUsage" | "SavingsPlanCoveredUsage" => ChargeCategory::Usage, + "Credit" => ChargeCategory::Credit, + "Tax" => ChargeCategory::Tax, + "Fee" | "RIFee" | "SavingsPlanUpfrontFee" | "SavingsPlanRecurringFee" | "Support" => { + ChargeCategory::Purchase + } + "Refund" + | "SavingsPlanNegation" + | "BundledDiscount" + | "PrivateRateDiscount" + | "Enterprise Discount Program Discount" + | "Solution Provider Program Discount" => ChargeCategory::Adjustment, + other => { + tracing::warn!( + "Unrecognized Cost Explorer record type {:?}; filed as an Adjustment", + other + ); + ChargeCategory::Adjustment + } + } +} + /// Turn a fetched Cost Explorer payload into ledger rows. /// -/// Pure — every input is in `batch`. Cost Explorer reports `UnblendedCost`, -/// which is what was actually charged, so `cost_basis` is `authoritative` -/// and `effective_cost` stays empty until PR4 also requests -/// `AmortizedCost`. Every row is `Usage` for the same reason: without -/// `RECORD_TYPE` in the grouping the payload cannot tell a credit from a -/// charge, and guessing would put refunds on the wrong side of the total. +/// Pure — every input is in `batch`. +/// +/// `UnblendedCost` is what was actually charged, so it is `billed_cost` +/// and `cost_basis` is `authoritative`; `AmortizedCost` spreads commitment +/// fees over the term they cover, which is `effective_cost`. Amounts keep +/// the sign Cost Explorer gave them, so a credit stays negative and a +/// total comes out right by summation alone. pub fn normalize(batch: &RawBatch) -> Result { #[derive(Deserialize)] struct CeResponse { + #[serde(rename = "GroupDefinitions")] + group_definitions: Option>, #[serde(rename = "ResultsByTime")] results_by_time: Option>, } + #[derive(Deserialize)] + struct GroupDefinition { + #[serde(rename = "Key")] + key: String, + } + #[derive(Deserialize)] struct TimeResult { #[serde(rename = "TimePeriod")] @@ -387,13 +469,7 @@ pub fn normalize(batch: &RawBatch) -> Result { #[serde(rename = "Keys")] keys: Vec, #[serde(rename = "Metrics")] - metrics: CostMetrics, - } - - #[derive(Deserialize)] - struct CostMetrics { - #[serde(rename = "UnblendedCost")] - unblended_cost: CostAmount, + metrics: std::collections::HashMap, } #[derive(Deserialize)] @@ -404,30 +480,74 @@ pub fn normalize(batch: &RawBatch) -> Result { unit: String, } + impl CostAmount { + fn value(&self) -> f64 { + self.amount.parse().unwrap_or(0.0) + } + } + let part = batch .part(PART_COST_AND_USAGE) .ok_or_else(|| anyhow!("Raw batch has no '{}' payload", PART_COST_AND_USAGE))?; let response: CeResponse = serde_json::from_str(&part.body) .map_err(|e| anyhow!("Failed to parse Cost Explorer payload: {}", e))?; + // Which key is which comes from the response itself rather than from + // the request this build would have sent, so a payload recorded by an + // older version still normalizes. + let definitions = response.group_definitions.unwrap_or_default(); + let position = |dimension: &str| definitions.iter().position(|d| d.key == dimension); + let service_at = position(DIMENSION_SERVICE).unwrap_or(0); + let record_type_at = position(DIMENSION_RECORD_TYPE); + let mut charges = Vec::new(); for result in response.results_by_time.unwrap_or_default() { let start = parse_day(&result.time_period.start)?; let end = parse_day(&result.time_period.end)?; for group in result.groups.unwrap_or_default() { - let amount: f64 = group.metrics.unblended_cost.amount.parse().unwrap_or(0.0); + let unblended = group.metrics.get(METRIC_UNBLENDED); + let amortized = group.metrics.get(METRIC_AMORTIZED); + let billed_cost = unblended.map(CostAmount::value); + let effective_cost = amortized.map(CostAmount::value); + // Cost Explorer returns a row for every service in the account, - // most of them zero. They carry no information and would bloat - // the fact table by an order of magnitude. - if amount == 0.0 { + // most of them zero on every metric. They carry no information + // and would bloat the fact table by an order of magnitude. A + // row that is zero unblended but non-zero amortized — usage a + // commitment already paid for — is not one of them. + if billed_cost.unwrap_or(0.0) == 0.0 && effective_cost.unwrap_or(0.0) == 0.0 { continue; } + // A quantity is only kept when the group leaves it in one unit. + let quantity = group + .metrics + .get(METRIC_USAGE_QUANTITY) + .filter(|q| q.unit != UNIT_NOT_APPLICABLE && !q.unit.is_empty()); + + // Without RECORD_TYPE in the grouping, credits and refunds are + // already netted into each service's amount and there is + // nothing left to label: such a payload is Usage throughout, + // which is what it was read as before the dimension was added. + let record_type = record_type_at.and_then(|at| group.keys.get(at)); + let category = record_type.map_or(ChargeCategory::Usage, |rt| charge_category(rt)); + charges.push(Charge { - service_name: group.keys.first().cloned(), - billed_cost: Some(amount), - ..Charge::new(start, end, group.metrics.unblended_cost.unit) + service_name: group.keys.get(service_at).cloned(), + charge_description: record_type.cloned(), + billed_cost, + effective_cost, + pricing_quantity: quantity.map(|q| q.value()), + pricing_unit: quantity.map(|q| q.unit.clone()), + charge_category: category, + ..Charge::new( + start, + end, + unblended + .or(amortized) + .map_or_else(|| "USD".to_string(), |amount| amount.unit.clone()), + ) }); } } @@ -644,10 +764,9 @@ impl BillingSource for AwsCloudService { } fn fetch(&self, period: &BillingPeriod) -> Result> { - let request = cost_and_usage_request( + let request = ledger_request( &period.start().to_string(), &period.end_exclusive().to_string(), - true, ); let body = self.cost_and_usage_raw(&request)?; @@ -804,8 +923,24 @@ mod tests { use super::*; use crate::ledger::{ChargeCategory, CostBasis}; - /// One recorded GetCostAndUsage response, DAILY and grouped by SERVICE. - const COST_AND_USAGE: &str = include_str!("testdata/aws_cost_and_usage.json"); + /// A recorded GetCostAndUsage response as this build asks for it: + /// three metrics, grouped by service and record type. + const COST_AND_USAGE: &str = include_str!("testdata/aws_cost_and_usage_record_type.json"); + + /// A response recorded before RECORD_TYPE was in the grouping, of the + /// kind already sitting in the raw store. + const LEGACY_COST_AND_USAGE: &str = include_str!("testdata/aws_cost_and_usage.json"); + + fn charge<'a>(normalized: &'a Normalized, service: &str, description: &str) -> &'a Charge { + normalized + .charges + .iter() + .find(|charge| { + charge.service_name.as_deref() == Some(service) + && charge.charge_description.as_deref() == Some(description) + }) + .unwrap_or_else(|| panic!("no {} / {} charge", service, description)) + } fn recorded_batch(body: &str) -> RawBatch { RawBatch { @@ -829,40 +964,126 @@ mod tests { fn a_recorded_response_normalizes_to_one_charge_per_service_day() { let normalized = normalize(&recorded_batch(COST_AND_USAGE)).unwrap(); - // Three non-zero groups across two days; the zero-cost KMS row is + // Eight non-zero groups across two days; the all-zero KMS row is // dropped. - assert_eq!(normalized.charges.len(), 3); + assert_eq!(normalized.charges.len(), 8); assert!(normalized.balances.is_empty()); - let first = &normalized.charges[0]; - assert_eq!( - first.service_name.as_deref(), - Some("Amazon Elastic Compute Cloud - Compute") + let ec2 = charge( + &normalized, + "Amazon Elastic Compute Cloud - Compute", + "Usage", ); - assert_eq!(first.billed_cost, Some(12.45)); - assert_eq!(first.billing_currency, "USD"); - assert_eq!(first.charge_category, ChargeCategory::Usage); - assert_eq!(first.cost_basis, CostBasis::Authoritative); + assert_eq!(ec2.billed_cost, Some(12.45)); + assert_eq!(ec2.effective_cost, Some(10.20)); + assert_eq!(ec2.billing_currency, "USD"); + assert_eq!(ec2.cost_basis, CostBasis::Authoritative); assert_eq!( - first.charge_period_start.to_rfc3339(), + ec2.charge_period_start.to_rfc3339(), "2026-08-01T00:00:00+00:00" ); assert_eq!( - first.charge_period_end.to_rfc3339(), + ec2.charge_period_end.to_rfc3339(), "2026-08-02T00:00:00+00:00" ); + } + + #[test] + fn the_record_type_decides_the_charge_category() { + let normalized = normalize(&recorded_batch(COST_AND_USAGE)).unwrap(); - // Amortization needs AmortizedCost, which this request does not ask - // for: the column stays empty rather than being filled with the - // unblended figure. - assert_eq!(first.effective_cost, None); + let category = |service: &str, record_type: &str| { + charge(&normalized, service, record_type).charge_category + }; + let ec2 = "Amazon Elastic Compute Cloud - Compute"; + + assert_eq!(category(ec2, "Usage"), ChargeCategory::Usage); + assert_eq!( + category(ec2, "SavingsPlanCoveredUsage"), + ChargeCategory::Usage + ); + assert_eq!(category(ec2, "Credit"), ChargeCategory::Credit); + assert_eq!(category(ec2, "Refund"), ChargeCategory::Adjustment); + assert_eq!(category("Tax", "Tax"), ChargeCategory::Tax); + assert_eq!( + category("AWS Support (Developer)", "Fee"), + ChargeCategory::Purchase + ); + // A record type this build has never seen still moved money, so it + // is kept and labelled as an adjustment rather than as usage. + assert_eq!( + category("Amazon Route 53", "SomeFutureRecordType"), + ChargeCategory::Adjustment + ); + } + + #[test] + fn credits_and_refunds_keep_their_sign_so_the_total_nets_out() { + let normalized = normalize(&recorded_batch(COST_AND_USAGE)).unwrap(); + let ec2 = "Amazon Elastic Compute Cloud - Compute"; + + assert_eq!(charge(&normalized, ec2, "Credit").billed_cost, Some(-3.0)); + assert_eq!(charge(&normalized, ec2, "Refund").billed_cost, Some(-1.5)); let total: f64 = normalized .charges .iter() .filter_map(|charge| charge.billed_cost) .sum(); - assert_eq!(total, 25.1); + // 12.45 + 0 + 0.75 + 29.00 - 3.00 - 1.50 + 2.10 + 1.23 + assert!((total - 41.03).abs() < 1e-9, "got {total}"); + } + + #[test] + fn usage_a_commitment_already_paid_for_is_not_mistaken_for_an_empty_row() { + let normalized = normalize(&recorded_batch(COST_AND_USAGE)).unwrap(); + let covered = charge( + &normalized, + "Amazon Elastic Compute Cloud - Compute", + "SavingsPlanCoveredUsage", + ); + + // Nothing was charged for it this day, but the amortized figure is + // what the commitment cost — dropping the row would lose it. + assert_eq!(covered.billed_cost, Some(0.0)); + assert_eq!(covered.effective_cost, Some(3.10)); + } + + #[test] + fn a_quantity_is_only_kept_when_it_has_one_real_unit() { + let normalized = normalize(&recorded_batch(COST_AND_USAGE)).unwrap(); + + let ec2 = charge( + &normalized, + "Amazon Elastic Compute Cloud - Compute", + "Usage", + ); + assert_eq!(ec2.pricing_quantity, Some(24.0)); + assert_eq!(ec2.pricing_unit.as_deref(), Some("Hrs")); + + // Grouping by service mixes usage types, and Cost Explorer says so + // with "N/A". A number in mixed units cannot be added to anything. + let s3 = charge(&normalized, "Amazon Simple Storage Service", "Usage"); + assert_eq!(s3.pricing_quantity, None); + assert_eq!(s3.pricing_unit, None); + } + + #[test] + fn a_payload_recorded_before_record_type_still_normalizes() { + let normalized = normalize(&recorded_batch(LEGACY_COST_AND_USAGE)).unwrap(); + + assert_eq!(normalized.charges.len(), 3); + // Credits were already netted into each service's amount, so there + // is nothing to label and nothing to amortize. + assert!(normalized + .charges + .iter() + .all(|charge| charge.charge_category == ChargeCategory::Usage)); + assert!(normalized + .charges + .iter() + .all(|charge| charge.effective_cost.is_none())); + assert_eq!(normalized.charges[0].billed_cost, Some(12.45)); } #[test] @@ -899,4 +1120,20 @@ mod tests { let totals = cost_and_usage_request("2026-08-01", "2026-09-01", false); assert!(totals.get("GroupBy").is_none()); } + + #[test] + fn the_ledger_request_carries_every_metric_in_one_call() { + let request = ledger_request("2026-08-01", "2026-09-01"); + + // Cost Explorer bills per request: three metrics, one call. + let metrics = request["Metrics"].as_array().unwrap(); + assert_eq!(metrics.len(), 3); + assert!(metrics.iter().any(|m| m == "UnblendedCost")); + assert!(metrics.iter().any(|m| m == "AmortizedCost")); + assert!(metrics.iter().any(|m| m == "UsageQuantity")); + + assert_eq!(request["GroupBy"][0]["Key"], "SERVICE"); + assert_eq!(request["GroupBy"][1]["Key"], "RECORD_TYPE"); + assert_eq!(request["Granularity"], "DAILY"); + } } diff --git a/src/cloud/testdata/aws_cost_and_usage_record_type.json b/src/cloud/testdata/aws_cost_and_usage_record_type.json new file mode 100644 index 0000000..3abb185 --- /dev/null +++ b/src/cloud/testdata/aws_cost_and_usage_record_type.json @@ -0,0 +1,106 @@ +{ + "GroupDefinitions": [ + { + "Type": "DIMENSION", + "Key": "SERVICE" + }, + { + "Type": "DIMENSION", + "Key": "RECORD_TYPE" + } + ], + "ResultsByTime": [ + { + "TimePeriod": { + "Start": "2026-08-01", + "End": "2026-08-02" + }, + "Total": {}, + "Groups": [ + { + "Keys": ["Amazon Elastic Compute Cloud - Compute", "Usage"], + "Metrics": { + "UnblendedCost": { "Amount": "12.4500000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "10.2000000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "24.0000000000", "Unit": "Hrs" } + } + }, + { + "Keys": ["Amazon Elastic Compute Cloud - Compute", "SavingsPlanCoveredUsage"], + "Metrics": { + "UnblendedCost": { "Amount": "0E-10", "Unit": "USD" }, + "AmortizedCost": { "Amount": "3.1000000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "12.0000000000", "Unit": "Hrs" } + } + }, + { + "Keys": ["Amazon Simple Storage Service", "Usage"], + "Metrics": { + "UnblendedCost": { "Amount": "0.7500000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "0.7500000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "30.0000000000", "Unit": "N/A" } + } + }, + { + "Keys": ["AWS Support (Developer)", "Fee"], + "Metrics": { + "UnblendedCost": { "Amount": "29.0000000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "29.0000000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "1.0000000000", "Unit": "N/A" } + } + }, + { + "Keys": ["AWS Key Management Service", "Usage"], + "Metrics": { + "UnblendedCost": { "Amount": "0", "Unit": "USD" }, + "AmortizedCost": { "Amount": "0", "Unit": "USD" }, + "UsageQuantity": { "Amount": "0", "Unit": "N/A" } + } + } + ], + "Estimated": false + }, + { + "TimePeriod": { + "Start": "2026-08-02", + "End": "2026-08-03" + }, + "Total": {}, + "Groups": [ + { + "Keys": ["Amazon Elastic Compute Cloud - Compute", "Credit"], + "Metrics": { + "UnblendedCost": { "Amount": "-3.0000000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "-3.0000000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "0", "Unit": "N/A" } + } + }, + { + "Keys": ["Amazon Elastic Compute Cloud - Compute", "Refund"], + "Metrics": { + "UnblendedCost": { "Amount": "-1.5000000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "-1.5000000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "0", "Unit": "N/A" } + } + }, + { + "Keys": ["Tax", "Tax"], + "Metrics": { + "UnblendedCost": { "Amount": "2.1000000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "2.1000000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "0", "Unit": "N/A" } + } + }, + { + "Keys": ["Amazon Route 53", "SomeFutureRecordType"], + "Metrics": { + "UnblendedCost": { "Amount": "1.2300000000", "Unit": "USD" }, + "AmortizedCost": { "Amount": "1.2300000000", "Unit": "USD" }, + "UsageQuantity": { "Amount": "0", "Unit": "N/A" } + } + } + ], + "Estimated": true + } + ] +} From adeab42294211df98e18ef66afeeb117dc2c7b99 Mon Sep 17 00:00:00 2001 From: JetSquirrel Date: Sun, 30 Aug 2026 01:16:21 +0800 Subject: [PATCH 4/5] Land Alibaba Cloud deductions and DeepSeek top-ups in the ledger MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Alibaba Cloud reports gross and net on the same line, so a voucher was only visible as the difference between two numbers. Each deduction — InvoiceDiscount, coupons, cash coupons, a stored-value card — is now a Credit row of its own beside a usage charge carrying the gross amount. Putting the net amount on the usage row and the deductions beside it would have counted them twice; AWS bills a discount as a separate line, Alibaba Cloud does not. Decomposed this way a product's rows sum to PretaxAmount, which is what was actually charged, so a period total stays a plain sum. When the deductions we can name do not close the gap between gross and net, the remainder becomes one Adjustment row and a warning naming the product: the bill accounts for that money even where this parser cannot, and dropping it would leave the ledger disagreeing with the invoice. DeepSeek publishes a balance, which is state. It says what is left, never what was bought, so a purchase can only be inferred from movement: a rise in the topped-up balance between two consecutive observations is money that went in. Those rows are derived at ingest rather than written once, because replacing a period clears whatever was there before — recomputing them keeps a re-ingest idempotent. The first observation of an account yields nothing: a balance that was simply there the first time it was looked at was not witnessed being paid for, and inventing a purchase for it would drop the whole opening balance into whichever month the account happened to be added. Balances that differ only in currency never derive a purchase from each other, and a falling topped-up balance is consumption, not a top-up. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 7 + docs/roadmap.md | 35 ++- src/cloud/aliyun.rs | 230 ++++++++++++++++--- src/cloud/testdata/aliyun_bill_overview.json | 27 +++ src/ingest.rs | 25 +- src/ledger/mod.rs | 210 +++++++++++++++++ 6 files changed, 484 insertions(+), 50 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1891288..446e7ec 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -38,6 +38,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - `charge_category` comes from the record type, so credits, refunds, taxes and support fees are each labelled as themselves instead of all reading as usage; amounts keep their sign +- **Alibaba Cloud and DeepSeek land in the ledger** (roadmap P0/PR5) + - Each Alibaba Cloud voucher, coupon and discount becomes its own + `Credit` row beside a gross usage charge, so a product's rows sum to + what was actually charged; an unexplained gap becomes one `Adjustment` + row instead of vanishing + - DeepSeek balances are recorded as snapshots, and a rise in the + topped-up balance between observations is derived as a `Purchase` - **DeepSeek Integration** - DeepSeek API integration for balance queries - Display account balance instead of cost for DeepSeek accounts diff --git a/docs/roadmap.md b/docs/roadmap.md index aa8a0f3..d18cadb 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -31,18 +31,18 @@ carry their weight for a personal ledger: ## Where we are -PR1 through PR4 have landed. A source is a registry row rather than an +PR1 through PR5 have landed. A source is a registry row rather than an enum variant; `billing.duckdb` holds `fct_charge`, `ingest_batch`, `fct_balance_snapshot` and `dim_fx_rate` behind a transactional whole-period write; every source fetches raw payloads to Parquet and normalizes them through a pure function, tested against a recorded -response; and AWS lands as real FOCUS rows, with credits, refunds, taxes -and fees each labelled as themselves. +response; and all three sources land as real FOCUS rows. -What is left in P0 is Alibaba Cloud and DeepSeek (PR5) and the read path -(PR6). The dashboard still reads the two per-account cache tables in -`cloudbridge.duckdb` — they go away in PR5, when the last source writes -through the ledger. +What is left in P0 is the read path (PR6). Nothing reads the ledger yet: +the dashboard still calls each provider's display method and caches the +result in the two per-account tables in `cloudbridge.duckdb`. Those go +away with PR6, not before — they are the only thing feeding the UI until +it reads through `v_charge_normalized`. One of the three structural problems is still open: @@ -162,18 +162,37 @@ Three decisions worth recording: the unit `N/A`: a quantity like that is not stored, because it cannot be added to anything. -### PR5 · Alibaba Cloud and DeepSeek +### PR5 · Alibaba Cloud and DeepSeek — landed Alibaba Cloud `QueryBillOverview`: `PretaxAmount` to `billed_cost`, `PretaxGrossAmount` to `list_cost`, each voucher/deduction as its own `Credit` row. Currency CNY. +As landed, the usage row carries the **gross** amount and each deduction +is a negative `Credit` beside it. Putting the net amount on the usage row +*and* the deductions next to it would count them twice — Alibaba Cloud +reports both figures on the same line, unlike AWS, which bills the +discount as a line of its own. Decomposed this way a product's rows sum to +`PretaxAmount`, which is what was actually charged, and a total stays a +plain sum. Where the named deductions do not close the gap between gross +and net, the remainder becomes one `Adjustment` row rather than +disappearing. + DeepSeek reports a balance, which is state, not a charge. It moves to `fct_balance_snapshot`; only top-ups become `fct_charge` rows with `charge_category = Purchase`. The current code stuffs the balance into `current_month_cost`, which is semantically wrong and blocks any correct total. +Top-ups are *derived*, not stored: a rise in the topped-up balance between +two consecutive observations is the only evidence of a purchase such a +source gives, and re-ingesting a period recomputes them, since replacing a +period clears what was there. The first observation of an account yields +nothing — a balance that was simply there the first time it was looked at +was not witnessed being paid for. The display path still reports the +balance as `current_month_cost`; that is PR6's to fix, along with +everything else the dashboard reads. + ### PR6 · Read through views, fix cross-currency Amounts are stored in their original currency. Conversion happens in a diff --git a/src/cloud/aliyun.rs b/src/cloud/aliyun.rs index be30c63..6d1aaba 100644 --- a/src/cloud/aliyun.rs +++ b/src/cloud/aliyun.rs @@ -12,7 +12,7 @@ use super::{ BillingPeriod, BillingSource, CostData, CostSummary, Normalized, RawBatch, ServiceCost, SourceId, }; -use crate::ledger::Charge; +use crate::ledger::{Charge, ChargeCategory}; type HmacSha1 = Hmac; @@ -412,17 +412,25 @@ fn parse_bill_overview(response: &BillOverviewResponse) -> (f64, Vec Result { let part = batch .part(PART_BILL_OVERVIEW) @@ -451,26 +459,61 @@ pub fn normalize(batch: &RawBatch) -> Result { let mut charges = Vec::new(); for item in items { - let billed = item.pretax_amount.unwrap_or(0.0); - let list = item.pretax_gross_amount.unwrap_or(0.0); - // A product with nothing on either side of the discount was not + let net = item.pretax_amount.unwrap_or(0.0); + let gross = item.pretax_gross_amount.unwrap_or(0.0); + // A product with nothing on either side of the deductions was not // used this month. - if billed == 0.0 && list == 0.0 { + if net == 0.0 && gross == 0.0 { continue; } - charges.push(Charge { - service_name: item.product_name, + let currency = item.currency.clone().unwrap_or_else(|| "CNY".to_string()); + let template = || Charge { + service_name: item.product_name.clone(), // ProductCode is stable across locales; ProductName is not. - service_category: item.product_code, - billed_cost: Some(billed), - list_cost: Some(list), - ..Charge::new( - start, - end, - item.currency.unwrap_or_else(|| "CNY".to_string()), - ) + service_category: item.product_code.clone(), + ..Charge::new(start, end, currency.clone()) + }; + + charges.push(Charge { + billed_cost: Some(gross), + list_cost: Some(gross), + ..template() }); + + let mut deducted = 0.0; + for (name, amount) in item.deductions() { + if amount == 0.0 { + continue; + } + deducted += amount; + charges.push(Charge { + charge_category: ChargeCategory::Credit, + charge_description: Some(name.to_string()), + billed_cost: Some(-amount), + ..template() + }); + } + + // Anything left between gross, the deductions we know the names of, + // and the net figure is money the bill accounts for and this parser + // does not. Recording it keeps the total honest and makes the gap + // visible instead of losing it. + let residual = gross - deducted - net; + if residual.abs() > RECONCILIATION_TOLERANCE { + tracing::warn!( + "Alibaba Cloud bill for {} does not reconcile: {:.2} {} unaccounted for", + item.product_name.as_deref().unwrap_or("?"), + residual, + currency + ); + charges.push(Charge { + charge_category: ChargeCategory::Adjustment, + charge_description: Some(UNRECONCILED.to_string()), + billed_cost: Some(-residual), + ..template() + }); + } } Ok(Normalized { @@ -525,12 +568,41 @@ struct BillOverviewItems { struct BillOverviewItem { product_code: Option, product_name: Option, + /// `Subscription` (prepaid) or `PayAsYouGo`. + subscription_type: Option, + /// What was charged, after every deduction below. pretax_amount: Option, + /// What it would have cost before any of them. #[serde(rename = "PretaxGrossAmount")] pretax_gross_amount: Option, + /// Negotiated or activity discount. + invoice_discount: Option, + /// Coupons (代金券), cash coupons and stored-value cards. + deducted_by_coupons: Option, + deducted_by_cash_coupons: Option, + deducted_by_prepaid_card: Option, currency: Option, } +impl BillOverviewItem { + /// The deductions between the gross and the net amount, each named as + /// Alibaba Cloud names it. + fn deductions(&self) -> [(&'static str, f64); 4] { + [ + ("InvoiceDiscount", self.invoice_discount.unwrap_or(0.0)), + ("DeductedByCoupons", self.deducted_by_coupons.unwrap_or(0.0)), + ( + "DeductedByCashCoupons", + self.deducted_by_cash_coupons.unwrap_or(0.0), + ), + ( + "DeductedByPrepaidCard", + self.deducted_by_prepaid_card.unwrap_or(0.0), + ), + ] + } +} + #[derive(Debug, Deserialize)] #[serde(rename_all = "PascalCase")] #[allow(dead_code)] @@ -572,7 +644,7 @@ struct InstanceBillItem { #[cfg(test)] mod tests { use super::*; - use crate::ledger::{ChargeCategory, CostBasis}; + use crate::ledger::CostBasis; /// One recorded QueryBillOverview response. const BILL_OVERVIEW: &str = include_str!("testdata/aliyun_bill_overview.json"); @@ -588,21 +660,109 @@ mod tests { } } + /// Rows for one product, in the order the normalizer emitted them. + fn rows<'a>(normalized: &'a Normalized, product: &str) -> Vec<&'a Charge> { + normalized + .charges + .iter() + .filter(|charge| charge.service_category.as_deref() == Some(product)) + .collect() + } + + #[test] + fn a_product_becomes_a_gross_usage_charge() { + let normalized = normalize(&recorded_batch(BILL_OVERVIEW)).unwrap(); + + let oss = rows(&normalized, "oss"); + assert_eq!(oss.len(), 1, "nothing was deducted from OSS"); + + let charge = oss[0]; + assert_eq!(charge.service_name.as_deref(), Some("对象存储 OSS")); + assert_eq!(charge.billed_cost, Some(42.0)); + assert_eq!(charge.list_cost, Some(42.0)); + assert_eq!(charge.billing_currency, "CNY"); + assert_eq!(charge.charge_category, ChargeCategory::Usage); + assert_eq!(charge.cost_basis, CostBasis::Authoritative); + + // The unused CDN product is dropped entirely. + assert!(rows(&normalized, "cdn").is_empty()); + } + + #[test] + fn each_deduction_becomes_its_own_credit_row() { + let normalized = normalize(&recorded_batch(BILL_OVERVIEW)).unwrap(); + let ecs = rows(&normalized, "ecs"); + + assert_eq!(ecs[0].billed_cost, Some(320.5)); + assert_eq!(ecs[0].charge_category, ChargeCategory::Usage); + + let credits: Vec<(&str, Option)> = ecs[1..] + .iter() + .map(|charge| { + ( + charge.charge_description.as_deref().unwrap(), + charge.billed_cost, + ) + }) + .collect(); + assert_eq!( + credits, + vec![ + ("InvoiceDiscount", Some(-22.05)), + ("DeductedByCoupons", Some(-10.0)), + ] + ); + assert!(ecs[1..] + .iter() + .all(|charge| charge.charge_category == ChargeCategory::Credit)); + } + + #[test] + fn a_products_rows_sum_to_what_was_actually_charged() { + let normalized = normalize(&recorded_batch(BILL_OVERVIEW)).unwrap(); + + // PretaxAmount for ECS is 288.45, for OSS 42.00, for RDS 62.00. + for (product, charged) in [("ecs", 288.45), ("oss", 42.0), ("rds", 62.0)] { + let total: f64 = rows(&normalized, product) + .iter() + .filter_map(|charge| charge.billed_cost) + .sum(); + assert!( + (total - charged).abs() < 1e-9, + "{product}: {total} != {charged}" + ); + } + } + #[test] - fn a_recorded_overview_normalizes_to_one_charge_per_product() { + fn a_deduction_this_parser_cannot_name_is_still_accounted_for() { let normalized = normalize(&recorded_batch(BILL_OVERVIEW)).unwrap(); + let rds = rows(&normalized, "rds"); + + // RDS: 100.00 gross, 30.00 off a stored-value card, 62.00 charged. + // The remaining 8.00 is a deduction under a name this parser does + // not know; it is recorded rather than dropped. + let unreconciled = rds + .iter() + .find(|charge| charge.charge_description.as_deref() == Some(UNRECONCILED)) + .expect("the gap is recorded"); + assert_eq!(unreconciled.billed_cost, Some(-8.0)); + assert_eq!(unreconciled.charge_category, ChargeCategory::Adjustment); + } - // The unused CDN product is dropped. - assert_eq!(normalized.charges.len(), 2); - - let ecs = &normalized.charges[0]; - assert_eq!(ecs.service_name.as_deref(), Some("云服务器 ECS")); - assert_eq!(ecs.service_category.as_deref(), Some("ecs")); - assert_eq!(ecs.billed_cost, Some(288.45)); - assert_eq!(ecs.list_cost, Some(320.5)); - assert_eq!(ecs.billing_currency, "CNY"); - assert_eq!(ecs.charge_category, ChargeCategory::Usage); - assert_eq!(ecs.cost_basis, CostBasis::Authoritative); + #[test] + fn rounding_does_not_produce_a_reconciliation_row() { + let normalized = normalize(&recorded_batch( + r#"{"Code":"Success","Data":{"Items":{"Item":[ + {"ProductCode":"ecs","ProductName":"ECS","PretaxGrossAmount":10.0, + "InvoiceDiscount":0.001,"PretaxAmount":9.999,"Currency":"CNY"}]}}}"#, + )) + .unwrap(); + + assert!(normalized + .charges + .iter() + .all(|charge| charge.charge_description.as_deref() != Some(UNRECONCILED))); } #[test] diff --git a/src/cloud/testdata/aliyun_bill_overview.json b/src/cloud/testdata/aliyun_bill_overview.json index 428d04d..718bc85 100644 --- a/src/cloud/testdata/aliyun_bill_overview.json +++ b/src/cloud/testdata/aliyun_bill_overview.json @@ -12,21 +12,48 @@ { "ProductCode": "ecs", "ProductName": "云服务器 ECS", + "SubscriptionType": "PayAsYouGo", "PretaxGrossAmount": 320.5, + "InvoiceDiscount": 22.05, + "DeductedByCoupons": 10.0, + "DeductedByCashCoupons": 0, + "DeductedByPrepaidCard": 0, "PretaxAmount": 288.45, "Currency": "CNY" }, { "ProductCode": "oss", "ProductName": "对象存储 OSS", + "SubscriptionType": "PayAsYouGo", "PretaxGrossAmount": 42.0, + "InvoiceDiscount": 0, + "DeductedByCoupons": 0, + "DeductedByCashCoupons": 0, + "DeductedByPrepaidCard": 0, "PretaxAmount": 42.0, "Currency": "CNY" }, + { + "ProductCode": "rds", + "ProductName": "云数据库 RDS", + "SubscriptionType": "Subscription", + "PretaxGrossAmount": 100.0, + "InvoiceDiscount": 0, + "DeductedByCoupons": 0, + "DeductedByCashCoupons": 0, + "DeductedByPrepaidCard": 30.0, + "PretaxAmount": 62.0, + "Currency": "CNY" + }, { "ProductCode": "cdn", "ProductName": "CDN", + "SubscriptionType": "PayAsYouGo", "PretaxGrossAmount": 0, + "InvoiceDiscount": 0, + "DeductedByCoupons": 0, + "DeductedByCashCoupons": 0, + "DeductedByPrepaidCard": 0, "PretaxAmount": 0, "Currency": "CNY" } diff --git a/src/ingest.rs b/src/ingest.rs index ee8f03f..b09ceb4 100644 --- a/src/ingest.rs +++ b/src/ingest.rs @@ -106,6 +106,10 @@ fn persist(batch: &RawBatch) -> Result { } /// Replace the period in the ledger with what the normalizer produced. +/// +/// Balances go in first. A source that reports only a balance reports no +/// purchases either, so its purchases are derived from the movement +/// between observations — including the one just made. fn record(batch: &RawBatch, normalized: &Normalized, raw_path: &Path) -> Result { let key = PeriodKey::new( batch.provider.clone(), @@ -113,28 +117,35 @@ fn record(batch: &RawBatch, normalized: &Normalized, raw_path: &Path) -> Result< batch.period.label(), ); + for balance in &normalized.balances { + ledger::record_balance(balance)?; + } + + let mut charges = normalized.charges.clone(); + if !normalized.balances.is_empty() { + // Recomputed on every ingest rather than written once, because + // replacing the period clears whatever was there before. + charges.extend(ledger::top_up_charges(&key)?); + } + ledger::replace_period( &key, &batch.batch_id, - &normalized.charges, + &charges, Some(&raw_path.to_string_lossy()), )?; - for balance in &normalized.balances { - ledger::record_balance(balance)?; - } - tracing::info!( "Ingested {} {}: {} charge(s), {} balance(s)", batch.provider, batch.period.label(), - normalized.charges.len(), + charges.len(), normalized.balances.len() ); Ok(IngestOutcome { batch_id: batch.batch_id.clone(), - charges: normalized.charges.len(), + charges: charges.len(), balances: normalized.balances.len(), raw_path: raw_path.to_path_buf(), }) diff --git a/src/ledger/mod.rs b/src/ledger/mod.rs index f3e6c50..cbc2db8 100644 --- a/src/ledger/mod.rs +++ b/src/ledger/mod.rs @@ -232,6 +232,129 @@ pub fn record_balance(snapshot: &BalanceSnapshot) -> Result<()> { with_connection(|conn| write_balance(conn, snapshot)) } +/// Purchases derived from the balance history of one period. +/// +/// A source that only reports a balance never reports a purchase: what it +/// publishes is state — what is left — not what was bought. A rise in the +/// topped-up balance between two consecutive observations is money that +/// went in, and it is the only evidence of a purchase such a source gives. +/// +/// Derived rather than stored, so that re-ingesting a period recomputes +/// them: [`replace_period`] clears the period first, and a top-up that was +/// only ever written once would not survive that. +/// +/// The first observation of an account produces nothing. A balance that is +/// simply *there* the first time it is looked at was not witnessed being +/// paid for, and inventing a purchase for it would put the whole opening +/// balance into whichever month the account happened to be added. +pub fn top_up_charges(key: &PeriodKey) -> Result> { + with_connection(|conn| derive_top_ups(conn, key)) +} + +fn derive_top_ups(conn: &Connection, key: &PeriodKey) -> Result> { + let (period_start, period_end) = period_bounds(&key.billing_period)?; + + // Everything up to the end of the period, so the observation that + // precedes the period can serve as the baseline for the first rise + // inside it. + let mut stmt = conn.prepare( + "SELECT currency, CAST(observed_at AS VARCHAR), topped_up_balance + FROM fct_balance_snapshot + WHERE provider = ? AND account_id = ? AND observed_at < CAST(? AS TIMESTAMP) + ORDER BY currency, observed_at", + )?; + + let rows = stmt + .query_map( + params![ + key.provider, + key.account_id, + period_end.format(TIMESTAMP_FORMAT).to_string() + ], + |row| { + Ok(( + row.get::<_, String>(0)?, + row.get::<_, String>(1)?, + row.get::<_, Option>(2)?, + )) + }, + )? + .collect::, _>>()?; + + let mut charges = Vec::new(); + let mut previous: Option<(String, f64)> = None; + + for (currency, observed_at, topped_up) in rows { + let observed_at = parse_timestamp(&observed_at)?; + let Some(topped_up) = topped_up else { + // Nothing to compare against, and nothing to compare from. + previous = None; + continue; + }; + + let baseline = previous + .take() + .filter(|(seen_currency, _)| seen_currency == ¤cy); + previous = Some((currency.clone(), topped_up)); + + let Some((_, before)) = baseline else { + continue; + }; + let added = topped_up - before; + if added <= 0.0 || observed_at < period_start { + // A falling topped-up balance is consumption, not a purchase. + continue; + } + + charges.push(Charge { + charge_category: ChargeCategory::Purchase, + charge_description: Some("Top-up".to_string()), + billed_cost: Some(added), + ..Charge::new(observed_at, observed_at, currency) + }); + } + + Ok(charges) +} + +/// First instant of a `YYYY-MM` period and of the one after it. +fn period_bounds(billing_period: &str) -> Result<(DateTime, DateTime)> { + let (year, month) = billing_period + .split_once('-') + .ok_or_else(|| anyhow!("Not a YYYY-MM billing period: {:?}", billing_period))?; + let year: i32 = year.parse()?; + let month: u32 = month.parse()?; + + let start = chrono::NaiveDate::from_ymd_opt(year, month, 1) + .ok_or_else(|| anyhow!("Not a real billing period: {:?}", billing_period))?; + let next = if month == 12 { + chrono::NaiveDate::from_ymd_opt(year + 1, 1, 1) + } else { + chrono::NaiveDate::from_ymd_opt(year, month + 1, 1) + } + .expect("the month after a real one exists"); + + Ok(( + start + .and_hms_opt(0, 0, 0) + .expect("midnight exists") + .and_utc(), + next.and_hms_opt(0, 0, 0) + .expect("midnight exists") + .and_utc(), + )) +} + +/// Parse a timestamp as DuckDB renders it, `YYYY-MM-DD HH:MM:SS` in UTC. +fn parse_timestamp(value: &str) -> Result> { + let stamp = value.split('.').next().unwrap_or(value); + Ok( + chrono::NaiveDateTime::parse_from_str(stamp, TIMESTAMP_FORMAT) + .map_err(|e| anyhow!("Unexpected timestamp {:?} in the ledger: {}", value, e))? + .and_utc(), + ) +} + fn write_period( conn: &mut Connection, key: &PeriodKey, @@ -612,6 +735,93 @@ mod tests { assert_eq!(cost, None); } + fn balance(day: u32, topped_up: f64) -> BalanceSnapshot { + BalanceSnapshot { + provider: "DeepSeek".to_string(), + account_id: "acct-3".to_string(), + observed_at: at(day), + balance: topped_up + 5.0, + granted_balance: Some(5.0), + topped_up_balance: Some(topped_up), + currency: "CNY".to_string(), + } + } + + fn deepseek_key() -> PeriodKey { + PeriodKey::new("DeepSeek", "acct-3", "2026-08") + } + + #[test] + fn a_rise_in_the_topped_up_balance_is_a_purchase() { + let mut conn = conn(); + + write_balance(&mut conn, &balance(1, 20.0)).unwrap(); + write_balance(&mut conn, &balance(2, 100.0)).unwrap(); + // Spending brings it back down; that is not a purchase. + write_balance(&mut conn, &balance(3, 60.0)).unwrap(); + write_balance(&mut conn, &balance(4, 160.0)).unwrap(); + + let charges = derive_top_ups(&conn, &deepseek_key()).unwrap(); + + assert_eq!(charges.len(), 2); + assert_eq!(charges[0].billed_cost, Some(80.0)); + assert_eq!(charges[1].billed_cost, Some(100.0)); + assert!(charges + .iter() + .all(|charge| charge.charge_category == ChargeCategory::Purchase)); + assert_eq!(charges[0].billing_currency, "CNY"); + assert_eq!(charges[0].charge_period_start, at(2)); + } + + #[test] + fn the_first_observation_of_an_account_is_not_a_purchase() { + let mut conn = conn(); + write_balance(&mut conn, &balance(1, 500.0)).unwrap(); + + // A balance that was simply there the first time it was looked at + // was not witnessed being paid for. + assert!(derive_top_ups(&conn, &deepseek_key()).unwrap().is_empty()); + } + + #[test] + fn a_top_up_across_a_month_boundary_belongs_to_the_month_it_was_seen_in() { + let mut conn = conn(); + + let july = BalanceSnapshot { + observed_at: Utc.with_ymd_and_hms(2026, 7, 31, 12, 0, 0).unwrap(), + ..balance(1, 20.0) + }; + write_balance(&mut conn, &july).unwrap(); + write_balance(&mut conn, &balance(1, 120.0)).unwrap(); + + // July has the baseline but no rise of its own. + let in_july = + derive_top_ups(&conn, &PeriodKey::new("DeepSeek", "acct-3", "2026-07")).unwrap(); + assert!(in_july.is_empty()); + + let in_august = derive_top_ups(&conn, &deepseek_key()).unwrap(); + assert_eq!(in_august.len(), 1); + assert_eq!(in_august[0].billed_cost, Some(100.0)); + } + + #[test] + fn balances_in_different_currencies_do_not_derive_purchases_from_each_other() { + let mut conn = conn(); + + write_balance(&mut conn, &balance(1, 20.0)).unwrap(); + write_balance( + &mut conn, + &BalanceSnapshot { + currency: "USD".to_string(), + topped_up_balance: Some(300.0), + ..balance(1, 300.0) + }, + ) + .unwrap(); + + assert!(derive_top_ups(&conn, &deepseek_key()).unwrap().is_empty()); + } + #[test] fn re_observing_a_balance_at_the_same_instant_overwrites() { let mut conn = conn(); From a4832d2abef6fd66d1c87e1efe4c6cc28d828f4f Mon Sep 17 00:00:00 2001 From: JetSquirrel Date: Sun, 30 Aug 2026 01:32:43 +0800 Subject: [PATCH 5/5] Read the dashboard out of the ledger and fix cross-currency totals MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dashboard added AWS dollars to Alibaba Cloud yuan and printed the result with a dollar sign. It also asked each provider for a display-shaped summary and kept a copy in two cache tables, so the numbers on screen were whatever the last API call happened to return. Charges are now read through v_charge_normalized, which converts each one at a rate dated no later than the charge itself. Conversion lives in a view rather than at write time because rates get corrected and the reporting currency is the user's to change; either would otherwise mean rewriting the fact table. Switching currency in Settings replaces a view and rewrites nothing. A charge already in the reporting currency converts at 1.0 without needing a row in the rate table. One whose currency no rate covers keeps a NULL converted amount: it is counted nowhere and reported on the dashboard as an explicit "not included" line, because a total that quietly folds in an unconverted amount at par is worse than one that admits what it is missing. The cache tables go with it (application schema v2). ingest_batch already records when each period was written, so that is what the freshness window checks now. The trend chart reads the daily rows the refresh already stored instead of paying for a third Cost Explorer call, and a source's trend window becomes how far back its own rows are worth charting — Alibaba Cloud's monthly rows get two billing periods rather than seven empty days. With nothing left reading them, get_cost_summary, get_cost_trend and get_cost_data are gone from the trait and from all three sources, along with the parsers and aggregators behind them. A source fetches and normalizes; what the numbers mean afterwards is the ledger's business. The clients also stop carrying an account id and name they no longer use. Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 13 ++ docs/roadmap.md | 51 ++++-- src/cloud/aliyun.rs | 250 +-------------------------- src/cloud/aws.rs | 348 +------------------------------------ src/cloud/deepseek.rs | 79 +-------- src/cloud/mod.rs | 115 ++----------- src/cloud/registry.rs | 23 ++- src/config.rs | 19 ++ src/db.rs | 346 ++++++------------------------------- src/ingest.rs | 69 +++++++- src/ledger/mod.rs | 22 ++- src/ledger/query.rs | 390 ++++++++++++++++++++++++++++++++++++++++++ src/ledger/schema.rs | 74 ++++++++ src/main.rs | 7 +- src/report.rs | 227 ++++++++++++++++++++++++ src/ui/chart.rs | 18 +- src/ui/dashboard.rs | 344 +++++++++++++++++-------------------- src/ui/settings.rs | 56 +++++- 18 files changed, 1154 insertions(+), 1297 deletions(-) create mode 100644 src/ledger/query.rs create mode 100644 src/report.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 446e7ec..b748a1d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -45,6 +45,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 row instead of vanishing - DeepSeek balances are recorded as snapshots, and a rise in the topped-up balance between observations is derived as a `Purchase` +- **The dashboard reads the ledger** (roadmap P0/PR6) + - Totals come from `v_charge_normalized`, which converts each charge at a + rate dated no later than the charge itself, so cross-cloud figures are + in one currency instead of adding dollars to yuan + - Reporting currency is a setting; switching it rebuilds a view and + rewrites nothing + - Charges in a currency no rate covers are reported on the dashboard + rather than being counted at par - **DeepSeek Integration** - DeepSeek API integration for balance queries - Display account balance instead of cost for DeepSeek accounts @@ -52,6 +60,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Support for multiple currencies (CNY, USD) ### Changed +- The response cache tables are gone (application schema v2). A refresh + checks when a period was last ingested, which the ledger already records +- A billing source only fetches and normalizes now; `get_cost_summary` and + `get_cost_trend` are gone, along with the per-call trend fetch — the + trend chart reads rows the refresh already stored - `CloudService` is now `BillingSource`, with `fetch` and `normalize` split apart: the first touches the network and interprets nothing, the second interprets and touches nothing diff --git a/docs/roadmap.md b/docs/roadmap.md index d18cadb..47d351a 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -31,31 +31,39 @@ carry their weight for a personal ledger: ## Where we are -PR1 through PR5 have landed. A source is a registry row rather than an -enum variant; `billing.duckdb` holds `fct_charge`, `ingest_batch`, +**P0 is done.** A source is a registry row rather than an enum variant. +`billing.duckdb` holds `fct_charge`, `ingest_batch`, `fct_balance_snapshot` and `dim_fx_rate` behind a transactional -whole-period write; every source fetches raw payloads to Parquet and +whole-period write. Every source fetches raw payloads to Parquet and normalizes them through a pure function, tested against a recorded -response; and all three sources land as real FOCUS rows. +response. The dashboard reads `v_charge_normalized`, so every figure on it +is in one currency, converted per charge at a rate dated no later than the +charge itself. -What is left in P0 is the read path (PR6). Nothing reads the ledger yet: -the dashboard still calls each provider's display method and caches the -result in the two per-account tables in `cloudbridge.duckdb`. Those go -away with PR6, not before — they are the only thing feeding the UI until -it reads through `v_charge_normalized`. +The acceptance test holds: all three sources land in one `fct_charge` +table, `SELECT sum(billed_cost_base) FROM v_charge_normalized WHERE +billing_period = ?` is the cross-cloud total, and a repeated ingest of an +unchanged bill produces identical rows. -One of the three structural problems is still open: +All three structural problems are closed: 1. ~~**`CloudProvider` is a compile-time enum**~~ — replaced by the source - registry in PR1. An unrecognized source id is now skipped with a - warning instead of being silently read as AWS. -2. **Amounts are summed across currencies.** The dashboard total adds AWS - USD to Alibaba Cloud CNY and shows the result as one number. + registry in PR1. An unrecognized source id is skipped with a warning + instead of being silently read as AWS. +2. ~~**Amounts are summed across currencies.**~~ — fixed in PR6. Charges + are stored in the currency they were billed in and converted in a view, + so a rate correction or a change of reporting currency costs nothing. + A charge in a currency no rate covers is counted nowhere and reported + on the dashboard rather than quietly folded in at par. 3. ~~**`fetch` and `normalize` are fused.**~~ — split in PR3. `fetch` persists what the provider returned and interprets nothing; `normalize` interprets and touches nothing, so a mapping fix replays payloads already on disk instead of paying Cost Explorer again. +What P0 did *not* do, and P1 owns: instance-level detail. Alibaba Cloud's +bill overview is one row per product per month, so its trend chart is as +coarse as its source data. + ## P0 — FOCUS normalization The one-sentence acceptance test for the whole phase: @@ -193,7 +201,7 @@ was not witnessed being paid for. The display path still reports the balance as `current_month_cost`; that is PR6's to fix, along with everything else the dashboard reads. -### PR6 · Read through views, fix cross-currency +### PR6 · Read through views, fix cross-currency — landed Amounts are stored in their original currency. Conversion happens in a view, never at write time, because rates get corrected and the user may @@ -212,6 +220,19 @@ ASOF LEFT JOIN dim_fx_rate f Ships with a built-in rate table and a reporting-currency setting. This is where the cross-currency total is actually fixed. +As landed, the view also carries `effective_cost_base` and the `fx_rate` it +used, and a charge already in the reporting currency converts at 1.0 +without needing a row in the rate table. A charge whose currency no rate +covers keeps a NULL `billed_cost_base`: it is left out of every converted +total and counted separately, so the dashboard can say how many charges it +is not showing rather than under-reporting silently. + +The freshness window moved into the ledger with the same change: +`ingest_batch` records when each period was last written, which is what a +refresh checks. The two response cache tables are gone, and so are +`get_cost_summary` and `get_cost_trend` — a source now fetches and +normalizes, and nothing else. + ## P1 - **Bill file export channel (S3 / OSS + Parquet).** Replaces per-request diff --git a/src/cloud/aliyun.rs b/src/cloud/aliyun.rs index 6d1aaba..6aaffe6 100644 --- a/src/cloud/aliyun.rs +++ b/src/cloud/aliyun.rs @@ -8,33 +8,20 @@ use sha1::Sha1; use std::collections::BTreeMap; use super::raw::RawPart; -use super::{ - BillingPeriod, BillingSource, CostData, CostSummary, Normalized, RawBatch, ServiceCost, - SourceId, -}; +use super::{BillingPeriod, BillingSource, Normalized, RawBatch}; use crate::ledger::{Charge, ChargeCategory}; type HmacSha1 = Hmac; /// Alibaba Cloud service pub struct AliyunCloudService { - account_id: String, - account_name: String, access_key_id: String, access_key_secret: String, } impl AliyunCloudService { - pub fn new( - account_id: String, - account_name: String, - access_key_id: String, - access_key_secret: String, - _region: Option, - ) -> Self { + pub fn new(access_key_id: String, access_key_secret: String, _region: Option) -> Self { Self { - account_id, - account_name, access_key_id, access_key_secret, } @@ -181,45 +168,6 @@ impl AliyunCloudService { serde_json::from_str(&body) .map_err(|e| anyhow!("Failed to parse bill overview: {} - {}", e, body)) } - - /// Query instance bill (daily details) - fn describe_instance_bill( - &self, - billing_cycle: &str, - granularity: &str, - ) -> Result { - let body = self.call_bss_api( - "DescribeInstanceBill", - &[ - ("BillingCycle", billing_cycle), - ("Granularity", granularity), // DAILY or MONTHLY - ("MaxResults", "300"), - ], - )?; - - serde_json::from_str(&body) - .map_err(|e| anyhow!("Failed to parse instance bill: {} - {}", e, body)) - } - - /// Query instance bill for a specific date (daily granularity requires BillingDate) - fn describe_instance_bill_by_date( - &self, - billing_cycle: &str, - billing_date: &str, - ) -> Result { - let body = self.call_bss_api( - "DescribeInstanceBill", - &[ - ("BillingCycle", billing_cycle), - ("BillingDate", billing_date), - ("Granularity", "DAILY"), - ("MaxResults", "300"), - ], - )?; - - serde_json::from_str(&body) - .map_err(|e| anyhow!("Failed to parse instance bill: {} - {}", e, body)) - } } impl BillingSource for AliyunCloudService { @@ -251,162 +199,6 @@ impl BillingSource for AliyunCloudService { fn normalize(&self, batch: &RawBatch) -> Result { normalize(batch) } - - fn get_cost_data(&self, start_date: &str, end_date: &str) -> Result> { - // Alibaba Cloud queries by month, extract year-month - let billing_cycle = &start_date[..7]; // YYYY-MM - - let response = self.describe_instance_bill(billing_cycle, "DAILY")?; - - let mut costs = Vec::new(); - if let Some(items) = response.data.and_then(|d| d.items) { - for item in items { - let date = item.billing_date.unwrap_or_default(); - // Filter by date range - if date.as_str() >= start_date && date.as_str() <= end_date { - costs.push(CostData { - account_id: self.account_id.clone(), - date, - service: item.product_name.unwrap_or_else(|| "Unknown".to_string()), - amount: item.pretax_amount.unwrap_or(0.0), - currency: "CNY".to_string(), - }); - } - } - } - - Ok(costs) - } - - fn get_cost_summary(&self) -> Result { - let now = Utc::now(); - - // Current month - let current_month = format!("{}-{:02}", now.year(), now.month()); - // Last month - let last_month_date = now - chrono::Duration::days(now.day() as i64 + 1); - let last_month = format!("{}-{:02}", last_month_date.year(), last_month_date.month()); - - // Query current month bill overview - let current_overview = self.query_bill_overview(¤t_month)?; - let last_overview = self.query_bill_overview(&last_month)?; - - // Parse current month costs - let (current_month_cost, current_month_details) = parse_bill_overview(¤t_overview); - let (last_month_cost, last_month_details) = parse_bill_overview(&last_overview); - - // Calculate month-over-month change - let month_over_month_change = if last_month_cost > 0.0 { - ((current_month_cost - last_month_cost) / last_month_cost) * 100.0 - } else if current_month_cost > 0.0 { - 100.0 - } else { - 0.0 - }; - - Ok(CostSummary { - account_id: self.account_id.clone(), - account_name: self.account_name.clone(), - source_id: SourceId::from("Aliyun"), - current_month_cost, - last_month_cost, - currency: "CNY".to_string(), - month_over_month_change, - current_month_details, - last_month_details, - }) - } - - fn get_cost_trend(&self, start_date: &str, end_date: &str) -> Result { - // Aggregate costs by date - let mut daily_map: std::collections::HashMap = - std::collections::HashMap::new(); - - // Use chrono to iterate through each day in the date range - use chrono::NaiveDate; - - let start = NaiveDate::parse_from_str(start_date, "%Y-%m-%d") - .map_err(|e| anyhow!("Invalid start date: {}", e))?; - let end = NaiveDate::parse_from_str(end_date, "%Y-%m-%d") - .map_err(|e| anyhow!("Invalid end date: {}", e))?; - - let mut current = start; - while current < end { - let date_str = current.format("%Y-%m-%d").to_string(); - let billing_cycle = current.format("%Y-%m").to_string(); - - match self.describe_instance_bill_by_date(&billing_cycle, &date_str) { - Ok(response) => { - if let Some(items) = response.data.and_then(|d| d.items) { - let mut day_total = 0.0; - for item in items { - let amount = item.pretax_amount.unwrap_or(0.0); - day_total += amount; - } - if day_total > 0.0 { - daily_map.insert(date_str.clone(), day_total); - } - } - } - Err(e) => { - tracing::warn!("Failed to query bill for {}: {}", date_str, e); - } - } - - current += chrono::Duration::days(1); - } - - // Convert to sorted list - let mut daily_costs: Vec = daily_map - .into_iter() - .map(|(date, amount)| super::DailyCost { date, amount }) - .collect(); - - daily_costs.sort_by(|a, b| a.date.cmp(&b.date)); - - Ok(super::CostTrend { - account_id: self.account_id.clone(), - currency: "CNY".to_string(), - daily_costs, - }) - } -} - -/// Parse bill overview -fn parse_bill_overview(response: &BillOverviewResponse) -> (f64, Vec) { - let mut total_cost = 0.0; - let mut details = Vec::new(); - - if let Some(data) = &response.data { - if let Some(items_wrapper) = &data.items { - if let Some(items) = &items_wrapper.item { - for item in items { - let amount = item.pretax_amount.unwrap_or(0.0); - total_cost += amount; - - if amount > 0.0 { - details.push(ServiceCost { - service: item - .product_name - .clone() - .unwrap_or_else(|| "Unknown".to_string()), - amount, - currency: "CNY".to_string(), - }); - } - } - } - } - } - - // Sort by amount in descending order - details.sort_by(|a, b| { - b.amount - .partial_cmp(&a.amount) - .unwrap_or(std::cmp::Ordering::Equal) - }); - - (total_cost, details) } /// Name the bill overview payload is stored under in a raw batch. @@ -603,44 +395,6 @@ impl BillOverviewItem { } } -#[derive(Debug, Deserialize)] -#[serde(rename_all = "PascalCase")] -#[allow(dead_code)] -struct InstanceBillResponse { - request_id: Option, - success: Option, - code: Option, - message: Option, - data: Option, -} - -#[derive(Debug, Deserialize)] -#[serde(rename_all = "PascalCase")] -#[allow(dead_code)] -struct InstanceBillData { - billing_cycle: Option, - account_id: Option, - total_count: Option, - next_token: Option, - max_results: Option, - /// DescribeInstanceBill returns Items as a direct array, not wrapped in an object - items: Option>, -} - -#[derive(Debug, Deserialize)] -#[serde(rename_all = "PascalCase")] -#[allow(dead_code)] -struct InstanceBillItem { - billing_date: Option, - product_code: Option, - product_name: Option, - instance_id: Option, - pretax_amount: Option, - #[serde(rename = "PretaxGrossAmount")] - pretax_gross_amount: Option, - currency: Option, -} - #[cfg(test)] mod tests { use super::*; diff --git a/src/cloud/aws.rs b/src/cloud/aws.rs index 9d54050..a55156d 100644 --- a/src/cloud/aws.rs +++ b/src/cloud/aws.rs @@ -1,37 +1,27 @@ //! AWS Cloud Service Implementation - Using ureq + AWS Signature V4 use anyhow::{anyhow, Result}; -use chrono::{DateTime, Datelike, Utc}; +use chrono::{DateTime, Utc}; use hmac::{Hmac, Mac}; use serde::Deserialize; use sha2::{Digest, Sha256}; use super::raw::RawPart; -use super::{BillingPeriod, BillingSource, CostData, CostSummary, Normalized, RawBatch, SourceId}; +use super::{BillingPeriod, BillingSource, Normalized, RawBatch}; use crate::ledger::{Charge, ChargeCategory}; type HmacSha256 = Hmac; /// AWS Cloud Service pub struct AwsCloudService { - account_id: String, - account_name: String, access_key_id: String, secret_access_key: String, region: String, } impl AwsCloudService { - pub fn new( - account_id: String, - account_name: String, - access_key_id: String, - secret_access_key: String, - region: Option, - ) -> Self { + pub fn new(access_key_id: String, secret_access_key: String, region: Option) -> Self { Self { - account_id, - account_name, access_key_id, secret_access_key, region: region.unwrap_or_else(|| "us-east-1".to_string()), @@ -234,18 +224,6 @@ impl AwsCloudService { Ok(body) } - /// Call Cost Explorer API - fn call_cost_explorer(&self, start_date: &str, end_date: &str) -> Result> { - let body = self.cost_and_usage_raw(&cost_and_usage_request(start_date, end_date, true))?; - parse_cost_explorer_response(&body, &self.account_id, &self.account_name) - } - - /// Call Cost Explorer API to get daily costs (not grouped by service, for trend charts) - fn call_cost_explorer_daily(&self, start_date: &str, end_date: &str) -> Result> { - let body = self.cost_and_usage_raw(&cost_and_usage_request(start_date, end_date, false))?; - parse_daily_cost_response(&body, &self.account_id) - } - /// Sign with specified region (for services like Cost Explorer that are only available in specific regions) #[allow(clippy::too_many_arguments)] fn sign_request_with_region( @@ -338,34 +316,6 @@ const UNIT_NOT_APPLICABLE: &str = "N/A"; const DIMENSION_SERVICE: &str = "SERVICE"; const DIMENSION_RECORD_TYPE: &str = "RECORD_TYPE"; -/// The GetCostAndUsage request body for the display path. -/// -/// `group_by_service` off is the trend query: one total per day, which is -/// a cheaper response than summing the grouped one client-side. -fn cost_and_usage_request( - start_date: &str, - end_date: &str, - group_by_service: bool, -) -> serde_json::Value { - let mut request = serde_json::json!({ - "TimePeriod": { - "Start": start_date, - "End": end_date - }, - "Granularity": "DAILY", - "Metrics": [METRIC_UNBLENDED] - }); - - if group_by_service { - request["GroupBy"] = serde_json::json!([{ - "Type": "DIMENSION", - "Key": DIMENSION_SERVICE - }]); - } - - request -} - /// The GetCostAndUsage request the ledger is built from. /// /// All three metrics ride in one request: Cost Explorer bills per request, @@ -599,152 +549,6 @@ fn parse_sts_response(xml: &str) -> Result { }) } -/// Parse Cost Explorer JSON response -fn parse_cost_explorer_response( - json: &str, - account_id: &str, - _account_name: &str, -) -> Result> { - #[derive(Deserialize)] - struct CeResponse { - #[serde(rename = "ResultsByTime")] - results_by_time: Option>, - } - - #[derive(Deserialize)] - struct TimeResult { - #[serde(rename = "TimePeriod")] - time_period: TimePeriod, - #[serde(rename = "Groups")] - groups: Option>, - } - - #[derive(Deserialize)] - struct TimePeriod { - #[serde(rename = "Start")] - start: String, - } - - #[derive(Deserialize)] - struct CostGroup { - #[serde(rename = "Keys")] - keys: Vec, - #[serde(rename = "Metrics")] - metrics: CostMetrics, - } - - #[derive(Deserialize)] - struct CostMetrics { - #[serde(rename = "UnblendedCost")] - unblended_cost: CostAmount, - } - - #[derive(Deserialize)] - struct CostAmount { - #[serde(rename = "Amount")] - amount: String, - #[serde(rename = "Unit")] - unit: String, - } - - let response: CeResponse = serde_json::from_str(json)?; - - let mut cost_data = Vec::new(); - if let Some(results) = response.results_by_time { - tracing::info!( - "Cost Explorer returned data for {} time periods", - results.len() - ); - for result in results { - if let Some(groups) = result.groups { - for group in groups { - let service_name = group.keys.first().cloned().unwrap_or_default(); - let amount: f64 = group.metrics.unblended_cost.amount.parse().unwrap_or(0.0); - let currency = group.metrics.unblended_cost.unit; - - if amount > 0.0 { - tracing::debug!("Service {}: {} {}", service_name, amount, currency); - cost_data.push(CostData { - account_id: account_id.to_string(), - date: result.time_period.start.clone(), - service: service_name, - amount, - currency, - }); - } - } - } - } - } - - tracing::info!("Parsed {} cost data records", cost_data.len()); - Ok(cost_data) -} - -/// Parse Cost Explorer daily cost response (not grouped by service) -fn parse_daily_cost_response(json: &str, account_id: &str) -> Result> { - #[derive(Deserialize)] - struct CeResponse { - #[serde(rename = "ResultsByTime")] - results_by_time: Option>, - } - - #[derive(Deserialize)] - struct TimeResult { - #[serde(rename = "TimePeriod")] - time_period: TimePeriod, - #[serde(rename = "Total")] - total: Option, - } - - #[derive(Deserialize)] - struct TimePeriod { - #[serde(rename = "Start")] - start: String, - } - - #[derive(Deserialize)] - struct CostMetrics { - #[serde(rename = "UnblendedCost")] - unblended_cost: CostAmount, - } - - #[derive(Deserialize)] - struct CostAmount { - #[serde(rename = "Amount")] - amount: String, - #[serde(rename = "Unit")] - unit: String, - } - - let response: CeResponse = serde_json::from_str(json)?; - - let mut cost_data = Vec::new(); - if let Some(results) = response.results_by_time { - tracing::debug!( - "Daily cost response returned {} time periods", - results.len() - ); - for result in results { - if let Some(total) = result.total { - let amount: f64 = total.unblended_cost.amount.parse().unwrap_or(0.0); - let currency = total.unblended_cost.unit; - - cost_data.push(CostData { - account_id: account_id.to_string(), - date: result.time_period.start.clone(), - service: "Total".to_string(), - amount, - currency, - }); - } - } - } - - tracing::debug!("Parsed {} daily cost data records", cost_data.len()); - Ok(cost_data) -} - impl BillingSource for AwsCloudService { fn validate_credentials(&self) -> Result { match self.call_sts_get_caller_identity() { @@ -780,142 +584,6 @@ impl BillingSource for AwsCloudService { fn normalize(&self, batch: &RawBatch) -> Result { normalize(batch) } - - fn get_cost_data(&self, start_date: &str, end_date: &str) -> Result> { - self.call_cost_explorer(start_date, end_date) - } - - fn get_cost_summary(&self) -> Result { - let now = Utc::now(); - let current_month_start = format!("{}-{:02}-01", now.year(), now.month()); - let current_month_end = format!("{}-{:02}-{:02}", now.year(), now.month(), now.day()); - - // Last month - let last_month = if now.month() == 1 { - chrono::NaiveDate::from_ymd_opt(now.year() - 1, 12, 1).unwrap() - } else { - chrono::NaiveDate::from_ymd_opt(now.year(), now.month() - 1, 1).unwrap() - }; - let last_month_start = format!("{}-{:02}-01", last_month.year(), last_month.month()); - let last_month_end = current_month_start.clone(); - - // Get current month costs - let current_costs = self.get_cost_data(¤t_month_start, ¤t_month_end)?; - let current_month_cost: f64 = current_costs.iter().map(|c| c.amount).sum(); - tracing::info!( - "Current month cost: {} USD ({} records)", - current_month_cost, - current_costs.len() - ); - - // Get last month costs - let last_costs = self.get_cost_data(&last_month_start, &last_month_end)?; - let last_month_cost: f64 = last_costs.iter().map(|c| c.amount).sum(); - tracing::info!( - "Last month cost: {} USD ({} records)", - last_month_cost, - last_costs.len() - ); - - // Calculate month-over-month change - let month_over_month_change = if last_month_cost > 0.0 { - ((current_month_cost - last_month_cost) / last_month_cost) * 100.0 - } else { - 0.0 - }; - - let currency = current_costs - .first() - .map(|c| c.currency.clone()) - .unwrap_or_else(|| "USD".to_string()); - - // Aggregate current month costs by service - let current_month_details = aggregate_costs_by_service(¤t_costs); - // Aggregate last month costs by service - let last_month_details = aggregate_costs_by_service(&last_costs); - - Ok(CostSummary { - account_id: self.account_id.clone(), - account_name: self.account_name.clone(), - source_id: SourceId::from("AWS"), - current_month_cost, - last_month_cost, - currency, - month_over_month_change, - current_month_details, - last_month_details, - }) - } - - fn get_cost_trend(&self, start_date: &str, end_date: &str) -> Result { - tracing::info!("Getting cost trend: {} to {}", start_date, end_date); - - // Call Cost Explorer API to get daily costs - let cost_data = self.call_cost_explorer_daily(start_date, end_date)?; - - // Aggregate daily costs - let (daily_costs, currency) = aggregate_daily_costs(&cost_data); - - Ok(super::CostTrend { - account_id: self.account_id.clone(), - currency, - daily_costs, - }) - } -} - -/// Aggregate cost data by service -fn aggregate_costs_by_service(costs: &[CostData]) -> Vec { - use std::collections::HashMap; - - let mut service_map: HashMap = HashMap::new(); - let mut currency = "USD".to_string(); - - for cost in costs { - *service_map.entry(cost.service.clone()).or_insert(0.0) += cost.amount; - currency = cost.currency.clone(); - } - - let mut result: Vec = service_map - .into_iter() - .map(|(service, amount)| super::ServiceCost { - service, - amount, - currency: currency.clone(), - }) - .collect(); - - // Sort by amount in descending order - result.sort_by(|a, b| { - b.amount - .partial_cmp(&a.amount) - .unwrap_or(std::cmp::Ordering::Equal) - }); - - result -} - -/// Aggregate daily costs by date, returns (daily cost list, currency) -fn aggregate_daily_costs(costs: &[CostData]) -> (Vec, String) { - use std::collections::HashMap; - - let mut date_map: HashMap = HashMap::new(); - let mut currency = "USD".to_string(); - - for cost in costs { - *date_map.entry(cost.date.clone()).or_insert(0.0) += cost.amount; - currency = cost.currency.clone(); - } - - let mut result: Vec = date_map - .into_iter() - .map(|(date, amount)| super::DailyCost { date, amount }) - .collect(); - - // Sort by date in ascending order - result.sort_by(|a, b| a.date.cmp(&b.date)); - - (result, currency) } #[cfg(test)] @@ -1111,16 +779,6 @@ mod tests { assert!(normalize(&batch).is_err()); } - #[test] - fn the_trend_request_asks_for_totals_and_the_summary_request_for_services() { - let grouped = cost_and_usage_request("2026-08-01", "2026-09-01", true); - assert_eq!(grouped["GroupBy"][0]["Key"], "SERVICE"); - assert_eq!(grouped["Granularity"], "DAILY"); - - let totals = cost_and_usage_request("2026-08-01", "2026-09-01", false); - assert!(totals.get("GroupBy").is_none()); - } - #[test] fn the_ledger_request_carries_every_metric_in_one_call() { let request = ledger_request("2026-08-01", "2026-09-01"); diff --git a/src/cloud/deepseek.rs b/src/cloud/deepseek.rs index c942ffb..98b5a61 100644 --- a/src/cloud/deepseek.rs +++ b/src/cloud/deepseek.rs @@ -4,10 +4,7 @@ use anyhow::{anyhow, Result}; use serde::Deserialize; use super::raw::RawPart; -use super::{ - BillingPeriod, BillingSource, CostData, CostSummary, CostTrend, Normalized, RawBatch, - ServiceCost, SourceId, -}; +use super::{BillingPeriod, BillingSource, Normalized, RawBatch}; use crate::ledger::BalanceSnapshot; /// DeepSeek balance info @@ -40,24 +37,16 @@ const PART_BALANCE: &str = "balance"; /// DeepSeek service pub struct DeepSeekService { - account_id: String, - account_name: String, api_key: String, } impl DeepSeekService { pub fn new( - account_id: String, - account_name: String, api_key: String, _secret: String, // Not used for DeepSeek, but kept for interface consistency _region: Option, // Not used for DeepSeek ) -> Self { - Self { - account_id, - account_name, - api_key, - } + Self { api_key } } /// Ask the balance endpoint and return the response body unchanged. @@ -142,70 +131,6 @@ impl BillingSource for DeepSeekService { fn normalize(&self, batch: &RawBatch) -> Result { normalize(batch) } - - fn get_cost_data(&self, _start_date: &str, _end_date: &str) -> Result> { - // DeepSeek doesn't provide detailed cost history, return empty - Ok(vec![]) - } - - fn get_cost_summary(&self) -> Result { - let balance = self.get_balance()?; - - // Prefer CNY balance first, then USD, then fallback to first - let balance_info = balance - .balance_infos - .iter() - .find(|b| b.currency == "CNY") - .or_else(|| balance.balance_infos.iter().find(|b| b.currency == "USD")) - .or_else(|| balance.balance_infos.first()) - .ok_or_else(|| anyhow!("No balance info found"))?; - - let total: f64 = balance_info.total_balance.parse().unwrap_or(0.0); - let granted: f64 = balance_info.granted_balance.parse().unwrap_or(0.0); - let topped_up: f64 = balance_info.topped_up_balance.parse().unwrap_or(0.0); - - // Build service details showing balance breakdown - let mut details = Vec::new(); - if granted > 0.0 { - details.push(ServiceCost { - service: "Granted Balance".to_string(), - amount: granted, - currency: balance_info.currency.clone(), - }); - } - if topped_up > 0.0 { - details.push(ServiceCost { - service: "Topped-up Balance".to_string(), - amount: topped_up, - currency: balance_info.currency.clone(), - }); - } - - // For DeepSeek, we show balance instead of cost - // current_month_cost = remaining balance (positive) - // last_month_cost = 0 (no historical data) - Ok(CostSummary { - account_id: self.account_id.clone(), - account_name: self.account_name.clone(), - source_id: SourceId::from("DeepSeek"), - current_month_cost: total, - last_month_cost: 0.0, - currency: balance_info.currency.clone(), - month_over_month_change: 0.0, // No comparison for balance - current_month_details: details, - last_month_details: vec![], - }) - } - - fn get_cost_trend(&self, _start_date: &str, _end_date: &str) -> Result { - // DeepSeek doesn't provide daily usage history - // Return empty trend with current balance as single point - Ok(CostTrend { - account_id: self.account_id.clone(), - currency: "USD".to_string(), - daily_costs: vec![], - }) - } } #[cfg(test)] diff --git a/src/cloud/mod.rs b/src/cloud/mod.rs index f2a0994..a3ed3ae 100644 --- a/src/cloud/mod.rs +++ b/src/cloud/mod.rs @@ -18,13 +18,13 @@ pub use registry::{SourceDescriptor, SourceId}; /// Accounts like that are filtered out on load, so this is a backstop. const UNKNOWN_SOURCE: &str = "Unknown"; -/// The credentials and identity a [`CloudService`] is built from. +/// The credentials a [`BillingSource`] is built from. /// /// Bundled into one struct so [`SourceDescriptor::build`] can be a plain -/// function pointer. +/// function pointer. Deliberately no account id or name: a client +/// authenticates and fetches, and which account the result is filed under +/// is the ingest's business, not its own. pub struct SourceContext { - pub account_id: String, - pub account_name: String, pub access_key_id: String, pub secret_access_key: String, pub region: Option, @@ -69,8 +69,6 @@ impl CloudAccount { /// source's default region filled in when the account stored none. pub fn context(&self, descriptor: &SourceDescriptor) -> SourceContext { SourceContext { - account_id: self.id.clone(), - account_name: self.name.clone(), access_key_id: self.access_key_id.clone(), secret_access_key: self.secret_access_key.clone(), region: descriptor.region_or_default(self.region.clone()), @@ -78,93 +76,6 @@ impl CloudAccount { } } -/// Cost data -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct CostData { - /// Account ID - pub account_id: String, - /// Date - pub date: String, - /// Service name - pub service: String, - /// Cost amount - pub amount: f64, - /// Currency - pub currency: String, -} - -/// Cost summary -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct CostSummary { - /// Account ID - pub account_id: String, - /// Account name - pub account_name: String, - /// Billing source this summary came from - pub source_id: SourceId, - /// Current month cost - pub current_month_cost: f64, - /// Last month cost - pub last_month_cost: f64, - /// Currency - pub currency: String, - /// Month-over-month change (percentage) - pub month_over_month_change: f64, - /// Current month service cost details - pub current_month_details: Vec, - /// Last month service cost details - pub last_month_details: Vec, -} - -impl CostSummary { - fn descriptor(&self) -> Option<&'static SourceDescriptor> { - self.source_id.descriptor() - } - - /// Short label for the source, for badges. - pub fn short_name(&self) -> &'static str { - self.descriptor().map_or(UNKNOWN_SOURCE, |s| s.short_name) - } - - /// Whether this is a point-in-time balance rather than a period cost. - /// The dashboard lists the two kinds in separate sections and labels - /// their amounts differently. - pub fn is_snapshot(&self) -> bool { - self.descriptor().is_some_and(SourceDescriptor::is_snapshot) - } -} - -/// Service cost detail -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct ServiceCost { - /// Service name - pub service: String, - /// Cost amount - pub amount: f64, - /// Currency - pub currency: String, -} - -/// Daily cost data (for chart display) -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct DailyCost { - /// Date (YYYY-MM-DD format) - pub date: String, - /// Daily cost - pub amount: f64, -} - -/// Cost trend data -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct CostTrend { - /// Account ID - pub account_id: String, - /// Currency - pub currency: String, - /// Daily costs list - pub daily_costs: Vec, -} - /// Budget information // TODO(v0.2.0): drop this allow once the budget UI is wired up #[allow(dead_code)] @@ -225,6 +136,15 @@ impl BillingPeriod { Self::new(instant.year(), instant.month()) } + /// The period before this one. + pub fn previous(&self) -> Self { + if self.month == 1 { + Self::new(self.year - 1, 12) + } else { + Self::new(self.year, self.month - 1) + } + } + /// `YYYY-MM`, as stored in `billing_period` and in the raw path. pub fn label(&self) -> String { format!("{:04}-{:02}", self.year, self.month) @@ -275,13 +195,4 @@ pub trait BillingSource: Send + Sync { /// Turn a fetched batch into ledger rows. Pure: no clock, no network, /// no database — everything it needs is in the batch. fn normalize(&self, batch: &RawBatch) -> Result; - - /// Get cost data - fn get_cost_data(&self, start_date: &str, end_date: &str) -> Result>; - - /// Get cost summary - fn get_cost_summary(&self) -> Result; - - /// Get cost trend (daily costs) - fn get_cost_trend(&self, start_date: &str, end_date: &str) -> Result; } diff --git a/src/cloud/registry.rs b/src/cloud/registry.rs index 7760bbb..d550009 100644 --- a/src/cloud/registry.rs +++ b/src/cloud/registry.rs @@ -18,9 +18,9 @@ use super::{BillingSource, SourceContext}; /// What a source reports, and therefore how it can be displayed. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Reporting { - /// Cost accrued over a period. `trend_window_days` is how far back a - /// daily trend is worth requesting — Alibaba Cloud needs one API call - /// per day, so it gets a shorter window than AWS. + /// Cost accrued over a period. `trend_window_days` is how far back the + /// trend chart reads the ledger — a window is only worth charting as + /// far as the source's own rows are detailed enough to fill it. Periodic { trend_window_days: i64 }, /// A point-in-time balance. There is no period cost and no history to /// chart. @@ -107,8 +107,8 @@ impl SourceDescriptor { region.or_else(|| self.default_region.map(str::to_string)) } - /// Days of daily trend worth requesting, or `None` for a source that has - /// no history to chart. + /// Days of trend worth charting, or `None` for a source that has no + /// history to chart. pub fn trend_window_days(&self) -> Option { match self.reporting { Reporting::Periodic { trend_window_days } => Some(trend_window_days), @@ -135,8 +135,6 @@ static SOURCES: &[SourceDescriptor] = &[ }, build: |ctx| { Box::new(AwsCloudService::new( - ctx.account_id, - ctx.account_name, ctx.access_key_id, ctx.secret_access_key, ctx.region, @@ -150,14 +148,15 @@ static SOURCES: &[SourceDescriptor] = &[ access_key_label: "AccessKey ID", secret_key_label: Some("AccessKey Secret"), default_region: Some("cn-hangzhou"), - // One API call per day, so a shorter window than AWS. + // QueryBillOverview reports one row per product per month, so a + // day-level chart has nothing finer to show: the window covers two + // billing periods rather than two weeks of empty days. Daily + // detail arrives with the bill export channel (P1). reporting: Reporting::Periodic { - trend_window_days: 7, + trend_window_days: 62, }, build: |ctx| { Box::new(AliyunCloudService::new( - ctx.account_id, - ctx.account_name, ctx.access_key_id, ctx.secret_access_key, ctx.region, @@ -174,8 +173,6 @@ static SOURCES: &[SourceDescriptor] = &[ reporting: Reporting::Snapshot, build: |ctx| { Box::new(DeepSeekService::new( - ctx.account_id, - ctx.account_name, ctx.access_key_id, ctx.secret_access_key, ctx.region, diff --git a/src/config.rs b/src/config.rs index bc1a00a..7b04154 100644 --- a/src/config.rs +++ b/src/config.rs @@ -9,6 +9,12 @@ use std::path::PathBuf; /// Default data refresh interval, in minutes pub const DEFAULT_REFRESH_INTERVAL_MINUTES: u32 = 60; +/// Currency every amount is shown in until the user picks another. +pub const DEFAULT_REPORTING_CURRENCY: &str = "USD"; + +/// Currencies the built-in rate table can convert between. +pub const SUPPORTED_REPORTING_CURRENCIES: &[&str] = &["USD", "CNY"]; + /// Application configuration #[derive(Debug, Clone, Serialize, Deserialize)] pub struct AppConfig { @@ -21,6 +27,15 @@ pub struct AppConfig { /// Persisted but not acted on yet: nothing schedules a refresh from it, /// so it has no Settings UI either. Wire both up together. pub refresh_interval_minutes: u32, + /// Currency every amount is converted to for display. Charges are + /// stored in the currency they were billed in; this only changes the + /// view they are read through. + #[serde(default = "default_reporting_currency")] + pub reporting_currency: String, +} + +fn default_reporting_currency() -> String { + DEFAULT_REPORTING_CURRENCY.to_string() } impl Default for AppConfig { @@ -29,6 +44,7 @@ impl Default for AppConfig { encryption_key: None, theme: ThemeConfig::default(), refresh_interval_minutes: DEFAULT_REFRESH_INTERVAL_MINUTES, + reporting_currency: default_reporting_currency(), } } } @@ -96,6 +112,9 @@ pub fn load_config() -> Result { if config.refresh_interval_minutes == 0 { config.refresh_interval_minutes = DEFAULT_REFRESH_INTERVAL_MINUTES; } + if config.reporting_currency.is_empty() { + config.reporting_currency = default_reporting_currency(); + } Ok(config) } else { // Return default config diff --git a/src/db.rs b/src/db.rs index 76dc286..84e32fd 100644 --- a/src/db.rs +++ b/src/db.rs @@ -7,25 +7,20 @@ //! ledger is the record that has to survive. use anyhow::Result; -use chrono::{DateTime, Duration, Utc}; +use chrono::{DateTime, Utc}; use duckdb::{params, Connection}; use std::sync::{Arc, Mutex}; -use crate::cloud::{ - BudgetInfo, BudgetStatus, CloudAccount, CostSummary, CostTrend, DailyCost, ServiceCost, - SourceId, -}; +use crate::cloud::{BillingPeriod, BudgetInfo, BudgetStatus, CloudAccount, SourceId}; use crate::config::get_database_path; use crate::crypto::get_crypto_manager; +use crate::ledger::{query, PeriodKey}; use crate::secret_store; lazy_static::lazy_static! { static ref DB_CONNECTION: Arc>> = Arc::new(Mutex::new(None)); } -/// Cache time-to-live (hours) -const CACHE_TTL_HOURS: i64 = 6; - /// Schema version of the application-state database. /// /// v1 is the first version to be recorded at all: it splits the billing @@ -33,7 +28,11 @@ const CACHE_TTL_HOURS: i64 = 6; /// `cost_data` table, renames `provider` to `source_id` now that a source /// is a registry row rather than an enum variant, and drops the credential /// columns for good — secrets live in the OS keyring. -const APP_SCHEMA_VERSION: i32 = 1; +/// +/// v2 drops the two response caches. The dashboard reads the ledger now, +/// which records when each period was ingested, so a separate copy of +/// display-shaped API responses has nothing left to do. +const APP_SCHEMA_VERSION: i32 = 2; /// Initialize database pub fn init_database() -> Result<()> { @@ -60,11 +59,18 @@ fn prepare_schema(conn: &Connection) -> Result<()> { "#, )?; - if current_schema_version(conn)? < 1 { + let version = current_schema_version(conn)?; + if version < 1 { migrate_to_v1(conn)?; } + if version < 2 { + conn.execute_batch( + "DROP TABLE IF EXISTS cost_summary_cache; + DROP TABLE IF EXISTS cost_trend_cache;", + )?; + } - create_v1_tables(conn)?; + create_tables(conn)?; conn.execute( "INSERT OR REPLACE INTO schema_version (version, applied_at) VALUES (?, ?)", @@ -74,13 +80,13 @@ fn prepare_schema(conn: &Connection) -> Result<()> { Ok(()) } -/// The v1 shape. Anything the migration already rebuilt is left alone. +/// The current shape. Anything a migration already rebuilt is left alone. /// /// Tables are declared without foreign keys: DuckDB will not drop or alter a /// table another table points at, which is what makes a rebuild like /// [`rebuild_accounts_v1`] necessary in the first place. `delete_account` /// cleans up dependants instead. -fn create_v1_tables(conn: &Connection) -> Result<()> { +fn create_tables(conn: &Connection) -> Result<()> { conn.execute_batch( r#" CREATE TABLE IF NOT EXISTS cloud_accounts ( @@ -104,29 +110,7 @@ fn create_v1_tables(conn: &Connection) -> Result<()> { updated_at VARCHAR NOT NULL ); - -- The two cache tables below hold display-shaped API responses, not - -- billing facts. They go away in PR5, once every source normalizes - -- into fct_charge and the dashboard reads through the ledger. - CREATE TABLE IF NOT EXISTS cost_summary_cache ( - account_id VARCHAR PRIMARY KEY, - current_month_cost DOUBLE NOT NULL, - last_month_cost DOUBLE NOT NULL, - currency VARCHAR NOT NULL, - month_over_month_change DOUBLE NOT NULL, - current_month_details TEXT, - last_month_details TEXT, - cached_at VARCHAR NOT NULL - ); - - CREATE TABLE IF NOT EXISTS cost_trend_cache ( - account_id VARCHAR NOT NULL, - date VARCHAR NOT NULL, - amount DOUBLE NOT NULL, - currency VARCHAR NOT NULL, - cached_at VARCHAR NOT NULL, - PRIMARY KEY (account_id, date) - ); - "#, + "#, )?; Ok(()) @@ -431,256 +415,6 @@ pub fn delete_account(account_id: &str) -> Result<()> { Ok(()) } -// ==================== Cache Functions ==================== - -/// Check if cost summary cache is valid -/// account_name and source_id are passed by the caller to avoid deadlock when acquiring lock while holding database lock -pub fn get_cached_cost_summary_with_account( - account_id: &str, - account_name: &str, - source_id: &SourceId, -) -> Result> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - let mut stmt = conn.prepare( - "SELECT current_month_cost, last_month_cost, currency, month_over_month_change, - current_month_details, last_month_details, cached_at - FROM cost_summary_cache WHERE account_id = ?", - )?; - - let result = stmt.query_row(params![account_id], |row| { - let cached_at_str: String = row.get(6)?; - let current_details_json: Option = row.get(4)?; - let last_details_json: Option = row.get(5)?; - - Ok(( - row.get::<_, f64>(0)?, - row.get::<_, f64>(1)?, - row.get::<_, String>(2)?, - row.get::<_, f64>(3)?, - current_details_json, - last_details_json, - cached_at_str, - )) - }); - - match result { - Ok(( - current, - last, - currency, - change, - current_details_json, - last_details_json, - cached_at_str, - )) => { - // Check if cache is expired - let cached_at = DateTime::parse_from_rfc3339(&cached_at_str) - .map(|dt| dt.with_timezone(&Utc)) - .unwrap_or_else(|_| Utc::now() - Duration::hours(CACHE_TTL_HOURS + 1)); - - let now = Utc::now(); - if now - cached_at > Duration::hours(CACHE_TTL_HOURS) { - tracing::info!("Cost summary cache expired (cached at: {})", cached_at_str); - return Ok(None); - } - - // Parse service details - let current_month_details: Vec = current_details_json - .and_then(|json| serde_json::from_str(&json).ok()) - .unwrap_or_default(); - let last_month_details: Vec = last_details_json - .and_then(|json| serde_json::from_str(&json).ok()) - .unwrap_or_default(); - - tracing::info!( - "Using cost summary cache (cached at: {}, {} hours remaining)", - cached_at_str, - CACHE_TTL_HOURS - (now - cached_at).num_hours() - ); - - Ok(Some(CostSummary { - account_id: account_id.to_string(), - account_name: account_name.to_string(), - source_id: source_id.clone(), - current_month_cost: current, - last_month_cost: last, - currency, - month_over_month_change: change, - current_month_details, - last_month_details, - })) - } - Err(_) => Ok(None), - } -} - -/// Save cost summary to cache -pub fn save_cost_summary_cache(summary: &CostSummary) -> Result<()> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - let current_details_json = serde_json::to_string(&summary.current_month_details)?; - let last_details_json = serde_json::to_string(&summary.last_month_details)?; - - conn.execute( - r#" - INSERT OR REPLACE INTO cost_summary_cache - (account_id, current_month_cost, last_month_cost, currency, month_over_month_change, - current_month_details, last_month_details, cached_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?) - "#, - params![ - summary.account_id, - summary.current_month_cost, - summary.last_month_cost, - summary.currency, - summary.month_over_month_change, - current_details_json, - last_details_json, - Utc::now().to_rfc3339(), - ], - )?; - - tracing::info!("Cached cost summary for account {}", summary.account_id); - Ok(()) -} - -/// Get cached cost trend -pub fn get_cached_cost_trend( - account_id: &str, - start_date: &str, - end_date: &str, -) -> Result> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - // First check if there's cache for this date range and if it's expired - let mut stmt = conn.prepare( - "SELECT date, amount, currency, cached_at FROM cost_trend_cache - WHERE account_id = ? AND date >= ? AND date < ? - ORDER BY date", - )?; - - let rows = stmt.query_map(params![account_id, start_date, end_date], |row| { - Ok(( - row.get::<_, String>(0)?, - row.get::<_, f64>(1)?, - row.get::<_, String>(2)?, - row.get::<_, String>(3)?, - )) - })?; - - let mut daily_costs = Vec::new(); - let mut oldest_cache: Option> = None; - let mut currency = "USD".to_string(); - - for row in rows { - let (date, amount, curr, cached_at_str) = row?; - - let cached_at = DateTime::parse_from_rfc3339(&cached_at_str) - .map(|dt| dt.with_timezone(&Utc)) - .unwrap_or_else(|_| Utc::now() - Duration::hours(CACHE_TTL_HOURS + 1)); - - // Track the oldest cache time - if oldest_cache.is_none() || cached_at < oldest_cache.unwrap() { - oldest_cache = Some(cached_at); - } - - currency = curr; - daily_costs.push(DailyCost { date, amount }); - } - - // Return None if no data or cache expired - if daily_costs.is_empty() { - return Ok(None); - } - - let now = Utc::now(); - if let Some(cached_at) = oldest_cache { - if now - cached_at > Duration::hours(CACHE_TTL_HOURS) { - tracing::info!("Cost trend cache expired"); - return Ok(None); - } - - tracing::info!( - "Using cost trend cache ({} data points, {} hours remaining)", - daily_costs.len(), - CACHE_TTL_HOURS - (now - cached_at).num_hours() - ); - } - - Ok(Some(CostTrend { - account_id: account_id.to_string(), - currency, - daily_costs, - })) -} - -/// Save cost trend to cache -pub fn save_cost_trend_cache(trend: &CostTrend) -> Result<()> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - let now = Utc::now().to_rfc3339(); - - for daily in &trend.daily_costs { - conn.execute( - r#" - INSERT OR REPLACE INTO cost_trend_cache - (account_id, date, amount, currency, cached_at) - VALUES (?, ?, ?, ?, ?) - "#, - params![ - trend.account_id, - daily.date, - daily.amount, - trend.currency, - now, - ], - )?; - } - - tracing::info!( - "Cached cost trend for account {} ({} days)", - trend.account_id, - trend.daily_costs.len() - ); - Ok(()) -} - -/// Clear all cache for specified account (for force refresh, reserved interface) -#[allow(dead_code)] -pub fn clear_account_cache(account_id: &str) -> Result<()> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - conn.execute( - "DELETE FROM cost_summary_cache WHERE account_id = ?", - params![account_id], - )?; - conn.execute( - "DELETE FROM cost_trend_cache WHERE account_id = ?", - params![account_id], - )?; - - tracing::info!("Cleared all cache for account {}", account_id); - Ok(()) -} - -/// Clear all cache (for global force refresh) -pub fn clear_all_cache() -> Result<()> { - let db = get_connection()?; - let conn = db.as_ref().unwrap(); - - conn.execute("DELETE FROM cost_summary_cache", [])?; - conn.execute("DELETE FROM cost_trend_cache", [])?; - - tracing::info!("Cleared all cost cache"); - Ok(()) -} - // ==================== Budget Functions ==================== /// Save or update budget for an account @@ -816,11 +550,16 @@ pub fn get_budget_status(account_id: &str) -> Result> { .find(|a| a.id == account_id) .ok_or_else(|| anyhow::anyhow!("Account not found"))?; - // Get cached cost summary - let cost_summary = - get_cached_cost_summary_with_account(account_id, &account.name, &account.source_id)?; - - let current_cost = cost_summary.map(|cs| cs.current_month_cost).unwrap_or(0.0); + // What the ledger says has been charged this month. It is in the + // reporting currency, while a budget carries a currency of its own; + // reconciling the two belongs with the budget alerts in P2, which is + // also where this function finally gets a caller. + let period = BillingPeriod::containing(Utc::now()); + let current_cost = query::period_total(&PeriodKey::new( + account.source_id.as_str().to_string(), + account.id.clone(), + period.label(), + ))?; // Calculate metrics let percentage_used = if budget.monthly_budget > 0.0 { @@ -944,7 +683,7 @@ mod tests { let columns = column_names(&conn, "cloud_accounts").unwrap(); rebuild_accounts_v1(&conn, &columns).unwrap(); - create_v1_tables(&conn).unwrap(); + create_tables(&conn).unwrap(); let (id, source_id, region): (String, String, String) = conn .query_row( @@ -973,6 +712,27 @@ mod tests { assert_eq!(budget, 100.0); } + #[test] + fn the_response_caches_are_dropped_on_upgrade() { + let conn = legacy_database(); + conn.execute_batch( + "CREATE TABLE cost_summary_cache (account_id VARCHAR PRIMARY KEY); + CREATE TABLE cost_trend_cache (account_id VARCHAR PRIMARY KEY);", + ) + .unwrap(); + + prepare_schema(&conn).unwrap(); + + assert!(!table_exists(&conn, "cost_summary_cache")); + assert!(!table_exists(&conn, "cost_trend_cache")); + assert_eq!(current_schema_version(&conn).unwrap(), APP_SCHEMA_VERSION); + // The account survives the upgrade that removed them. + let accounts: i64 = conn + .query_row("SELECT count(*) FROM cloud_accounts", [], |row| row.get(0)) + .unwrap(); + assert_eq!(accounts, 1); + } + #[test] fn the_v1_rebuild_leaves_an_already_renamed_column_alone() { // A database that got as far as source_id before being interrupted. @@ -986,7 +746,7 @@ mod tests { let columns = column_names(&conn, "cloud_accounts").unwrap(); rebuild_accounts_v1(&conn, &columns).unwrap(); - create_v1_tables(&conn).unwrap(); + create_tables(&conn).unwrap(); let source_id: String = conn .query_row("SELECT source_id FROM cloud_accounts", [], |row| row.get(0)) diff --git a/src/ingest.rs b/src/ingest.rs index b09ceb4..1875330 100644 --- a/src/ingest.rs +++ b/src/ingest.rs @@ -9,18 +9,22 @@ //! partition, and one whole-period replacement in `fct_charge` — all under //! the same batch id. -// The dashboard still reads through the response caches; it starts calling -// this in PR6, when the ledger becomes the read path. -#![allow(dead_code)] - use anyhow::{anyhow, Result}; -use chrono::Utc; +use chrono::{DateTime, Duration, Utc}; use std::path::{Path, PathBuf}; use crate::cloud::raw::{self, RawBatch}; +use crate::cloud::registry::SourceDescriptor; use crate::cloud::{BillingPeriod, CloudAccount, Normalized}; use crate::config::get_raw_data_dir; -use crate::ledger::{self, PeriodKey}; +use crate::ledger::{self, query, PeriodKey}; + +/// How long a period stays fresh after it is ingested. +/// +/// Cost Explorer bills per request and the Alibaba Cloud bill only moves a +/// few times a day, so refreshing on every window focus would cost money +/// for nothing. +const FRESH_FOR_HOURS: i64 = 6; /// What one ingest did, for logging and for the UI to report. #[derive(Debug, Clone, PartialEq, Eq)] @@ -32,9 +36,52 @@ pub struct IngestOutcome { pub raw_path: PathBuf, } -/// Ingest the period that is currently accruing — the UI's "refresh now". -pub fn ingest_current_period(account: &CloudAccount) -> Result { - ingest_period(account, &BillingPeriod::containing(Utc::now())) +/// Bring an account's ledger up to date, skipping periods ingested +/// recently enough unless `force` says otherwise. +/// +/// The current period and the one before it, because the dashboard shows +/// both — and because a provider keeps correcting last month for a while +/// after it ends. A source that reports only a balance has no history to +/// backfill, so it gets the current period alone. +pub fn refresh_account(account: &CloudAccount, now: DateTime, force: bool) -> Result<()> { + let current = BillingPeriod::containing(now); + let is_snapshot = account + .descriptor() + .is_some_and(SourceDescriptor::is_snapshot); + + let periods = if is_snapshot { + vec![current] + } else { + vec![current.previous(), current] + }; + + for period in periods { + if !force && is_fresh(account, &period, now)? { + tracing::debug!( + "Skipping {} {}: ingested within the last {} hours", + account.name, + period.label(), + FRESH_FOR_HOURS + ); + continue; + } + + ingest_period(account, &period)?; + } + + Ok(()) +} + +/// Whether a period was ingested recently enough to leave alone. +fn is_fresh(account: &CloudAccount, period: &BillingPeriod, now: DateTime) -> Result { + let key = PeriodKey::new( + account.source_id.as_str().to_string(), + account.id.clone(), + period.label(), + ); + + Ok(query::last_ingest(&key)? + .is_some_and(|ingested_at| now - ingested_at < Duration::hours(FRESH_FOR_HOURS))) } /// Fetch one account's billing period and land it in the ledger. @@ -65,9 +112,13 @@ pub fn ingest_period(account: &CloudAccount, period: &BillingPeriod) -> Result Result { let descriptor = account.descriptor().ok_or_else(|| { anyhow!( diff --git a/src/ledger/mod.rs b/src/ledger/mod.rs index cbc2db8..3d383af 100644 --- a/src/ledger/mod.rs +++ b/src/ledger/mod.rs @@ -20,6 +20,7 @@ // Written by PR4/PR5 and read by PR6; remove once the AWS normalizer lands. #![allow(dead_code)] +pub mod query; pub mod schema; use anyhow::{anyhow, Result}; @@ -183,10 +184,11 @@ pub struct BalanceSnapshot { } /// Open (creating if needed) the ledger database and apply its schema. -pub fn init_ledger() -> Result<()> { +pub fn init_ledger(reporting_currency: &str) -> Result<()> { let path = get_ledger_database_path()?; let conn = Connection::open(&path)?; schema::apply(&conn)?; + schema::apply_reporting_currency(&conn, reporting_currency)?; let mut ledger = LEDGER_CONNECTION.lock().unwrap(); *ledger = Some(conn); @@ -205,6 +207,24 @@ fn with_connection(f: impl FnOnce(&mut Connection) -> Result) -> Result f(conn) } +/// Same, for the reads in [`query`], which need no transaction. +fn with_connection_ref(f: impl FnOnce(&Connection) -> Result) -> Result { + let guard = LEDGER_CONNECTION + .lock() + .map_err(|e| anyhow!("Failed to lock ledger connection: {}", e))?; + let conn = guard + .as_ref() + .ok_or_else(|| anyhow!("Ledger not initialized"))?; + f(conn) +} + +/// Point the reading view at a different currency. +/// +/// Cheap: the fact table is untouched, only the view is replaced. +pub fn set_reporting_currency(currency: &str) -> Result<()> { + with_connection_ref(|conn| schema::apply_reporting_currency(conn, currency)) +} + /// Identifier for one ingest. /// /// Minted by the caller rather than in here, because the raw payloads are diff --git a/src/ledger/query.rs b/src/ledger/query.rs new file mode 100644 index 0000000..e35b62f --- /dev/null +++ b/src/ledger/query.rs @@ -0,0 +1,390 @@ +//! Reading the ledger. +//! +//! Everything the UI shows comes through [`schema::NORMALIZED_VIEW`], so +//! amounts arrive already expressed in the reporting currency. Nothing in +//! here adds up two currencies. + +use anyhow::Result; +use chrono::{DateTime, Utc}; +use duckdb::{params, Connection}; + +use super::schema::{NORMALIZED_VIEW, TIMESTAMP_FORMAT}; +use super::{with_connection_ref, PeriodKey}; + +/// The most recent balance a source reported for an account. +#[derive(Debug, Clone, PartialEq)] +pub struct Balance { + pub balance: f64, + pub granted_balance: Option, + pub topped_up_balance: Option, + /// The currency the source reports in, which is not converted: a + /// balance is what is left in an account, not an amount spent. + pub currency: String, + pub observed_at: DateTime, +} + +/// One day's charges, as `(YYYY-MM-DD, amount)` in the reporting currency. +pub type DailyTotal = (String, f64); + +/// Total charged in one billing period, in the reporting currency. +pub fn period_total(key: &PeriodKey) -> Result { + with_connection_ref(|conn| period_total_of(conn, key)) +} + +/// Total charged across every account in a billing period. +/// +/// This is the cross-cloud, cross-currency number: one query, one currency +/// out, no summing of amounts that were never comparable. +pub fn total_for_period(billing_period: &str) -> Result { + with_connection_ref(|conn| total_for_period_of(conn, billing_period)) +} + +/// Charges of one period grouped by service, largest first. +pub fn service_breakdown(key: &PeriodKey) -> Result> { + with_connection_ref(|conn| service_breakdown_of(conn, key)) +} + +/// Daily charge totals for an account since an instant, oldest first. +pub fn daily_totals( + provider: &str, + account_id: &str, + since: DateTime, +) -> Result> { + with_connection_ref(|conn| daily_totals_of(conn, provider, account_id, since)) +} + +/// The newest balance snapshot for an account, if it reports one. +pub fn latest_balance(provider: &str, account_id: &str) -> Result> { + with_connection_ref(|conn| latest_balance_of(conn, provider, account_id)) +} + +/// When a period was last ingested, or `None` if it never was. +/// +/// This is what a refresh checks against its cache window: the ledger +/// records when it was written, so nothing else has to. +pub fn last_ingest(key: &PeriodKey) -> Result>> { + with_connection_ref(|conn| last_ingest_of(conn, key)) +} + +/// How many charges could not be converted, because no rate covers their +/// currency. They are missing from every converted total. +pub fn unconverted_charges(billing_period: &str) -> Result { + with_connection_ref(|conn| unconverted_charges_of(conn, billing_period)) +} + +fn period_total_of(conn: &Connection, key: &PeriodKey) -> Result { + let total: Option = conn.query_row( + &format!( + "SELECT sum(billed_cost_base) FROM {NORMALIZED_VIEW} + WHERE provider = ? AND account_id = ? AND billing_period = ?" + ), + params![key.provider, key.account_id, key.billing_period], + |row| row.get(0), + )?; + + Ok(total.unwrap_or(0.0)) +} + +fn total_for_period_of(conn: &Connection, billing_period: &str) -> Result { + let total: Option = conn.query_row( + &format!("SELECT sum(billed_cost_base) FROM {NORMALIZED_VIEW} WHERE billing_period = ?"), + params![billing_period], + |row| row.get(0), + )?; + + Ok(total.unwrap_or(0.0)) +} + +fn service_breakdown_of(conn: &Connection, key: &PeriodKey) -> Result> { + let mut stmt = conn.prepare(&format!( + "SELECT coalesce(service_name, 'Other') AS service, sum(billed_cost_base) AS amount + FROM {NORMALIZED_VIEW} + WHERE provider = ? AND account_id = ? AND billing_period = ? + GROUP BY service + HAVING amount > 0 + ORDER BY amount DESC" + ))?; + + let rows = stmt + .query_map( + params![key.provider, key.account_id, key.billing_period], + |row| Ok((row.get::<_, String>(0)?, row.get::<_, f64>(1)?)), + )? + .collect::, _>>()?; + + Ok(rows) +} + +fn daily_totals_of( + conn: &Connection, + provider: &str, + account_id: &str, + since: DateTime, +) -> Result> { + let mut stmt = conn.prepare(&format!( + "SELECT strftime(charge_period_start, '%Y-%m-%d') AS day, sum(billed_cost_base) AS amount + FROM {NORMALIZED_VIEW} + WHERE provider = ? AND account_id = ? AND charge_period_start >= CAST(? AS TIMESTAMP) + GROUP BY day + ORDER BY day" + ))?; + + let rows = stmt + .query_map( + params![ + provider, + account_id, + since.format(TIMESTAMP_FORMAT).to_string() + ], + |row| Ok((row.get::<_, String>(0)?, row.get::<_, Option>(1)?)), + )? + .collect::, _>>()?; + + Ok(rows + .into_iter() + .map(|(day, amount)| (day, amount.unwrap_or(0.0))) + .collect()) +} + +fn latest_balance_of( + conn: &Connection, + provider: &str, + account_id: &str, +) -> Result> { + let mut stmt = conn.prepare( + "SELECT balance, granted_balance, topped_up_balance, currency, + CAST(observed_at AS VARCHAR) + FROM fct_balance_snapshot + WHERE provider = ? AND account_id = ? + ORDER BY observed_at DESC + LIMIT 1", + )?; + + let mut rows = stmt.query_map(params![provider, account_id], |row| { + Ok(( + row.get::<_, f64>(0)?, + row.get::<_, Option>(1)?, + row.get::<_, Option>(2)?, + row.get::<_, String>(3)?, + row.get::<_, String>(4)?, + )) + })?; + + let Some(row) = rows.next().transpose()? else { + return Ok(None); + }; + let (balance, granted_balance, topped_up_balance, currency, observed_at) = row; + + Ok(Some(Balance { + balance, + granted_balance, + topped_up_balance, + currency, + observed_at: super::parse_timestamp(&observed_at)?, + })) +} + +fn last_ingest_of(conn: &Connection, key: &PeriodKey) -> Result>> { + let mut stmt = conn.prepare( + "SELECT CAST(max(completed_at) AS VARCHAR) FROM ingest_batch + WHERE provider = ? AND account_id = ? AND billing_period = ? AND status = 'complete'", + )?; + + let completed_at: Option = stmt.query_row( + params![key.provider, key.account_id, key.billing_period], + |row| row.get(0), + )?; + + completed_at + .map(|stamp| super::parse_timestamp(&stamp)) + .transpose() +} + +fn unconverted_charges_of(conn: &Connection, billing_period: &str) -> Result { + conn.query_row( + &format!( + "SELECT count(*) FROM {NORMALIZED_VIEW} + WHERE billing_period = ? AND billed_cost IS NOT NULL AND billed_cost_base IS NULL" + ), + params![billing_period], + |row| row.get(0), + ) + .map_err(Into::into) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ledger::schema; + use crate::ledger::{BalanceSnapshot, Charge, ChargeCategory}; + use chrono::TimeZone; + + fn conn(reporting_currency: &str) -> Connection { + let conn = Connection::open_in_memory().expect("in-memory duckdb"); + schema::apply(&conn).expect("schema applies"); + schema::apply_reporting_currency(&conn, reporting_currency).expect("view applies"); + conn + } + + fn at(day: u32) -> DateTime { + Utc.with_ymd_and_hms(2026, 8, day, 0, 0, 0).unwrap() + } + + fn charge(service: &str, amount: f64, currency: &str, day: u32) -> Charge { + Charge { + service_name: Some(service.to_string()), + billed_cost: Some(amount), + ..Charge::new(at(day), at(day + 1), currency) + } + } + + fn aws() -> PeriodKey { + PeriodKey::new("AWS", "acct-1", "2026-08") + } + + fn aliyun() -> PeriodKey { + PeriodKey::new("Aliyun", "acct-2", "2026-08") + } + + fn write(conn: &mut Connection, key: &PeriodKey, charges: &[Charge]) { + let batch_id = crate::ledger::new_batch_id(); + crate::ledger::write_period(conn, key, &batch_id, charges, None).unwrap(); + } + + #[test] + fn a_cross_cloud_total_is_one_query_in_one_currency() { + let mut conn = conn("USD"); + write(&mut conn, &aws(), &[charge("EC2", 12.5, "USD", 1)]); + write(&mut conn, &aliyun(), &[charge("ECS", 710.0, "CNY", 1)]); + + // 710 CNY at the built-in 0.1408 is 99.968 USD. + let total = total_for_period_of(&conn, "2026-08").unwrap(); + assert!((total - 112.468).abs() < 1e-6, "got {total}"); + + // Each account still reports in the same currency as the total. + assert!((period_total_of(&conn, &aws()).unwrap() - 12.5).abs() < 1e-9); + assert!((period_total_of(&conn, &aliyun()).unwrap() - 99.968).abs() < 1e-6); + } + + #[test] + fn changing_the_reporting_currency_rereads_the_same_rows() { + let mut conn = conn("USD"); + write(&mut conn, &aliyun(), &[charge("ECS", 710.0, "CNY", 1)]); + + assert!((period_total_of(&conn, &aliyun()).unwrap() - 99.968).abs() < 1e-6); + + // No rewrite of the fact table: only the view changes. + schema::apply_reporting_currency(&conn, "CNY").unwrap(); + assert!((period_total_of(&conn, &aliyun()).unwrap() - 710.0).abs() < 1e-9); + } + + #[test] + fn a_charge_in_a_currency_no_rate_covers_is_reported_rather_than_counted() { + let mut conn = conn("USD"); + write( + &mut conn, + &aws(), + &[ + charge("EC2", 12.5, "USD", 1), + charge("Something", 100.0, "JPY", 1), + ], + ); + + // The unconvertible row is left out of the total... + let total = total_for_period_of(&conn, "2026-08").unwrap(); + assert!((total - 12.5).abs() < 1e-9, "got {total}"); + // ...and is countable, so the UI can say so. + assert_eq!(unconverted_charges_of(&conn, "2026-08").unwrap(), 1); + } + + #[test] + fn a_rate_is_taken_from_the_charges_own_time() { + let mut conn = conn("USD"); + conn.execute_batch( + "INSERT OR REPLACE INTO dim_fx_rate VALUES ('CNY', 'USD', DATE '2026-08-15', 0.2, 'test')", + ) + .unwrap(); + + write( + &mut conn, + &aliyun(), + &[ + charge("ECS", 100.0, "CNY", 1), + charge("ECS", 100.0, "CNY", 20), + ], + ); + + // The 1 August charge predates the new rate and keeps the old one; + // the 20 August charge takes the newer. + let daily = daily_totals_of(&conn, "Aliyun", "acct-2", at(1)).unwrap(); + assert_eq!(daily.len(), 2); + assert!((daily[0].1 - 14.08).abs() < 1e-9, "got {:?}", daily[0]); + assert!((daily[1].1 - 20.0).abs() < 1e-9, "got {:?}", daily[1]); + } + + #[test] + fn a_breakdown_is_by_service_largest_first() { + let mut conn = conn("USD"); + write( + &mut conn, + &aws(), + &[ + charge("S3", 0.75, "USD", 1), + charge("EC2", 12.5, "USD", 1), + charge("EC2", 4.0, "USD", 2), + Charge { + charge_category: ChargeCategory::Credit, + billed_cost: Some(-2.0), + ..charge("EC2", -2.0, "USD", 2) + }, + ], + ); + + let breakdown = service_breakdown_of(&conn, &aws()).unwrap(); + assert_eq!( + breakdown, + vec![("EC2".to_string(), 14.5), ("S3".to_string(), 0.75)] + ); + } + + #[test] + fn a_period_that_was_never_ingested_has_no_ingest_time() { + let mut conn = conn("USD"); + assert!(last_ingest_of(&conn, &aws()).unwrap().is_none()); + + write(&mut conn, &aws(), &[charge("EC2", 1.0, "USD", 1)]); + assert!(last_ingest_of(&conn, &aws()).unwrap().is_some()); + } + + #[test] + fn the_newest_balance_is_the_one_reported() { + let mut conn = conn("USD"); + for (day, amount) in [(1, 50.0), (3, 30.0), (2, 40.0)] { + crate::ledger::write_balance( + &mut conn, + &BalanceSnapshot { + provider: "DeepSeek".to_string(), + account_id: "acct-3".to_string(), + observed_at: at(day), + balance: amount, + granted_balance: Some(5.0), + topped_up_balance: Some(amount - 5.0), + currency: "CNY".to_string(), + }, + ) + .unwrap(); + } + + let balance = latest_balance_of(&conn, "DeepSeek", "acct-3") + .unwrap() + .expect("a balance was recorded"); + assert_eq!(balance.balance, 30.0); + assert_eq!(balance.observed_at, at(3)); + // Not converted: a balance is what is left, not what was spent. + assert_eq!(balance.currency, "CNY"); + + assert!(latest_balance_of(&conn, "DeepSeek", "unknown") + .unwrap() + .is_none()); + } +} diff --git a/src/ledger/schema.rs b/src/ledger/schema.rs index 272e0f2..487608a 100644 --- a/src/ledger/schema.rs +++ b/src/ledger/schema.rs @@ -17,6 +17,24 @@ use duckdb::{params, Connection}; /// Bumped whenever the statements below change shape. pub const SCHEMA_VERSION: i32 = 1; +/// The view the application reads: every charge with its amount also +/// expressed in the reporting currency. +pub const NORMALIZED_VIEW: &str = "v_charge_normalized"; + +/// Rates shipped with the build, as `(from, to, date, rate)`. +/// +/// Static and approximate. They are dated because a rate gets corrected +/// and because a charge must be converted at a rate from its own time, not +/// from today's — so a real feed, when it arrives, only has to insert rows +/// with later dates. Nothing here is overwritten by it. +/// +/// Only the currencies the sources actually bill in are covered: USD (AWS, +/// DeepSeek) and CNY (Alibaba Cloud, DeepSeek). +pub const BUILTIN_RATES: &[(&str, &str, &str, f64)] = &[ + ("USD", "CNY", "2026-01-01", 7.10), + ("CNY", "USD", "2026-01-01", 0.1408), +]; + /// Format used for every `TIMESTAMP` bind and parse in this module. pub const TIMESTAMP_FORMAT: &str = "%Y-%m-%d %H:%M:%S"; @@ -117,6 +135,8 @@ pub fn apply(conn: &Connection) -> Result<()> { "#, )?; + seed_builtin_rates(conn)?; + conn.execute( "INSERT OR REPLACE INTO schema_version (version, applied_at) VALUES (?, CAST(? AS TIMESTAMP))", params![ @@ -127,3 +147,57 @@ pub fn apply(conn: &Connection) -> Result<()> { Ok(()) } + +/// Insert the rates that ship with the build, leaving any other row alone. +fn seed_builtin_rates(conn: &Connection) -> Result<()> { + for (from_ccy, to_ccy, rate_date, rate) in BUILTIN_RATES { + conn.execute( + "INSERT OR REPLACE INTO dim_fx_rate (from_ccy, to_ccy, rate_date, rate, source) + VALUES (?, ?, CAST(? AS DATE), ?, 'builtin')", + params![from_ccy, to_ccy, rate_date, rate], + )?; + } + + Ok(()) +} + +/// (Re)create the reading view for a reporting currency. +/// +/// Conversion happens here rather than at write time because a rate gets +/// corrected after the fact and because the user may change the currency +/// they want to read in — either would mean rewriting the fact table if +/// the amounts had been converted on the way in. +/// +/// The join is ASOF: a charge takes the newest rate dated on or before the +/// charge itself, never a later one. A charge already in the reporting +/// currency needs no rate at all, and one for which no rate exists keeps a +/// NULL `billed_cost_base` — it is left out of a converted total rather +/// than silently counted at par. +pub fn apply_reporting_currency(conn: &Connection, currency: &str) -> Result<()> { + if !currency.chars().all(|c| c.is_ascii_alphabetic()) || currency.is_empty() { + return Err(anyhow::anyhow!("Not a currency code: {:?}", currency)); + } + + conn.execute_batch(&format!( + r#" + CREATE OR REPLACE VIEW {NORMALIZED_VIEW} AS + SELECT + c.*, + CASE WHEN c.billing_currency = '{currency}' THEN 1.0 ELSE f.rate END AS fx_rate, + c.billed_cost + * CASE WHEN c.billing_currency = '{currency}' THEN 1.0 ELSE f.rate END + AS billed_cost_base, + c.effective_cost + * CASE WHEN c.billing_currency = '{currency}' THEN 1.0 ELSE f.rate END + AS effective_cost_base, + '{currency}' AS reporting_currency + FROM fct_charge c + ASOF LEFT JOIN dim_fx_rate f + ON f.from_ccy = c.billing_currency + AND f.to_ccy = '{currency}' + AND f.rate_date <= c.charge_period_start::DATE; + "# + ))?; + + Ok(()) +} diff --git a/src/main.rs b/src/main.rs index cb7edc1..b008b5d 100644 --- a/src/main.rs +++ b/src/main.rs @@ -5,6 +5,7 @@ mod crypto; mod db; mod ingest; mod ledger; +mod report; mod secret_store; mod ui; @@ -39,7 +40,9 @@ fn main() { // Apply the persisted theme before the first window opens, so the app // does not flash the default appearance on startup. - let dark_mode = config::load_config().unwrap_or_default().theme.dark_mode; + let settings = config::load_config().unwrap_or_default(); + let dark_mode = settings.theme.dark_mode; + let reporting_currency = settings.reporting_currency.clone(); Theme::change( if dark_mode { ThemeMode::Dark @@ -55,7 +58,7 @@ fn main() { if let Err(e) = db::init_database() { tracing::error!("Database initialization failed: {}", e); } - if let Err(e) = ledger::init_ledger() { + if let Err(e) = ledger::init_ledger(&reporting_currency) { tracing::error!("Ledger initialization failed: {}", e); } diff --git a/src/report.rs b/src/report.rs new file mode 100644 index 0000000..0d725e4 --- /dev/null +++ b/src/report.rs @@ -0,0 +1,227 @@ +//! What the dashboard shows, read out of the ledger. +//! +//! Nothing here talks to a provider. The numbers come from +//! `v_charge_normalized`, so they are already in one currency by the time +//! anything adds them up — the dashboard used to sum AWS dollars and +//! Alibaba Cloud yuan into a single figure. + +use anyhow::Result; +use chrono::{DateTime, Utc}; + +use crate::cloud::registry::SourceDescriptor; +use crate::cloud::{BillingPeriod, CloudAccount, SourceId}; +use crate::ledger::query::{self, Balance}; +use crate::ledger::PeriodKey; + +/// Shown in place of a source's name when its id is not in the registry. +const UNKNOWN_SOURCE: &str = "Unknown"; + +/// One service's share of a period. +#[derive(Debug, Clone)] +pub struct ServiceCost { + pub service: String, + pub amount: f64, + pub currency: String, +} + +/// One day of a trend chart. +#[derive(Debug, Clone)] +pub struct DailyCost { + pub date: String, + pub amount: f64, +} + +/// One account's card on the dashboard. +#[derive(Debug, Clone)] +pub struct AccountReport { + pub account_id: String, + pub account_name: String, + pub source_id: SourceId, + /// The headline figure: what was charged this period, or — for a + /// source that only reports state — what is left in the account. + pub amount: f64, + /// Currency `amount` is in: the reporting currency for charges, the + /// source's own for a balance. A balance is not converted, because it + /// is not spend and does not belong in a spend total. + pub currency: String, + /// Change against the period before, or `None` for a source that + /// reports a balance and has no period to compare against. + pub month_over_month_change: Option, + /// What the headline figure is made of: services for a period cost, + /// the granted and topped-up parts for a balance. + pub services: Vec, +} + +impl AccountReport { + fn descriptor(&self) -> Option<&'static SourceDescriptor> { + self.source_id.descriptor() + } + + /// Short label for the source, for badges. + pub fn short_name(&self) -> &'static str { + self.descriptor().map_or(UNKNOWN_SOURCE, |s| s.short_name) + } + + /// Whether the headline figure is a balance rather than a period cost. + /// The dashboard lists the two kinds separately and labels them + /// differently. + pub fn is_snapshot(&self) -> bool { + self.descriptor().is_some_and(SourceDescriptor::is_snapshot) + } +} + +/// Everything one dashboard render needs. +#[derive(Debug, Clone)] +pub struct DashboardReport { + pub accounts: Vec, + /// Cross-cloud totals, in the reporting currency. + pub current_month: f64, + pub last_month: f64, + pub month_over_month_change: f64, + pub reporting_currency: String, + /// Charges left out of the totals because no rate covers their + /// currency. Zero unless a source starts billing in something the + /// built-in rate table does not know. + pub unconverted_charges: i64, +} + +/// Read the ledger for a set of accounts. No network access. +pub fn build( + accounts: &[CloudAccount], + now: DateTime, + reporting_currency: &str, +) -> Result { + let current = BillingPeriod::containing(now); + let previous = current.previous(); + + let mut reports = Vec::new(); + for account in accounts { + if !account.enabled { + continue; + } + reports.push(account_report( + account, + ¤t, + &previous, + reporting_currency, + )?); + } + + let current_month = query::total_for_period(¤t.label())?; + let last_month = query::total_for_period(&previous.label())?; + + Ok(DashboardReport { + accounts: reports, + current_month, + last_month, + month_over_month_change: change(current_month, last_month).unwrap_or(0.0), + reporting_currency: reporting_currency.to_string(), + unconverted_charges: query::unconverted_charges(¤t.label())?, + }) +} + +/// Daily charges for an account over the last `days` days. +pub fn trend(account: &CloudAccount, days: i64, now: DateTime) -> Result> { + let since = now - chrono::Duration::days(days); + let daily = query::daily_totals(account.source_id.as_str(), &account.id, since)?; + + Ok(daily + .into_iter() + .map(|(date, amount)| DailyCost { date, amount }) + .collect()) +} + +/// The symbol an amount is shown with, or the code itself when there is +/// no familiar one. +pub fn symbol(currency: &str) -> &str { + match currency { + "USD" => "$", + "CNY" => "¥", + other => other, + } +} + +fn account_report( + account: &CloudAccount, + current: &BillingPeriod, + previous: &BillingPeriod, + reporting_currency: &str, +) -> Result { + let provider = account.source_id.as_str(); + let key = |period: &BillingPeriod| { + PeriodKey::new(provider.to_string(), account.id.clone(), period.label()) + }; + + let is_snapshot = account + .descriptor() + .is_some_and(SourceDescriptor::is_snapshot); + let balance = query::latest_balance(provider, &account.id)?; + + let (amount, currency, last_month, services) = if is_snapshot { + // A balance source has charges too — a top-up is a purchase — but + // the number worth showing on the card is what is left. + let currency = balance + .as_ref() + .map_or_else(|| reporting_currency.to_string(), |b| b.currency.clone()); + ( + balance.as_ref().map_or(0.0, |b| b.balance), + currency.clone(), + None, + balance + .as_ref() + .map_or_else(Vec::new, |b| breakdown(b, ¤cy)), + ) + } else { + let services = query::service_breakdown(&key(current))? + .into_iter() + .map(|(service, amount)| ServiceCost { + service, + amount, + currency: reporting_currency.to_string(), + }) + .collect(); + + ( + query::period_total(&key(current))?, + reporting_currency.to_string(), + Some(query::period_total(&key(previous))?), + services, + ) + }; + + Ok(AccountReport { + account_id: account.id.clone(), + account_name: account.name.clone(), + source_id: account.source_id.clone(), + amount, + currency, + month_over_month_change: last_month.and_then(|last| change(amount, last)), + services, + }) +} + +/// What a balance is made of, for the expanded card. Only the parts the +/// source actually reports are listed. +fn breakdown(balance: &Balance, currency: &str) -> Vec { + [ + ("Granted balance", balance.granted_balance), + ("Topped-up balance", balance.topped_up_balance), + ] + .into_iter() + .filter_map(|(service, amount)| { + amount + .filter(|amount| *amount != 0.0) + .map(|amount| ServiceCost { + service: service.to_string(), + amount, + currency: currency.to_string(), + }) + }) + .collect() +} + +/// Percentage change, or `None` when there is no base to compare against — +/// everything is up infinitely from nothing. +fn change(current: f64, last: f64) -> Option { + (last != 0.0).then(|| ((current - last) / last) * 100.0) +} diff --git a/src/ui/chart.rs b/src/ui/chart.rs index 835db19..1245b21 100644 --- a/src/ui/chart.rs +++ b/src/ui/chart.rs @@ -10,7 +10,7 @@ use gpui::*; use gpui_component::chart::{BarChart, LineChart, PieChart}; use gpui_component::{ActiveTheme, StyledExt}; -use crate::cloud::{DailyCost, ServiceCost}; +use crate::report::{self, DailyCost, ServiceCost}; // ==================== Bar Chart ==================== @@ -102,15 +102,18 @@ pub struct CostBarChart { height: f32, /// Show labels on bars show_labels: bool, + /// Currency the amounts are in, for the value labels + currency: String, } impl CostBarChart { - pub fn new(daily_costs: Vec, width: f32, height: f32) -> Self { + pub fn new(daily_costs: Vec, width: f32, height: f32, currency: String) -> Self { Self { daily_costs, width, height, show_labels: false, // Default: no labels (cleaner look) + currency, } } @@ -159,6 +162,7 @@ impl CostBarChart { let tick_margin = (chart_data.len() / 6).max(1); let show_labels = self.show_labels; + let symbol = report::symbol(&self.currency).to_string(); div() .w(px(self.width)) @@ -170,7 +174,7 @@ impl CostBarChart { .fill(move |_| chart_color) .tick_margin(tick_margin) .when(show_labels, |chart| { - chart.label(|d| format!("${:.2}", d.amount)) + chart.label(move |d| format!("{}{:.2}", symbol, d.amount)) }), ) .into_any_element() @@ -380,7 +384,7 @@ impl ServicePieChart { }; // Use full service name for legend (truncate only if very long) let name = Self::truncate_legend_name(&s.service); - let amount = s.amount; + let amount = format!("{}{:.2}", report::symbol(&s.currency), s.amount); (color, name, amount, percentage) }) .collect(); @@ -437,7 +441,7 @@ impl ServicePieChart { .text_sm() .font_weight(FontWeight::MEDIUM) .text_color(cx.theme().foreground) - .child(format!("${:.2}", amount)), + .child(amount), ) .child( div() @@ -496,7 +500,7 @@ pub struct CostStats { pub average: f64, pub max: f64, pub min: f64, - #[allow(dead_code)] + /// Currency every figure above is in. pub currency: String, } @@ -539,7 +543,7 @@ impl CostStats { .text_sm() .font_weight(FontWeight::MEDIUM) .text_color(cx.theme().foreground) - .child(format!("${:.2}", value)), + .child(format!("{}{:.2}", report::symbol(&self.currency), value)), ) } } diff --git a/src/ui/dashboard.rs b/src/ui/dashboard.rs index 031e156..b9825c5 100644 --- a/src/ui/dashboard.rs +++ b/src/ui/dashboard.rs @@ -6,20 +6,20 @@ use gpui_component::{button::*, scroll::ScrollableElement, *}; use std::collections::HashMap; use super::chart::{CostBarChart, CostStats, ServicePieChart}; -use crate::cloud::{CostSummary, CostTrend}; +use crate::report::{self, AccountReport, DailyCost, DashboardReport}; /// Dashboard View pub struct DashboardView { - /// Cost summary data - summaries: Vec, + /// Everything on screen, read out of the ledger + report: Option, /// Whether loading is in progress loading: bool, /// Error message error: Option, /// Currently expanded account ID (for drill-down) expanded_account: Option, - /// Cost trend cache (account_id -> CostTrend) - cost_trends: HashMap, + /// Daily charges per account, loaded when a card is expanded + cost_trends: HashMap>, /// Accounts currently loading trends loading_trends: HashMap, } @@ -41,7 +41,7 @@ impl DashboardView { .detach(); Self { - summaries: Vec::new(), + report: None, loading: true, // Initial state is loading error: None, expanded_account: None, @@ -52,73 +52,59 @@ impl DashboardView { /// Refresh data pub fn refresh(&mut self, cx: &mut Context) { + self.load(false, cx); + } + + /// Refresh past the freshness window, paying for another fetch. + fn force_refresh(&mut self, cx: &mut Context) { + self.cost_trends.clear(); + self.load(true, cx); + } + + /// Ingest what is stale, then read the ledger. + /// + /// A provider that fails is logged and skipped rather than failing the + /// whole render: what is already in the ledger is still worth showing, + /// and it is the last thing that was true. + fn load(&mut self, force: bool, cx: &mut Context) { self.loading = true; self.error = None; cx.notify(); // Use channel to fetch data in background thread - let (tx, rx) = std::sync::mpsc::channel::, String>>(); + let (tx, rx) = std::sync::mpsc::channel::>(); std::thread::spawn(move || { - match crate::db::get_all_accounts() { - Ok(accounts) => { - let mut summaries = Vec::new(); - - for account in accounts { - if !account.enabled { - continue; - } - - // Try to get from cache first - match crate::db::get_cached_cost_summary_with_account( - &account.id, - &account.name, - &account.source_id, - ) { - Ok(Some(cached)) => { - summaries.push(cached); - continue; - } - Ok(None) => {} - Err(_) => {} - } + let reporting_currency = crate::config::load_config() + .unwrap_or_default() + .reporting_currency; - let Some(descriptor) = account.descriptor() else { - continue; - }; - let service = (descriptor.build)(account.context(descriptor)); - - match service.get_cost_summary() { - Ok(summary) => { - // Save to cache - if let Err(e) = crate::db::save_cost_summary_cache(&summary) { - tracing::warn!("Failed to save cost cache: {}", e); - } - summaries.push(summary); - } - Err(e) => { - tracing::error!( - "Failed to get {} cost for {}: {}", - descriptor.short_name, - account.name, - e - ); - } - } - } - let _ = tx.send(Ok(summaries)); - } + let accounts = match crate::db::get_all_accounts() { + Ok(accounts) => accounts, Err(e) => { tracing::error!("Failed to get account list: {}", e); let _ = tx.send(Err(format!("Failed to load data: {}", e))); + return; + } + }; + + let now = chrono::Utc::now(); + for account in accounts.iter().filter(|account| account.enabled) { + if let Err(e) = crate::ingest::refresh_account(account, now, force) { + tracing::error!("Failed to refresh {}: {}", account.name, e); } } + + let _ = tx.send( + crate::report::build(&accounts, now, &reporting_currency) + .map_err(|e| format!("Failed to read the ledger: {}", e)), + ); }); // Use gpui spawn to wait for results cx.spawn(async move |this, cx| { let result = smol::unblock(move || { - rx.recv_timeout(std::time::Duration::from_secs(60)) + rx.recv_timeout(std::time::Duration::from_secs(120)) .unwrap_or(Err("Data retrieval timeout".to_string())) }) .await; @@ -126,8 +112,8 @@ impl DashboardView { cx.update(|cx| { this.update(cx, |this, cx| { match result { - Ok(summaries) => { - this.summaries = summaries; + Ok(report) => { + this.report = Some(report); this.loading = false; this.error = None; } @@ -180,20 +166,16 @@ impl DashboardView { ) } - /// Force refresh (clear cache and refetch) - fn force_refresh(&mut self, cx: &mut Context) { - // Clear all cache - if let Err(e) = crate::db::clear_all_cache() { - tracing::warn!("Failed to clear cache: {}", e); - } - // Clear trend cache in memory - self.cost_trends.clear(); - // Then refresh - self.refresh(cx); - } - fn render_summary_cards(&self, cx: &Context) -> impl IntoElement { - if self.summaries.is_empty() { + let Some(report) = self.report.as_ref() else { + return div().w_full().p_8().items_center().justify_center().child( + div() + .text_color(cx.theme().muted_foreground) + .child("No data available, please add a cloud account first"), + ); + }; + + if report.accounts.is_empty() { return div().w_full().p_8().items_center().justify_center().child( div() .text_color(cx.theme().muted_foreground) @@ -201,19 +183,15 @@ impl DashboardView { ); } - let total_current: f64 = self.summaries.iter().map(|s| s.current_month_cost).sum(); - let total_last: f64 = self.summaries.iter().map(|s| s.last_month_cost).sum(); - let total_change = if total_last > 0.0 { - ((total_current - total_last) / total_last) * 100.0 - } else { - 0.0 - }; + let symbol = report::symbol(&report.reporting_currency); + let money = |amount: f64| format!("{}{:.2}", symbol, amount); div() .w_full() .v_flex() .gap_4() - // Overview cards + // Overview cards. Every figure here is in the reporting + // currency, converted per charge at a rate from its own time. .child( div() .w_full() @@ -221,29 +199,38 @@ impl DashboardView { .gap_4() .child(self.render_stat_card( "Current Month", - &format!("${:.2}", total_current), - None, - cx, - )) - .child(self.render_stat_card( - "Last Month", - &format!("${:.2}", total_last), + &money(report.current_month), None, cx, )) + .child(self.render_stat_card("Last Month", &money(report.last_month), None, cx)) .child(self.render_stat_card( "Month-over-Month", - &format!("{:+.1}%", total_change), - Some(total_change >= 0.0), + &format!("{:+.1}%", report.month_over_month_change), + Some(report.month_over_month_change >= 0.0), cx, )) .child(self.render_stat_card( "Active Accounts", - &self.summaries.len().to_string(), + &report.accounts.len().to_string(), None, cx, )), ) + // A charge in a currency no rate covers is missing from the + // totals above; say so rather than quietly under-reporting. + .when(report.unconverted_charges > 0, |el| { + el.child( + div() + .w_full() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(format!( + "{} charge(s) are not included: no exchange rate to {}", + report.unconverted_charges, report.reporting_currency + )), + ) + }) // Per-account costs (split into cost accounts and balance accounts) .child( div() @@ -255,26 +242,36 @@ impl DashboardView { ) // Sources that report a period cost .child({ - let cost_summaries: Vec<&CostSummary> = - self.summaries.iter().filter(|s| !s.is_snapshot()).collect(); - - div().w_full().v_flex().gap_4().children( - cost_summaries - .into_iter() - .enumerate() - .map(|(index, summary)| { - let is_expanded = - self.expanded_account.as_ref() == Some(&summary.account_id); - self.render_account_card(summary, is_expanded, index, cx) - }), - ) + let cost_accounts: Vec<&AccountReport> = report + .accounts + .iter() + .filter(|account| !account.is_snapshot()) + .collect(); + + div() + .w_full() + .v_flex() + .gap_4() + .children( + cost_accounts + .into_iter() + .enumerate() + .map(|(index, account)| { + let is_expanded = + self.expanded_account.as_ref() == Some(&account.account_id); + self.render_account_card(account, is_expanded, index, cx) + }), + ) }) // Sources that report a point-in-time balance instead .child({ - let balance_summaries: Vec<&CostSummary> = - self.summaries.iter().filter(|s| s.is_snapshot()).collect(); + let balance_accounts: Vec<&AccountReport> = report + .accounts + .iter() + .filter(|account| account.is_snapshot()) + .collect(); - if balance_summaries.is_empty() { + if balance_accounts.is_empty() { div() } else { div() @@ -289,13 +286,14 @@ impl DashboardView { ) .child( div().w_full().v_flex().gap_4().children( - balance_summaries.into_iter().enumerate().map( - |(index, summary)| { + balance_accounts + .into_iter() + .enumerate() + .map(|(index, account)| { let is_expanded = self.expanded_account.as_ref() - == Some(&summary.account_id); - self.render_account_card(summary, is_expanded, index, cx) - }, - ), + == Some(&account.account_id); + self.render_account_card(account, is_expanded, index, cx) + }), ), ) } @@ -341,23 +339,24 @@ impl DashboardView { fn render_account_card( &self, - summary: &CostSummary, + account: &AccountReport, is_expanded: bool, index: usize, cx: &Context, ) -> impl IntoElement { - let change_color = if summary.month_over_month_change >= 0.0 { - gpui::red() - } else { - gpui::green() + let change = account.month_over_month_change; + let change_color = match change { + Some(change) if change < 0.0 => gpui::green(), + Some(_) => gpui::red(), + None => cx.theme().muted_foreground, }; - let account_id = summary.account_id.clone(); - let details = summary.current_month_details.clone(); + let account_id = account.account_id.clone(); + let details = account.services.clone(); // Pre-render trend chart (render outside closure to avoid borrow issues) let trend_chart = if is_expanded { - Some(self.render_trend_chart(&summary.account_id, cx)) + Some(self.render_trend_chart(&account.account_id, cx)) } else { None }; @@ -394,7 +393,7 @@ impl DashboardView { div() .font_weight(FontWeight::SEMIBOLD) .text_color(cx.theme().foreground) - .child(summary.account_name.clone()), + .child(account.account_name.clone()), ) .child( div() @@ -411,7 +410,7 @@ impl DashboardView { .rounded_md() .bg(cx.theme().accent.opacity(0.1)) .text_color(cx.theme().accent) - .child(summary.short_name()), + .child(account.short_name()), ), ) // Cost overview @@ -420,19 +419,14 @@ impl DashboardView { .h_flex() .justify_between() .child({ - // Format per-account amount using account currency. - let label = if summary.is_snapshot() { + // A balance stays in the currency the source keeps + // it in; a period cost is in the reporting one. + let label = if account.is_snapshot() { "Balance" } else { "This Month" }; - let symbol = match summary.currency.as_str() { - "CNY" => "¥", - "USD" => "$", - other => other, - }; - div() .v_flex() .child( @@ -441,12 +435,11 @@ impl DashboardView { .text_color(cx.theme().muted_foreground) .child(label), ) - .child( - div() - .text_lg() - .font_weight(FontWeight::BOLD) - .child(format!("{}{:.2}", symbol, summary.current_month_cost)), - ) + .child(div().text_lg().font_weight(FontWeight::BOLD).child(format!( + "{}{:.2}", + report::symbol(&account.currency), + account.amount + ))) }) .child( div() @@ -463,7 +456,12 @@ impl DashboardView { .text_lg() .font_weight(FontWeight::BOLD) .text_color(change_color) - .child(format!("{:+.1}%", summary.month_over_month_change)), + // Nothing to compare against reads as + // an em dash, not as a flat 0%. + .child(change.map_or_else( + || "—".to_string(), + |change| format!("{:+.1}%", change), + )), ), ), ) @@ -523,28 +521,28 @@ impl DashboardView { .into_any_element(); } - // Check for cached data - if let Some(trend) = self.cost_trends.get(account_id) { + // Charges are already in the ledger; the chart just reads them. + if let Some(daily_costs) = self.cost_trends.get(account_id) { + let currency = self + .report + .as_ref() + .map(|report| report.reporting_currency.clone()) + .unwrap_or_default(); + // Use BarChart with labels for daily cost visualization - let bar_chart = CostBarChart::new(trend.daily_costs.clone(), 550.0, 150.0); + let bar_chart = CostBarChart::new(daily_costs.clone(), 550.0, 150.0, currency.clone()); - // Calculate statistics from daily_costs - let total: f64 = trend.daily_costs.iter().map(|d| d.amount).sum(); - let count = trend.daily_costs.len() as f64; + let total: f64 = daily_costs.iter().map(|d| d.amount).sum(); + let count = daily_costs.len() as f64; let average = if count > 0.0 { total / count } else { 0.0 }; - let max = trend - .daily_costs - .iter() - .map(|d| d.amount) - .fold(0.0_f64, f64::max); - let min = trend - .daily_costs + let max = daily_costs.iter().map(|d| d.amount).fold(0.0_f64, f64::max); + let min = daily_costs .iter() .map(|d| d.amount) .fold(f64::MAX, f64::min); let min = if min == f64::MAX { 0.0 } else { min }; - let stats = CostStats::new(total, average, max, min, trend.currency.clone()); + let stats = CostStats::new(total, average, max, min, currency); return div() .w_full() @@ -588,6 +586,9 @@ impl DashboardView { } /// Load cost trend data (lazy loading) + /// + /// A read of the ledger, not a fetch: the daily rows are already there + /// from the refresh that filled the cards. fn load_cost_trend(&mut self, account_id: &str, cx: &mut Context) { let account_id_clone = account_id.to_string(); self.loading_trends.insert(account_id.to_string(), true); @@ -603,11 +604,9 @@ impl DashboardView { return; }; - let (tx, rx) = std::sync::mpsc::channel::>(); + let (tx, rx) = std::sync::mpsc::channel::, String>>(); std::thread::spawn(move || { - use chrono::{Datelike, Duration, Utc}; - let Some(descriptor) = account.descriptor() else { let _ = tx.send(Err("Unknown billing source".to_string())); return; @@ -621,35 +620,11 @@ impl DashboardView { return; }; - let now = Utc::now(); - let start = now - Duration::days(days); - let start_date = format!("{}-{:02}-{:02}", start.year(), start.month(), start.day()); - let end_date = format!("{}-{:02}-{:02}", now.year(), now.month(), now.day()); - - // Try to get from cache first - if let Ok(Some(cached)) = - crate::db::get_cached_cost_trend(&account.id, &start_date, &end_date) - { - let _ = tx.send(Ok(cached)); - return; - } - - let service = (descriptor.build)(account.context(descriptor)); - match service.get_cost_trend(&start_date, &end_date) { - Ok(trend) => { - // Save to cache - if let Err(e) = crate::db::save_cost_trend_cache(&trend) { - tracing::warn!("Failed to save trend cache: {}", e); - } - let _ = tx.send(Ok(trend)); - } - Err(e) => { - let _ = tx.send(Err(format!( - "Failed to get {} trend data: {}", - descriptor.short_name, e - ))); - } - } + let _ = tx.send( + report::trend(&account, days, chrono::Utc::now()).map_err(|e| { + format!("Failed to read {} trend data: {}", descriptor.short_name, e) + }), + ); }); let account_id_for_update = account_id.to_string(); @@ -665,8 +640,11 @@ impl DashboardView { this.loading_trends .insert(account_id_for_update.clone(), false); - if let Ok(trend) = result { - this.cost_trends.insert(account_id_for_update, trend); + match result { + Ok(daily_costs) => { + this.cost_trends.insert(account_id_for_update, daily_costs); + } + Err(e) => tracing::warn!("{}", e), } cx.notify(); }) diff --git a/src/ui/settings.rs b/src/ui/settings.rs index 49ea5b7..1bb9f4c 100644 --- a/src/ui/settings.rs +++ b/src/ui/settings.rs @@ -2,9 +2,9 @@ use gpui::prelude::FluentBuilder; use gpui::*; -use gpui_component::{switch::*, *}; +use gpui_component::{button::*, switch::*, *}; -use crate::config::{load_config, save_config, AppConfig}; +use crate::config::{load_config, save_config, AppConfig, SUPPORTED_REPORTING_CURRENCIES}; /// Settings View pub struct SettingsView { @@ -38,6 +38,23 @@ impl SettingsView { self.save_config(cx); } + /// Change the currency every amount is shown in. + /// + /// Only the reading view is rebuilt — charges stay in the currency + /// they were billed in, so this costs nothing and loses nothing. + fn set_reporting_currency(&mut self, currency: &str, cx: &mut Context) { + self.config.reporting_currency = currency.to_string(); + + if let Err(e) = crate::ledger::set_reporting_currency(currency) { + tracing::error!("Failed to switch reporting currency: {}", e); + self.save_status = Some(format!("Could not switch currency: {}", e)); + cx.notify(); + return; + } + + self.save_config(cx); + } + fn save_config(&mut self, cx: &mut Context) { match save_config(&self.config) { Ok(_) => { @@ -79,6 +96,7 @@ impl SettingsView { impl Render for SettingsView { fn render(&mut self, _window: &mut Window, cx: &mut Context) -> impl IntoElement { let dark_mode = self.config.theme.dark_mode; + let reporting_currency = self.config.reporting_currency.clone(); div() .size_full() @@ -119,6 +137,40 @@ impl Render for SettingsView { cx, ), ) + // Reporting currency + .child( + self.render_section( + "Reporting", + div() + .h_flex() + .justify_between() + .items_center() + .child( + div().v_flex().child(div().child("Currency")).child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child( + "Totals are converted to this currency. \ + Charges keep the currency they were billed in.", + ), + ), + ) + .child(div().h_flex().gap_2().children( + SUPPORTED_REPORTING_CURRENCIES.iter().map(|currency| { + Button::new(SharedString::from(format!("currency-{currency}"))) + .label(*currency) + .when(*currency == reporting_currency, |button| { + button.primary() + }) + .on_click(cx.listener(move |this, _, _, cx| { + this.set_reporting_currency(currency, cx); + })) + }), + )), + cx, + ), + ) // About .child( self.render_section(