diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9ce36ea..3a4b64f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,9 +2,10 @@ name: CI on: push: - branches: [main] + branches: [main, "feat/**"] pull_request: branches: [main] + workflow_dispatch: jobs: test: @@ -12,7 +13,7 @@ jobs: strategy: fail-fast: false matrix: - python-version: ["3.9", "3.10", "3.11", "3.12"] + python-version: ["3.10", "3.11", "3.12"] steps: - name: Checkout @@ -26,10 +27,14 @@ jobs: - name: Install dependencies run: | python -m pip install --upgrade pip - python -m pip install -e ".[all]" pytest pytest-cov + python -m pip install -e . networkx pytest pytest-cov ruff mypy build - name: Run open-source guard run: python scripts/opensource_guard.py - name: Run tests run: pytest -q + + - name: Run release gates + if: matrix.python-version == '3.12' + run: PYTHON=python bash tools/ci_checks.sh diff --git a/.github/workflows/scale.yml b/.github/workflows/scale.yml new file mode 100644 index 0000000..cad713e --- /dev/null +++ b/.github/workflows/scale.yml @@ -0,0 +1,22 @@ +name: Scale Benchmarks + +on: + workflow_dispatch: + schedule: + - cron: "0 3 * * 1" + +jobs: + benchmark: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + - run: python -m pip install --upgrade pip + - run: python -m pip install -e . + - run: python tools/benchmark_scale.py --counts 1000,10000 --output .benchmarks/result.json + - uses: actions/upload-artifact@v4 + with: + name: scale-benchmark + path: .benchmarks/result.json diff --git a/.gitignore b/.gitignore index 921b60f..ba6a4df 100644 --- a/.gitignore +++ b/.gitignore @@ -20,6 +20,7 @@ wheels/ *.egg .pytest_cache/ .coverage +coverage.json htmlcov/ .tox/ .venv @@ -49,6 +50,12 @@ output_csv/ sqlgraph_output/ failed_cases.json smoke_result.json +*.duckdb +.benchmarks/ +comparison-screenshots/ +generated_illustrations*/ +/book*/ +*.zip # Large or private local datasets df.csv diff --git a/CAPABILITIES.yaml b/CAPABILITIES.yaml new file mode 100644 index 0000000..ea4b5ff --- /dev/null +++ b/CAPABILITIES.yaml @@ -0,0 +1,86 @@ +schema_version: capabilities-v1 +capabilities: + - id: input-baseline + status: implemented + description: Versioned SQL, dialect, schema, UDF, parameter, and scheduler baseline. + modules: [sqlgraph/baseline, sqlgraph/input] + tests: [tests/contract/test_baseline_manifest.py] + examples: [examples/minimal/scenario.yaml] + limitations: [Dynamic SQL must be materialized before parsing.] + - id: deterministic-sql-compiler + status: implemented + description: Statement-aware SQLGlot compilation with stable graph identities. + modules: [sqlgraph/parser, sqlgraph/builder, sqlgraph/identity] + tests: [tests/test_identity/test_determinism.py, tests/test_parser] + examples: [examples/ads_pipeline] + limitations: [Unresolvable multi-source columns are marked UNKNOWN.] + - id: two-level-graph + status: implemented + description: Heterogeneous graph plus reconstructable physical TableGraph projection. + modules: [sqlgraph/model, sqlgraph/analyze/table_graph.py, sqlgraph/lineage] + tests: [tests/test_graph_advanced/test_drilldown.py] + examples: [examples/video_commercial_warehouse] + limitations: [Nested struct fields remain attached to top-level physical columns.] + - id: evidence-subgraph + status: implemented + description: Intent-bound evidence collection, expansion, versioning, and stable hashes. + modules: [sqlgraph/evidence] + tests: [tests/test_evidence/test_engine.py] + examples: [examples/book_cases/caliber_consistency] + limitations: [External facts must be supplied by the caller.] + - id: evidence-discipline + status: implemented + description: Coverage obligations, counterevidence, exclusions, gaps, and residual unknowns. + modules: [sqlgraph/evidence] + tests: [tests/safety/test_evidence_gates.py] + examples: [examples/book_cases/caliber_consistency] + limitations: [Counterevidence search is structural rather than semantic.] + - id: offline-governance-metrics + status: implemented + description: Versioned inventory, topology, impact, consistency, similarity, motif, and anomaly analysis. + modules: [sqlgraph/analyze, sqlgraph/metrics] + tests: [tests/test_analyze] + examples: [examples/ads_pipeline] + limitations: [Optional embedding and model-based anomaly metrics require extras.] + - id: seven-step-loop + status: implemented + description: Observe, Explain, Propose, Authorize, Execute, Verify, and Learn orchestration. + modules: [sqlgraph/reasoning, sqlgraph/agent] + tests: [tests/test_reasoning/test_runner.py] + examples: [examples/minimal/scenario.yaml] + limitations: [Learn writes signals only and never expands authorization.] + - id: autonomy-policy + status: implemented + description: Evidence, authorization, reversibility, impact, risk, and reliability gates. + modules: [sqlgraph/autonomy] + tests: [tests/test_autonomy, tests/safety/test_autonomy_gates.py] + examples: [examples/book_cases/irreversible_drop] + limitations: [Thresholds are reference defaults and must be calibrated per deployment.] + - id: reversible-actions + status: implemented + description: Idempotent file and DuckDB actions with dry run, circuit breaker, and verified rollback. + modules: [sqlgraph/actions] + tests: [tests/test_actions, tests/safety/test_execution_recovery.py] + examples: [examples/book_cases/cold_table_retirement] + limitations: [Only local SQL files and DuckDB are included adapters.] + - id: three-layer-verification + status: implemented + description: Actual code comparison, independent graph rebuild, and runtime checks. + modules: [sqlgraph/verification, sqlgraph/verify] + tests: [tests/test_verification, tests/safety/test_verification_failure.py] + examples: [examples/book_cases/caliber_consistency] + limitations: [Production runtime checks require a deployment-specific adapter.] + - id: graphrag-grounding + status: implemented + description: Assertion-to-citation validation constrained by baseline and evidence versions. + modules: [sqlgraph/graphrag, sqlgraph/serialize/graphrag.py] + tests: [tests/test_graphrag] + examples: [examples/minimal/scenario.yaml] + limitations: [The repository validates supplied assertions but does not bundle an LLM.] + - id: audit-replay + status: implemented + description: Append-only hash-chained events with integrity verification and task replay. + modules: [sqlgraph/audit] + tests: [tests/test_audit, tests/safety/test_audit_integrity.py] + examples: [examples/minimal/scenario.yaml] + limitations: [Local JSONL storage is a reference adapter, not a distributed ledger.] diff --git a/README.md b/README.md index f28dac2..c1b2582 100644 --- a/README.md +++ b/README.md @@ -89,7 +89,7 @@ pip install -e . pip install sqlgraph-lineage ``` -Requires Python 3.9 through 3.12. +Requires Python 3.10 through 3.12. ## Quick start @@ -130,6 +130,22 @@ sqlgraph playground sqlgraph stats ./sql --dialect spark ``` +### Evidence-bound governance quickstart + +Run the minimal book scenario and verify its complete evidence package: + +```bash +sqlgraph governance run examples/minimal/scenario.yaml -o demo_output/minimal +sqlgraph governance verify demo_output/minimal +sqlgraph governance replay demo_output/minimal/audit.jsonl +``` + +The output contains the deterministic input baseline, graph snapshot, evidence +bundle, autonomy decision, action result, independent verification report, and a +hash-chained seven-step audit log. See [CAPABILITIES.yaml](CAPABILITIES.yaml) for +the status and test evidence of every advertised capability, and +[docs/limitations.md](docs/limitations.md) for the supported boundary. + ### Lineage Explorer (search + local subgraphs) For large inputs, avoid one giant HTML file. Serve a searchable explorer instead: diff --git a/README.zh-CN.md b/README.zh-CN.md index 6a0929e..2f7dfab 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -54,7 +54,7 @@ pip install -e . pip install sqlgraph-lineage ``` -需要 Python 3.9 到 3.12。 +需要 Python 3.10 到 3.12。 ## 快速开始 @@ -95,6 +95,19 @@ sqlgraph playground sqlgraph stats ./sql --dialect spark ``` +### 证据约束治理 Quickstart + +```bash +sqlgraph governance run examples/minimal/scenario.yaml -o demo_output/minimal +sqlgraph governance verify demo_output/minimal +sqlgraph governance replay demo_output/minimal/audit.jsonl +``` + +输出包含确定性输入基线、图快照、证据包、自治决议、动作结果、独立验证报告, +以及带哈希链的七步审计日志。所有对外能力的状态与测试证据见 +[`CAPABILITIES.yaml`](CAPABILITIES.yaml),支持边界见 +[`docs/limitations.md`](docs/limitations.md)。 + ### 血缘检索浏览器(检索 + 局部子图) 大规模输入不再依赖单个巨大 HTML,改用可检索的本地浏览器: diff --git a/docs/architecture.md b/docs/architecture.md index 5c6e71b..4d4d7fe 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -94,3 +94,24 @@ An in-memory **`PropertyGraph`** of typed nodes and edges. text, so lookalike expressions on different columns stay distinct. - **Scale** — 96/128-bit fingerprints keep collisions negligible even for warehouses with millions of columns. + +## Governance reference pipeline + +The governance modules build on the graph without coupling the compiler to an +execution environment: + +```text +Baseline -> HeteroGraph/TableGraph -> Evidence/Grounding -> Autonomy + -> Action Adapter -> Independent Verification -> Audit Replay +``` + +- `baseline` identifies the effective input and discloses missing dependencies. +- `evidence` collects an intent-bound, versioned subgraph with coverage duties. +- `graphrag` rejects assertions with missing, stale, or out-of-scope citations. +- `autonomy` applies evidence, authorization, and reversibility gates before scoring. +- `actions` provides idempotent file and DuckDB adapters with verified rollback. +- `verification` compares actual code, rebuilds structure, and checks runtime effects. +- `audit` appends hash-chained events; `reasoning` orchestrates the seven steps. + +No analysis score directly authorizes an action. The action and verification +adapters are explicit seams for production integrations. diff --git a/docs/book-map.md b/docs/book-map.md new file mode 100644 index 0000000..0b87141 --- /dev/null +++ b/docs/book-map.md @@ -0,0 +1,20 @@ +# Book to Code Map + +| Book topic | Module | Tests | Observable artifact | +|---|---|---|---| +| Object and provenance model | `sqlgraph.baseline`, `sqlgraph.model` | `tests/contract/test_baseline_manifest.py` | `baseline.json`, `graph.json` | +| Semantic runtime | `sqlgraph.parser`, `sqlgraph.builder`, `sqlgraph.analyze` | `tests/test_identity`, `tests/test_analyze` | deterministic graph and analysis snapshot | +| Evidence chain | `sqlgraph.evidence`, `sqlgraph.graphrag` | `tests/test_evidence`, `tests/test_graphrag` | `evidence.json` | +| Graded autonomy | `sqlgraph.autonomy` | `tests/test_autonomy` | `decision.json` | +| Reversibility red line | `sqlgraph.actions` | `tests/safety/test_execution_recovery.py` | execution and rollback events | +| Seven-step loop | `sqlgraph.reasoning` | `tests/test_reasoning/test_runner.py` | seven hash-chained audit events | +| Structural insight | `sqlgraph.analyze` | `tests/test_analyze` | profile and governance snapshot | +| Caliber consistency | `examples/book_cases/caliber_consistency` | `tests/test_integration/test_book_cases.py` | successful L3 repair | +| Cost governance | `examples/book_cases/cold_table_retirement` | `tests/test_integration/test_book_cases.py` | reversible quarantine proposal | +| Verification | `sqlgraph.verification` | `tests/test_verification` | `verification.json` | +| Three objections | `sqlgraph.graphrag`, `sqlgraph.autonomy`, `sqlgraph.audit` | `tests/safety` | rejection, escalation, and replay records | +| Engineering implementation | `sqlgraph.cli`, `sqlgraph.serve` | `tests/test_cli`, `tests/test_serve` | CLI package and Explorer | +| Boundaries and future work | `CAPABILITIES.yaml`, `docs/limitations.md` | `tests/contract/test_capabilities.py` | explicit capability status | + +Claims without a module, current test, and observable artifact must be marked +`experimental`, `planned`, or `concept-only` in `CAPABILITIES.yaml`. diff --git a/docs/limitations.md b/docs/limitations.md new file mode 100644 index 0000000..b585284 --- /dev/null +++ b/docs/limitations.md @@ -0,0 +1,22 @@ +# Limitations + +- SQL parsing depends on SQLGlot. Unsupported or dynamic SQL must be materialized + before analysis; the system does not execute templates to guess their output. +- Multi-source columns without enough schema information are marked `UNKNOWN`. + Evidence containing them cannot pass the `no_unresolved` obligation. +- Expression fingerprints are conservative. Algebraically equivalent expressions + such as reordered addition are intentionally not merged. +- Counterevidence search is structural. Domain counterexamples and external facts + must be supplied through adapters. +- The included action adapters support local SQL files and DuckDB only. They are + reference implementations, not production warehouse credentials or schedulers. +- JSONL audit storage detects local mutation, deletion, and reordering but is not + a distributed consensus ledger. +- Runtime verification trusts the configured runtime adapter's observations. + Production deployments must isolate that adapter from the execution path. +- Embeddings and model-based anomaly detection are optional analysis features and + cannot authorize governance actions. +- The ordinary CI scale smoke test uses 100 generated SQL statements. The 1,000 + and 10,000 statement benchmarks run in a separate workflow. +- SqlGraph does not bundle an LLM, approval platform, identity provider, metadata + catalog, or production scheduler. diff --git a/docs/plans/2026-08-25-book-governance-reference-release.md b/docs/plans/2026-08-25-book-governance-reference-release.md new file mode 100644 index 0000000..f89689f --- /dev/null +++ b/docs/plans/2026-08-25-book-governance-reference-release.md @@ -0,0 +1,934 @@ +# AI 原生数仓治理参考实现发布 Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** 在不破坏 GitHub 现有 Explorer 和离线分析能力的前提下,交付满足六条对外发布门槛的 AI 原生数仓治理参考实现,并将功能分支推送到 GitHub。 + +**Architecture:** 以 `origin/main` 为唯一整合基线,先补确定性 baseline 和可下钻事实图,再通过 Evidence、Grounding、Autonomy、Actions、Verification、Audit 六个深模块形成受约束闭环。`GovernanceRunner` 只做编排;Explorer 和离线报告共享 GitHub CSS,报告在渲染时将 CSS 内联以保持单文件离线运行。 + +**Tech Stack:** Python 3.9-3.12、SQLGlot、Typer、Jinja2、DuckDB、JSON Schema、pytest、Ruff、mypy、原生 HTML/CSS/JavaScript。 + +--- + +## 文件结构 + +新增模块: + +```text +sqlgraph/ +├── baseline/ # 输入清单、baseline_id、缺失依赖 +├── evidence/ # 证据收集、扩图、充分性 +├── graphrag/ # 断言与结构引用校验 +├── autonomy/ # 自治闸门和三维决议 +├── actions/ # 动作计划、adapter、幂等、熔断、回滚 +├── verification/ # 代码、独立结构、运行态验证 +├── audit/ # JSONL 哈希链和重放 +└── reasoning/ # 七步 GovernanceRunner +``` + +新增发布资产: + +```text +schemas/ +examples/minimal/ +examples/book_cases/ +tests/contract/ +tests/safety/ +tests/scale/ +docs/book-map.md +docs/theory-vs-implementation.md +docs/limitations.md +CAPABILITIES.yaml +``` + +兼容层: + +- `sqlgraph.agent` 委托 `sqlgraph.reasoning` +- `sqlgraph.verify` 委托 `sqlgraph.verification` +- 现有 `build`、`stats`、`analyze`、`profile`、`serve`、`playground`、`demo` 保持兼容 + +--- + +### Task 1: 固化远端基线与回归保护 + +**Files:** +- Modify: `.gitignore` +- Modify: `.github/workflows/ci.yml` +- Create: `tests/test_release/test_remote_capabilities.py` +- Create: `tests/test_release/__init__.py` + +- [ ] **Step 1: 写远端能力回归测试** + +```python +from typer.testing import CliRunner + +from sqlgraph.cli import app + + +def test_public_cli_keeps_existing_github_commands(): + result = CliRunner().invoke(app, ["--help"]) + assert result.exit_code == 0 + for command in ("build", "stats", "analyze", "profile", "serve", "playground", "demo"): + assert command in result.stdout +``` + +- [ ] **Step 2: 运行测试并确认当前远端基线通过** + +Run: `python -m pytest tests/test_release/test_remote_capabilities.py -q` + +Expected: PASS。 + +- [ ] **Step 3: 扩充生成物忽略规则** + +加入: + +```gitignore +*.duckdb +demo_output/ +comparison-screenshots/ +generated_illustrations*/ +book*/ +*.zip +.benchmarks/ +``` + +- [ ] **Step 4: 运行远端完整测试** + +Run: `python -m pytest -q` + +Expected: 远端测试全部通过。 + +- [ ] **Step 5: 提交** + +```bash +git add .gitignore .github/workflows/ci.yml tests/test_release +git commit -m "test: protect github baseline capabilities" +``` + +### Task 2: 移植确定性编译与多语句事实层 + +**Files:** +- Create: `sqlgraph/identity/__init__.py` +- Modify: `sqlgraph/input/sql_source.py` +- Modify: `sqlgraph/parser/base.py` +- Modify: `sqlgraph/parser/expr_dag.py` +- Modify: `sqlgraph/builder/graph_builder.py` +- Modify: `sqlgraph/model/nodes.py` +- Modify: `sqlgraph/model/edges.py` +- Modify: `sqlgraph/model/graph.py` +- Create: `tests/test_identity/test_determinism.py` +- Create: `tests/test_identity/__init__.py` +- Create: `tests/test_graph_advanced/test_statement_lineage.py` +- Create: `tests/test_graph_advanced/__init__.py` + +- [ ] **Step 1: 写确定性和多语句失败测试** + +```python +def test_two_builds_have_identical_full_ids(): + sql = "INSERT INTO d SELECT a + b AS c FROM s" + assert all_ids(build_graph(sql, dialect="spark")) == all_ids( + build_graph(sql, dialect="spark") + ) + + +def test_independent_statements_do_not_cross_link(): + graph = build_graph( + "INSERT INTO d1 SELECT * FROM s1; INSERT INTO d2 SELECT * FROM s2;", + dialect="spark", + ) + assert table_pairs(graph) == {("s1", "d1"), ("s2", "d2")} +``` + +- [ ] **Step 2: 运行测试确认缺失能力** + +Run: `python -m pytest tests/test_identity tests/test_graph_advanced/test_statement_lineage.py -q` + +Expected: FAIL,边或 SQL 身份不稳定,或多语句归属不完整。 + +- [ ] **Step 3: 实现稳定身份与语句归属** + +实现: + +```python +IDENTITY_RULE_VERSION = "id-v2" + + +def stable_id(prefix: str, semantic_key: str, bits: int = 96) -> str: + width = bits // 4 + digest = hashlib.sha256(semantic_key.encode("utf-8")).hexdigest()[:width] + return f"{prefix}_{digest}" +``` + +每条语句保存 `stmt_index`;所有边 ID 使用源、目标、类型和语句上下文派生。表达式继续使用远端保守规范化,并增加 `expr_operand` 子表达式边,不引入交换律等价推断。 + +- [ ] **Step 4: 增加来源位置和 UNRESOLVED 诊断** + +节点和诊断属性至少包含: + +```python +{ + "source_name": item.name, + "stmt_index": stmt_index, + "line": line, + "column": column, + "resolution": "resolved" | "unresolved", +} +``` + +- [ ] **Step 5: 运行解析、构建、分析和 Explorer 回归** + +Run: `python -m pytest tests/test_parser tests/test_builder tests/test_analyze tests/test_serve tests/test_identity tests/test_graph_advanced -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add sqlgraph/identity sqlgraph/input sqlgraph/parser sqlgraph/builder sqlgraph/model tests/test_identity tests/test_graph_advanced +git commit -m "feat: add deterministic statement-aware compiler" +``` + +### Task 3: 增加统一 Baseline Registry + +**Files:** +- Create: `sqlgraph/baseline/__init__.py` +- Create: `sqlgraph/baseline/model.py` +- Create: `sqlgraph/baseline/builder.py` +- Create: `schemas/baseline-manifest-v1.schema.json` +- Create: `tests/contract/test_baseline_manifest.py` +- Create: `tests/contract/__init__.py` + +- [ ] **Step 1: 写 baseline 确定性和缺失依赖测试** + +```python +def test_baseline_id_is_stable_and_timestamp_is_not_identity(tmp_path): + first = build_test_baseline(tmp_path) + second = build_test_baseline(tmp_path) + assert first.baseline_id == second.baseline_id + assert first.created_at != "" + + +def test_missing_schema_is_disclosed(): + baseline = build_baseline(SqlSource.from_sql("SELECT a FROM t")) + assert "schema" in baseline.missing_dependencies +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/contract/test_baseline_manifest.py -q` + +Expected: FAIL with `ModuleNotFoundError: sqlgraph.baseline`。 + +- [ ] **Step 3: 实现 BaselineManifest** + +```python +@dataclass(frozen=True) +class BaselineManifest: + schema_version: str + baseline_id: str + source_hashes: tuple[SourceHash, ...] + dialect: str + parser_version: str + identity_rule_version: str + dependency_hashes: Mapping[str, str] + missing_dependencies: tuple[str, ...] + created_at: str +``` + +`baseline_id` 仅由规范化 source hashes、方言、解析器规则、Schema/UDF/参数/调度哈希生成。 + +- [ ] **Step 4: 添加 JSON Schema 并验证产物** + +使用 `jsonschema` 开发依赖验证 `BaselineManifest.to_dict()`;错误字段必须产生可读异常。 + +- [ ] **Step 5: 运行契约测试** + +Run: `python -m pytest tests/contract/test_baseline_manifest.py -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add sqlgraph/baseline schemas tests/contract pyproject.toml +git commit -m "feat: add deterministic input baselines" +``` + +### Task 4: 统一 TableGraph 投影与下钻 + +**Files:** +- Modify: `sqlgraph/analyze/table_graph.py` +- Create: `sqlgraph/lineage/__init__.py` +- Create: `tests/test_graph_advanced/test_drilldown.py` + +- [ ] **Step 1: 写投影可重建测试** + +```python +def test_table_edge_drills_to_statement_columns_and_transforms(): + graph = build_graph( + "INSERT INTO dst SELECT CASE WHEN status='A' THEN amount ELSE 0 END AS value FROM src", + dialect="spark", + ) + result = drilldown(graph, "src", "dst") + assert result.found + assert result.statements + assert result.column_paths[0].source_columns == ("amount", "status") +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_graph_advanced/test_drilldown.py -q` + +Expected: FAIL,当前无统一 `drilldown` 契约。 + +- [ ] **Step 3: 实现 TableGraph 来源映射** + +`TableGraphEdge` 增加: + +```python +statement_refs: tuple[StatementRef, ...] +column_dependency_ids: tuple[str, ...] +transform_ids: tuple[str, ...] +``` + +`drilldown()` 仅从这些引用返回细粒度证据,不按名称重新猜测。 + +- [ ] **Step 4: 运行分析与下钻测试** + +Run: `python -m pytest tests/test_analyze tests/test_graph_advanced/test_drilldown.py -q` + +Expected: PASS。 + +- [ ] **Step 5: 提交** + +```bash +git add sqlgraph/analyze/table_graph.py sqlgraph/lineage tests/test_graph_advanced/test_drilldown.py +git commit -m "feat: make table lineage fully drillable" +``` + +### Task 5: 深化 Evidence Engine + +**Files:** +- Create: `sqlgraph/evidence/__init__.py` +- Create: `sqlgraph/evidence/model.py` +- Create: `sqlgraph/evidence/engine.py` +- Create: `schemas/evidence-bundle-v1.schema.json` +- Create: `tests/test_evidence/test_engine.py` +- Create: `tests/test_evidence/__init__.py` +- Create: `tests/safety/test_evidence_gates.py` +- Create: `tests/safety/__init__.py` + +- [ ] **Step 1: 写覆盖、反证和稳定哈希测试** + +```python +def test_evidence_bundle_contains_support_counterevidence_and_exclusions(): + bundle = engine.collect(request) + assert bundle.supporting + assert bundle.counterevidence + assert bundle.excluded + assert bundle.coverage_contract.required + + +def test_removed_required_evidence_cannot_remain_sufficient(): + bundle = replace(bundle, included_edges=()) + decision = engine.assess(bundle) + assert decision.action in {"expand", "degrade", "refuse", "escalate"} + assert not decision.sufficient +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_evidence tests/safety/test_evidence_gates.py -q` + +Expected: FAIL with missing EvidenceEngine。 + +- [ ] **Step 3: 实现请求、证据包和充分性决议** + +实现不可变数据模型: + +```python +@dataclass(frozen=True) +class EvidenceRequest: + task_id: str + baseline_id: str + intent: str + anchors: tuple[str, ...] + direction: str + max_depth: int + coverage_obligations: tuple[str, ...] + required_external_facts: tuple[str, ...] = () +``` + +`subgraph_hash` 对 included、excluded、支持、反对、缺口和规则版本规范化后使用 SHA256。 + +- [ ] **Step 4: 实现扩图和判停** + +`expand()` 必须返回新版本,不可原地篡改;`assess()` 在存在未满足覆盖义务、UNKNOWN/UNRESOLVED 或外部事实缺失时禁止 `stop`。 + +- [ ] **Step 5: 运行证据和安全测试** + +Run: `python -m pytest tests/test_evidence tests/safety/test_evidence_gates.py -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add sqlgraph/evidence schemas/evidence-bundle-v1.schema.json tests/test_evidence tests/safety +git commit -m "feat: enforce evidence coverage and counterevidence" +``` + +### Task 6: 增加 GraphRAG 引用校验 + +**Files:** +- Create: `sqlgraph/graphrag/__init__.py` +- Create: `sqlgraph/graphrag/grounding.py` +- Create: `tests/test_graphrag/test_grounding.py` +- Create: `tests/test_graphrag/__init__.py` +- Modify: `sqlgraph/serialize/graphrag.py` + +- [ ] **Step 1: 写伪造引用拒绝测试** + +```python +def test_nonexistent_reference_rejects_assertion(): + report = validate_assertions( + [GroundedAssertion("dst depends on src", citations=("edge_missing",))], + evidence, + ) + assert report.status == "rejected" + assert report.invalid_citations == ("edge_missing",) +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_graphrag/test_grounding.py -q` + +Expected: FAIL with missing grounding module。 + +- [ ] **Step 3: 实现 GroundedAssertion 和 GroundingReport** + +校验 citation 存在性、baseline/evidence 版本一致性和 coverage 射程。无 citation 的结构性断言状态为 `rejected`。 + +- [ ] **Step 4: 运行 GraphRAG 测试** + +Run: `python -m pytest tests/test_graphrag tests/test_serialize -q` + +Expected: PASS。 + +- [ ] **Step 5: 提交** + +```bash +git add sqlgraph/graphrag sqlgraph/serialize/graphrag.py tests/test_graphrag +git commit -m "feat: validate graphrag assertions against evidence" +``` + +### Task 7: 移植自治策略并绑定证据版本 + +**Files:** +- Create: `sqlgraph/autonomy/__init__.py` +- Create: `sqlgraph/autonomy/decision.py` +- Create: `schemas/autonomy-decision-v1.schema.json` +- Create: `tests/test_autonomy/test_decision.py` +- Create: `tests/test_autonomy/__init__.py` +- Create: `tests/safety/test_autonomy_gates.py` + +- [ ] **Step 1: 写表驱动安全测试** + +```python +@pytest.mark.parametrize( + ("reversible", "grounded", "scope", "expected"), + [ + (False, True, "continuous_l5", "L4"), + (True, False, "single_l3", "L0"), + (True, True, "none", "L2"), + (True, True, "single_l3", "L3"), + ], +) +def test_hard_gates_precede_scoring(reversible, grounded, scope, expected): + decision = decide_autonomy(action(reversible, grounded, scope, roi=999)) + assert decision.level.value == expected +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_autonomy tests/safety/test_autonomy_gates.py -q` + +Expected: FAIL with missing autonomy module。 + +- [ ] **Step 3: 实现三维决议和可逆性三证据** + +`GovernanceAction` 使用: + +```python +reversibility = ReversibilityEvidence( + state_restorable=True, + external_effects_controlled=True, + rollback_verified=True, +) +``` + +输出包含 `level`、`requires_human_review`、`within_l5_scope`、`policy_version`、`evidence_version` 和命中闸门。 + +- [ ] **Step 4: 运行测试** + +Run: `python -m pytest tests/test_autonomy tests/safety/test_autonomy_gates.py -q` + +Expected: PASS。 + +- [ ] **Step 5: 提交** + +```bash +git add sqlgraph/autonomy schemas/autonomy-decision-v1.schema.json tests/test_autonomy tests/safety/test_autonomy_gates.py +git commit -m "feat: enforce evidence and reversibility autonomy gates" +``` + +### Task 8: 实现 Action Engine、幂等和回滚 + +**Files:** +- Create: `sqlgraph/actions/__init__.py` +- Create: `sqlgraph/actions/model.py` +- Create: `sqlgraph/actions/engine.py` +- Create: `sqlgraph/actions/adapters.py` +- Create: `schemas/action-plan-v1.schema.json` +- Create: `tests/test_actions/test_engine.py` +- Create: `tests/test_actions/__init__.py` +- Create: `tests/safety/test_execution_recovery.py` + +- [ ] **Step 1: 写幂等、故障和回滚测试** + +```python +def test_repeat_execution_is_noop(engine, plan): + first = engine.execute(plan) + second = engine.execute(plan) + assert first.status == "success" + assert second.status == "noop" + assert second.execution_id == first.execution_id + + +def test_injected_failure_opens_circuit_and_restores_snapshot(engine, failing_plan): + result = engine.execute(failing_plan) + assert result.status == "rolled_back" + assert result.circuit_open + assert result.rollback.verified +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_actions tests/safety/test_execution_recovery.py -q` + +Expected: FAIL with missing ActionEngine。 + +- [ ] **Step 3: 实现动作模型和 engine** + +`ActionPlan` 必须包含 `task_id`、`baseline_id`、`evidence_version`、`decision_id`、`idempotency_key`、`adapter`、`operations` 和 `rollback_plan`。 + +- [ ] **Step 4: 实现两个 adapter** + +`SqlFilePatchAdapter` 使用前置 SHA256 和备份恢复;`DuckDBTaskAdapter` 使用数据库快照、任务白名单和故障注入点。二者都实现 `dry_run/execute/rollback/verify_rollback`。 + +- [ ] **Step 5: 运行测试** + +Run: `python -m pytest tests/test_actions tests/safety/test_execution_recovery.py -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add sqlgraph/actions schemas/action-plan-v1.schema.json tests/test_actions tests/safety/test_execution_recovery.py +git commit -m "feat: add idempotent actions and verified rollback" +``` + +### Task 9: 实现独立三层 Verification + +**Files:** +- Create: `sqlgraph/verification/__init__.py` +- Create: `sqlgraph/verification/model.py` +- Create: `sqlgraph/verification/engine.py` +- Create: `schemas/verification-report-v1.schema.json` +- Create: `sqlgraph/verify/__init__.py` +- Create: `tests/test_verification/test_engine.py` +- Create: `tests/test_verification/__init__.py` +- Create: `tests/safety/test_verification_failure.py` + +- [ ] **Step 1: 写独立重抽和失败决议测试** + +```python +def test_structure_verification_rebuilds_from_changed_source(tmp_path): + report = verifier.verify(plan, execution) + assert report.structure.evidence["rebuilt_from_source"] is True + + +def test_any_critical_failure_blocks_success(): + report = VerificationReport(code=passed(), structure=failed(), runtime=passed()) + assert report.outcome == "failed" +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_verification tests/safety/test_verification_failure.py -q` + +Expected: FAIL with missing verification module。 + +- [ ] **Step 3: 实现 code/structure/runtime verifier** + +结构 verifier 从变更后的 SQL 路径调用 `build_baseline()` 和 `build_graph()`,不得读取执行阶段保存的图成功标志。 + +- [ ] **Step 4: 提供旧接口兼容层** + +`sqlgraph.verify.verify()` 构造新 verifier 所需的只读输入并返回兼容字典,不复制验证规则。 + +- [ ] **Step 5: 运行验证和分析回归** + +Run: `python -m pytest tests/test_verification tests/safety/test_verification_failure.py tests/test_analyze -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add sqlgraph/verification sqlgraph/verify schemas/verification-report-v1.schema.json tests/test_verification tests/safety/test_verification_failure.py +git commit -m "feat: add independent three-layer verification" +``` + +### Task 10: 实现追加式 Audit 与 GovernanceRunner + +**Files:** +- Create: `sqlgraph/audit/__init__.py` +- Create: `sqlgraph/audit/model.py` +- Create: `sqlgraph/audit/log.py` +- Create: `schemas/audit-event-v1.schema.json` +- Create: `sqlgraph/reasoning/__init__.py` +- Create: `sqlgraph/reasoning/runner.py` +- Create: `sqlgraph/agent/__init__.py` +- Create: `tests/test_audit/test_log.py` +- Create: `tests/test_audit/__init__.py` +- Create: `tests/test_reasoning/test_runner.py` +- Create: `tests/test_reasoning/__init__.py` +- Create: `tests/safety/test_audit_integrity.py` + +- [ ] **Step 1: 写哈希链、篡改和七步重放测试** + +```python +def test_tampered_event_breaks_integrity(audit_path): + log = AuditLog(audit_path) + append_three_events(log) + tamper_second_line(audit_path) + assert not log.verify_integrity().valid + + +def test_runner_exports_all_seven_steps(runner): + result = runner.run(request) + assert [event.step for event in result.events] == [ + "observe", "explain", "propose", "authorize", "execute", "verify", "learn" + ] +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_audit tests/test_reasoning tests/safety/test_audit_integrity.py -q` + +Expected: FAIL with missing audit/reasoning modules。 + +- [ ] **Step 3: 实现 canonical JSON 哈希链** + +`event_hash = SHA256(canonical_json(event_without_hash))`,每条事件保存 `previous_hash`。首事件使用 64 个零字符。 + +- [ ] **Step 4: 实现 GovernanceRunner** + +runner 通过构造参数接收 Baseline、Evidence、Autonomy、Action、Verification 和 Audit 模块。证据不足在 `authorize` 前停止;执行或验证失败时记录熔断和回滚事件。 + +- [ ] **Step 5: 增加兼容入口并运行测试** + +Run: `python -m pytest tests/test_audit tests/test_reasoning tests/safety/test_audit_integrity.py -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add sqlgraph/audit sqlgraph/reasoning sqlgraph/agent schemas/audit-event-v1.schema.json tests/test_audit tests/test_reasoning tests/safety/test_audit_integrity.py +git commit -m "feat: add replayable governance audit chain" +``` + +### Task 11: 增加治理 CLI 和三条书中案例 + +**Files:** +- Modify: `sqlgraph/cli.py` +- Create: `examples/minimal/scenario.yaml` +- Create: `examples/minimal/sql/source.sql` +- Create: `examples/book_cases/caliber_consistency/scenario.yaml` +- Create: `examples/book_cases/cold_table_retirement/scenario.yaml` +- Create: `examples/book_cases/irreversible_drop/scenario.yaml` +- Create: `sqlgraph/reasoning/scenario.py` +- Create: `tests/test_integration/test_book_cases.py` +- Create: `tests/test_cli/test_governance_cli.py` + +- [ ] **Step 1: 写 CLI 和场景验收测试** + +```python +@pytest.mark.parametrize( + ("case", "outcome"), + [ + ("caliber_consistency", "success"), + ("cold_table_retirement", "success"), + ("irreversible_drop", "held_for_human_review"), + ], +) +def test_book_case_exports_replayable_bundle(case, outcome, tmp_path): + result = run_case(case, tmp_path) + assert result.outcome == outcome + assert AuditLog(result.audit_path).verify_integrity().valid +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_cli/test_governance_cli.py tests/test_integration/test_book_cases.py -q` + +Expected: FAIL,`governance` 命令和场景不存在。 + +- [ ] **Step 3: 实现 YAML 场景加载与 CLI** + +新增: + +```text +sqlgraph governance run -o +sqlgraph governance replay +sqlgraph governance verify +``` + +场景只允许引用注册 adapter 和动作模板,不执行任意 Python。 + +- [ ] **Step 4: 运行三条案例** + +Run: `python -m pytest tests/test_cli/test_governance_cli.py tests/test_integration/test_book_cases.py -q` + +Expected: PASS。 + +- [ ] **Step 5: 提交** + +```bash +git add sqlgraph/cli.py sqlgraph/reasoning/scenario.py examples/minimal examples/book_cases tests/test_cli/test_governance_cli.py tests/test_integration/test_book_cases.py +git commit -m "feat: add runnable book governance cases" +``` + +### Task 12: 移植视频商业化数仓与 GitHub 样式工作台 + +**Files:** +- Create: `examples/video_commercial_warehouse/` +- Create: `tests/test_video_warehouse/` +- Create: `sqlgraph/serve/theme.py` +- Modify: `examples/video_commercial_warehouse/report.py` +- Modify: `examples/video_commercial_warehouse/templates/workbench.html` +- Modify: `sqlgraph/serve/web/static/app.css` + +- [ ] **Step 1: 写共享样式和离线报告测试** + +```python +def test_workbench_inlines_explorer_theme(tmp_path): + html = render_demo_report(tmp_path).read_text() + assert "--bg: #0b1020" in html or "--bg:#0b1020" in html + assert "SqlGraph Explorer" in html + assert "https://" not in html + assert 'class="app-tabs product-nav"' in html +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/test_video_warehouse/test_report.py -q` + +Expected: FAIL,综合案例尚未移植。 + +- [ ] **Step 3: 移植 94 表、62 任务 DuckDB 案例** + +只移植 `examples/video_commercial_warehouse/` 和对应测试;数据库、生成报告和截图不进入 Git。 + +- [ ] **Step 4: 共享 Explorer CSS** + +`sqlgraph.serve.theme.load_explorer_css()` 读取包内 `app.css`。报告渲染器将其内容注入 `__EXPLORER_CSS__`,再注入工作台专用 CSS,确保离线单文件运行。 + +- [ ] **Step 5: 运行报告和浏览器测试** + +Run: `python -m pytest tests/test_video_warehouse tests/test_serve -q` + +Expected: PASS。浏览器验证 1440x1000 和 390x844 无横向溢出。 + +- [ ] **Step 6: 提交** + +```bash +git add examples/video_commercial_warehouse tests/test_video_warehouse sqlgraph/serve/theme.py sqlgraph/serve/web/static/app.css +git commit -m "feat: add github-aligned governance workbench" +``` + +### Task 13: 建立能力账本和书—代码映射 + +**Files:** +- Create: `CAPABILITIES.yaml` +- Create: `schemas/capabilities-v1.schema.json` +- Create: `docs/book-map.md` +- Create: `docs/theory-vs-implementation.md` +- Create: `docs/limitations.md` +- Modify: `README.md` +- Modify: `README.zh-CN.md` +- Modify: `docs/architecture.md` +- Create: `tools/check_capabilities.py` +- Create: `tests/contract/test_capabilities.py` + +- [ ] **Step 1: 写能力账本路径真实性测试** + +```python +def test_implemented_capabilities_reference_existing_assets(): + ledger = load_capabilities("CAPABILITIES.yaml") + for capability in ledger["capabilities"]: + if capability["status"] == "implemented": + assert all(Path(path).exists() for path in capability["modules"]) + assert all(Path(path).exists() for path in capability["tests"]) + assert all(Path(path).exists() for path in capability["examples"]) +``` + +- [ ] **Step 2: 运行测试确认失败** + +Run: `python -m pytest tests/contract/test_capabilities.py -q` + +Expected: FAIL,能力账本不存在。 + +- [ ] **Step 3: 写能力账本和三份边界文档** + +十二类最低能力逐项填写状态。embedding、多智能体和生产 adapter 标为 `experimental` 或 `planned`;未通过当前测试的能力不得标为 `implemented`。 + +- [ ] **Step 4: 更新中英文 README** + +Quickstart 必须从 clone、安装到生成并验证完整事件包;命令仅引用当前实际存在的 CLI。 + +- [ ] **Step 5: 运行账本检查** + +Run: `python tools/check_capabilities.py && python -m pytest tests/contract/test_capabilities.py -q` + +Expected: PASS。 + +- [ ] **Step 6: 提交** + +```bash +git add CAPABILITIES.yaml schemas/capabilities-v1.schema.json docs README.md README.zh-CN.md tools/check_capabilities.py tests/contract/test_capabilities.py +git commit -m "docs: publish capability and book mappings" +``` + +### Task 14: 扩展黄金、安全和规模验证 + +**Files:** +- Modify: `tests/golden/` +- Create: `tests/scale/test_generated_warehouse.py` +- Create: `tools/benchmark_scale.py` +- Create: `.github/workflows/scale.yml` +- Modify: `.github/workflows/ci.yml` +- Create: `tools/ci_checks.sh` +- Create: `tools/check_layering.py` +- Create: `tools/check_random_ids.py` +- Create: `tools/check_release_assets.py` + +- [ ] **Step 1: 增加至少 20 组黄金 SQL** + +每组 fixture 配套规范化 snapshot,覆盖设计规格列出的语法、方言和失败诊断。测试断言 fixture 数量不少于 20。 + +- [ ] **Step 2: 增加分层规模基准** + +```python +@pytest.mark.parametrize("count", [100]) +def test_generated_sql_scale_smoke(count, tmp_path): + result = run_scale_benchmark(count, tmp_path) + assert result.parsed == count + assert result.deterministic + assert result.peak_memory_mb > 0 +``` + +手工 workflow 执行: + +```bash +python tools/benchmark_scale.py --counts 1000,10000 --output .benchmarks/result.json +``` + +- [ ] **Step 3: 接入 CI 门禁** + +普通 CI 运行 lint、类型、分层、随机 ID、单元、契约、黄金、端到端、安全、100 SQL 规模、能力账本、开源 guard 和构建包 Quickstart。 + +- [ ] **Step 4: 运行完整本地门禁** + +Run: `PYTHON=python bash tools/ci_checks.sh` + +Expected: 所有阶段通过,覆盖率达到脚本设定阈值。 + +- [ ] **Step 5: 提交** + +```bash +git add tests/golden tests/scale tools .github/workflows +git commit -m "test: enforce release governance gates" +``` + +### Task 15: 发布验收、敏感文件审计和 GitHub 推送 + +**Files:** +- Modify: `pyproject.toml` +- Modify: `LICENSE` +- Modify: `CONTRIBUTING.md` +- Create: `docs/release-evidence.md` + +- [ ] **Step 1: 验证打包和十分钟 Quickstart** + +Run: + +```bash +python -m build +python -m venv /tmp/sqlgraph-release-venv +/tmp/sqlgraph-release-venv/bin/pip install dist/*.whl +/tmp/sqlgraph-release-venv/bin/sqlgraph governance run examples/minimal/scenario.yaml -o /tmp/sqlgraph-quickstart +/tmp/sqlgraph-release-venv/bin/sqlgraph governance verify /tmp/sqlgraph-quickstart +``` + +Expected: 安装成功、案例成功、审计完整性通过。 + +- [ ] **Step 2: 运行全部测试与规模基准** + +Run: + +```bash +PYTHON=python bash tools/ci_checks.sh +python tools/benchmark_scale.py --counts 1000,10000 --output .benchmarks/release.json +``` + +Expected: 全部通过,并输出资源指标。 + +- [ ] **Step 3: 浏览器验证** + +启动本地服务,验证 Explorer 和治理工作台在 1440x1000、390x844 下无空白、遮挡或横向溢出,且控制台无错误。 + +- [ ] **Step 4: 写发布证据** + +`docs/release-evidence.md` 记录六条门槛对应命令、测试、产物和结果摘要,不提交 `.benchmarks/`、HTML、DuckDB 或截图。 + +- [ ] **Step 5: 审计提交范围** + +Run: + +```bash +git status --short +git diff origin/main...HEAD --check +python scripts/opensource_guard.py +git ls-files | rg '(^book|\\.zip$|\\.duckdb$|comparison-screenshots|generated_illustrations)' +``` + +Expected: 最后一条无输出;其余检查通过。 + +- [ ] **Step 6: 最终提交** + +```bash +git add pyproject.toml LICENSE CONTRIBUTING.md docs/release-evidence.md +git commit -m "chore: prepare governance reference release" +``` + +- [ ] **Step 7: 推送功能分支** + +```bash +git push -u origin feat/book-governance-reference +``` + +Expected: 推送成功,远端分支可用于 Pull Request;不直接合并或强推 `main`。 diff --git a/docs/release-evidence.md b/docs/release-evidence.md new file mode 100644 index 0000000..f581c6e --- /dev/null +++ b/docs/release-evidence.md @@ -0,0 +1,44 @@ +# Governance Reference Release Evidence + +## Release Gates + +| Gate | Evidence | Result | +|---|---|---| +| Ten-minute runnable | Wheel installed in a clean Python 3.12 virtual environment; `governance run`, `verify`, and `replay` completed | Pass | +| Honest capability status | `CAPABILITIES.yaml` validated against JSON Schema and repository paths | Pass | +| Core safety | Evidence, Grounding, autonomy, idempotency, rollback, verification, and audit tamper tests | Pass | +| Three book cases | Caliber consistency, cold-table quarantine, irreversible drop to L4 | Pass | +| Full replay | Every case emits seven hash-chained events and passes integrity verification | Pass | +| Explicit limitations | `docs/limitations.md` and capability-level limitations | Pass | + +## Automated Validation + +- Unified release gate: `PYTHON=python bash tools/ci_checks.sh` +- Tests: 344 passed at the first complete release-gate run. +- Foundation coverage: 89.29%, above the 85% gate. +- Golden corpus: 5 byte-for-byte graph snapshots plus 15 syntax/dialect cases. +- Package build: source distribution and wheel completed. +- Open-source guard, layering, deterministic-ID, capability, and release-asset + checks passed. + +## Scale Evidence + +| SQL statements | Elapsed | Peak memory | Nodes | Edges | Artifact | Deterministic | +|---:|---:|---:|---:|---:|---:|---| +| 1,000 | 997 ms | 11.45 MB | 5,003 | 9,002 | 2.52 MB | yes | +| 10,000 | 10.16 s | 113.17 MB | 50,003 | 90,002 | 25.25 MB | yes | + +The benchmark uses a deterministic generated SQL chain. Results are local +reference measurements, not production latency guarantees. + +## Browser Evidence + +- Desktop viewport: 1440x1000, no horizontal overflow. +- Mobile viewport: 390x844, no horizontal overflow. +- Mobile lineage canvas switches to the list representation. +- No task tabs are off-screen. +- Explorer brand, dark tokens, and active navigation are present. +- Browser console produced no errors. + +Screenshots and generated HTML remain local test artifacts and are intentionally +excluded from Git. diff --git a/docs/specs/2026-08-25-book-governance-reference-release-design.md b/docs/specs/2026-08-25-book-governance-reference-release-design.md new file mode 100644 index 0000000..047977e --- /dev/null +++ b/docs/specs/2026-08-25-book-governance-reference-release-design.md @@ -0,0 +1,456 @@ +# AI 原生数仓治理参考实现发布设计 + +## 1. 目标 + +将 SqlGraph 建设为《AI 原生数仓治理》的可运行参考实现与证据仓库,并满足以下六条对外发布门槛: + +1. 全新 Python 3.10-3.12 环境可在十分钟内完成安装、运行示例并生成证据产物。 +2. `CAPABILITIES.yaml` 如实记录每项能力的状态、模块、测试和示例。 +3. 证据缺失、不可逆动作、授权缺失和验证失败均有自动化拒绝测试。 +4. 至少提供口径一致性、成本或冷表治理、不可逆动作转 L4 三条完整案例。 +5. 每条案例均能导出从 baseline、证据、提案、授权、执行到验证的可校验事件包。 +6. 支持范围、已知限制、外部依赖和规模边界均有公开文档。 + +最终读者看到的不只是血缘图,而是能验证一个治理结论如何从确定性事实中产生、如何受证据和安全闸门约束,以及执行后如何验证、恢复和追责。 + +## 2. 范围 + +### 2.1 本轮包含 + +- 以 GitHub `origin/main` 为基线,保留现有 `analyze`、`serve`、Explorer、CLI 和开源治理文件。 +- 移植并深化本地已有的确定性身份、多语句血缘、表达式 DAG、证据子图、自治决策、七步循环、三层验证和 DuckDB 数仓案例。 +- 新增统一 baseline、证据纪律、Grounding、动作安全、独立验证、追加式审计、能力账本和书—代码—测试映射。 +- 将治理工作台接入 GitHub Explorer 的共享视觉基线,同时继续输出单文件离线 HTML。 +- 建立单元、契约、黄金、端到端、安全和规模六级验证。 +- 在功能分支通过全部门禁后推送 GitHub,不直接修改或强推远端 `main`。 + +### 2.2 本轮不包含 + +- 企业生产权限中心、审批平台或调度平台的真实接入。 +- Neo4j、ByteGraph、云数据库或大模型供应商的强依赖。 +- 生产级多租户、在线高可用、分布式锁和长期任务调度。 +- 将所有代码迁移到新的 `src/` 目录。 +- 把实验性 embedding 或多智能体能力声明为核心能力。 + +这些能力可以通过后续 adapter 或实验模块接入,但不影响本轮参考实现验收。 + +## 3. 基线与整合策略 + +### 3.1 Git 基线 + +- 功能分支:`feat/book-governance-reference` +- 基线:最新 `origin/main` +- 工作树:独立 clean worktree +- 禁止从当前本地脏 `main` 直接合并或推送 +- 禁止提交书稿 ZIP、生成图片、DuckDB 文件、浏览器截图、缓存和本地输出 + +### 3.2 远端能力必须保留 + +- `sqlgraph/analyze/` 离线治理分析及 CLI `analyze`、`profile` +- `sqlgraph/serve/` Explorer 搜索、图查看、统计和 Playground +- `sqlgraph/input/ddl_schema.py` 与 DDL 转 Schema 工具 +- GitHub issue/PR 模板、`SECURITY.md`、`CODE_OF_CONDUCT.md` +- MIT License、开源检查脚本和发布包资源声明 +- 远端保守表达式指纹语义,不恢复代数交换律推断 + +### 3.3 本地能力按模块移植 + +不复制整个目录覆盖远端,而是以测试为入口逐模块移植。发生语义冲突时,以以下优先级裁决: + +1. 本设计中的安全和证据约束。 +2. GitHub 当前公开接口和兼容性。 +3. 本地实现细节。 + +## 4. 目标架构 + +```text +Input Adapters + | + v +Baseline Registry + | + v +Deterministic Compiler ---> HeteroGraph ---> TableGraph / Analyze + | + v + Evidence Engine + | coverage + | counterevidence + | citations + | + v + Autonomy Policy + | + v + Action Engine + | dry run + | idempotency + | execute + | rollback + | + v + Independent Verification + | + v + Append-only Audit Log +``` + +七步治理循环只负责编排这些深模块,不复制 baseline、证据、授权、执行或验证规则。 + +## 5. 模块设计 + +### 5.1 Baseline Registry + +位置:`sqlgraph/baseline/` + +主要接口: + +```python +def build_baseline( + source: SqlSource, + *, + dialect: str | None, + schema: SchemaRegistry | None, + parameters: Mapping[str, object] | None = None, + udf_manifest: Mapping[str, object] | None = None, + scheduler_manifest: Mapping[str, object] | None = None, +) -> BaselineManifest +``` + +`BaselineManifest` 必须包含: + +- `schema_version` +- `baseline_id` +- 输入文件及内容 SHA256 +- 方言、SQLGlot 版本、身份规则版本 +- Schema、UDF、参数和调度快照的哈希 +- 创建时间,仅作为元数据,不参与语义身份 +- 缺失依赖及其显式状态 + +同一有效输入必须生成相同 `baseline_id`。缺失 Schema 或 UDF 时不得伪造已解析结论。 + +### 5.2 Deterministic Compiler + +位置:现有 `sqlgraph/input/`、`parser/`、`builder/`、`model/`、`identity/` + +要求: + +- 全部 SQL 语句独立建模,禁止跨语句笛卡尔伪边。 +- 表、字段、表达式、语句和边均使用稳定语义身份。 +- Transform 指纹沿用远端保守规范化,不推断 `a+b` 与 `b+a` 等价。 +- 物理列无法判定时标记 `UNRESOLVED`,不得使用默认表名猜测。 +- 节点、边和诊断包含可定位的来源文件、语句序号和源码范围。 +- TableGraph 是 HeteroGraph 的可重建投影,每条边保存来源语句、字段依赖和 transform 引用。 + +### 5.3 Evidence Engine + +位置:`sqlgraph/evidence/` + +外部接口收敛为: + +```python +class EvidenceEngine: + def collect(self, request: EvidenceRequest) -> EvidenceBundle: ... + def expand(self, bundle: EvidenceBundle, reason: str) -> EvidenceBundle: ... + def assess(self, bundle: EvidenceBundle) -> SufficiencyDecision: ... +``` + +`EvidenceRequest` 必须声明: + +- `task_id`、`baseline_id` +- 治理对象和治理意图 +- 锚点 +- 方向和最大深度 +- 覆盖义务 +- 外部事实要求 + +`EvidenceBundle` 必须输出: + +- included / excluded 节点和边 +- 支持证据和反对证据 +- 覆盖契约、缺口、残余未知 +- 来源引用 +- 规则版本和稳定 `subgraph_hash` +- 扩图历史 + +证据不足只能返回 `expand`、`degrade`、`refuse` 或 `escalate`,不得进入自主执行。 + +### 5.4 GraphRAG Grounding + +位置:`sqlgraph/graphrag/` + +主要接口: + +```python +def validate_assertions( + assertions: Sequence[GroundedAssertion], + evidence: EvidenceBundle, +) -> GroundingReport +``` + +每条结构性断言必须引用当前 baseline 和 evidence 中存在的节点、边或路径。以下情况必须失败: + +- 引用不存在。 +- 引用属于其他 baseline 或 evidence 版本。 +- 断言射程超过覆盖契约。 +- 只有自然语言结论,没有结构证据。 + +GraphRAG 序列化只负责上下文,不自动赋予结论真实性。 + +### 5.5 Autonomy Policy + +位置:`sqlgraph/autonomy/` + +保留当前“闸门在前、标定在后”模型,并补充: + +- 输入必须引用 `evidence_version` 和授权主体。 +- 可逆性由三项证据组成:状态可恢复、外部后果可控、回滚路径已验证。 +- 证据、授权或可逆性任一闸门失败时,收益分不得补偿。 +- 输出继续使用动作深度、L4 开关和 L5 范围三维模型。 +- 所有规则携带 `policy_version`。 + +### 5.6 Action Engine + +位置:`sqlgraph/actions/` + +核心接口: + +```python +class ActionEngine: + def plan(self, request: ActionRequest) -> ActionPlan: ... + def dry_run(self, plan: ActionPlan) -> DryRunResult: ... + def execute(self, plan: ActionPlan) -> ExecutionResult: ... + def rollback(self, execution: ExecutionResult) -> RollbackResult: ... +``` + +首批两个 adapter: + +- `SqlFilePatchAdapter`:精确修改、前置内容哈希、备份和恢复验证。 +- `DuckDBTaskAdapter`:事务或快照执行、增量任务重跑、故障注入和恢复。 + +执行前必须验证: + +- 自治决议允许该动作。 +- 幂等键未产生成功副作用。 +- 预演结果与计划一致。 +- 回滚计划已实际验证,而不只是存在文本。 + +执行失败或验证失败时触发熔断;自动回滚仅适用于已验证可恢复的 adapter。 + +### 5.7 Independent Verification + +位置:`sqlgraph/verification/` + +三层定义: + +1. `code`:计划变更与实际变更是否一致。 +2. `structure`:从变更后的实际 SQL 和 baseline 独立重建图,再比较边界。 +3. `runtime`:由 DuckDB adapter 查询真实结果、质量信号和下游重算状态。 + +验证模块不得复用执行阶段缓存的“成功”布尔值。任一关键层 `fail` 时结果只能是 `failed`;运行态未执行时只能是 `incomplete`。 + +为兼容现有调用,`sqlgraph.verify` 暂时作为迁移适配器,内部委托给 `sqlgraph.verification`。 + +### 5.8 Append-only Audit + +位置:`sqlgraph/audit/` + +主要接口: + +```python +class AuditLog: + def append(self, event: AuditEvent) -> AuditEvent: ... + def verify_integrity(self) -> IntegrityReport: ... + def replay(self, task_id: str) -> ReplayResult: ... +``` + +每个事件包含: + +- `event_id`、`task_id`、`baseline_id` +- 事件类型、schema 版本和策略版本 +- 授权身份、证据引用、幂等键 +- 前一事件哈希和当前事件哈希 +- 输入、输出和状态转换 + +事件采用 JSONL 追加写入。篡改、删除或重排事件必须被完整性检查发现。 + +### 5.9 Governance Runner + +位置:`sqlgraph/reasoning/` + +`GovernanceRunner.run()` 编排 Observe、Explain、Propose、Authorize、Execute、Verify、Learn。每步产生版本化产物和审计事件;失败后可从允许的步骤恢复,但不能绕过 Authorize 或 Verify。 + +现有 `sqlgraph.agent` 保留为兼容入口,委托给新 runner。 + +## 6. 数据契约 + +在 `schemas/` 提供以下 JSON Schema: + +- `baseline-manifest-v1.schema.json` +- `evidence-bundle-v1.schema.json` +- `autonomy-decision-v1.schema.json` +- `action-plan-v1.schema.json` +- `verification-report-v1.schema.json` +- `audit-event-v1.schema.json` +- `capabilities-v1.schema.json` + +所有持久化产物包含 `schema_version`。契约测试验证必填字段、拒绝语义和兼容读取。 + +## 7. 示例与用户体验 + +### 7.1 十分钟示例 + +`examples/minimal/` 提供一条小型 SQL 流程和一条命令: + +```bash +sqlgraph governance run examples/minimal/scenario.yaml -o demo_output/minimal +``` + +输出: + +- baseline manifest +- graph snapshot +- evidence bundle +- autonomy decision +- action/verification records +- audit JSONL 和 replay summary +- 单文件 HTML 报告 + +### 7.2 三条书中案例 + +- `caliber_consistency`:CTR 百分数与比率冲突,允许 L3 可逆修复。 +- `cold_table_retirement`:先停写或隔离,保留恢复路径,验证成本与下游影响。 +- `irreversible_drop`:无已验证恢复路径的删除动作,稳定拒绝 L3/L5 并转 L4。 + +现有 94 表、62 任务视频商业化数仓作为综合案例保留,不作为十分钟 Quickstart 的前置条件。 + +### 7.3 GitHub 样式 + +- `sqlgraph/serve/web/static/app.css` 是共享视觉基线。 +- 治理报告渲染器读取并内联共享 CSS,再追加治理页面专用规则。 +- 页面保持 `#0b1020`、`#0f172a`、`#34d399`、`#38bdf8` 等远端 token。 +- 保留顶部 `app-tabs` 导航、深色网格图谱、键盘焦点和移动端列表模式。 +- 报告不得引用外部 CDN,必须可离线打开并适配 390x844。 + +## 8. 能力账本与文档 + +根目录新增 `CAPABILITIES.yaml`,每项包含: + +- `id` +- `status` +- `description` +- `modules` +- `tests` +- `examples` +- `limitations` + +状态只能是: + +- `implemented` +- `experimental` +- `planned` +- `concept-only` + +新增: + +- `docs/book-map.md` +- `docs/theory-vs-implementation.md` +- `docs/limitations.md` +- 更新后的中英文 README 和 architecture + +CI 校验账本中的模块、测试和示例路径真实存在;书中尚未实现的能力不得标记为 implemented。 + +## 9. 验证体系 + +### 9.1 单元测试 + +覆盖规范化、稳定身份、baseline、证据扩图、反证、Grounding、自治规则、幂等、哈希链和回滚状态机。 + +### 9.2 契约测试 + +位置:`tests/contract/` + +验证 JSON Schema、CLI 输出、版本兼容、错误码和拒绝语义。 + +### 9.3 黄金语料 + +位置:`tests/golden/` + +从当前 5 组扩展到至少 20 组,覆盖: + +- 多语句 +- CTE 与嵌套子查询 +- JOIN 同名列与缺 Schema +- 窗口、CASE、CAST、聚合 +- UDF 与缺 UDF +- 动态 SQL 的明确不支持诊断 +- Spark、Hive、Presto、BigQuery、MySQL、Postgres、DuckDB +- 仅格式变化的稳定性 + +### 9.4 端到端测试 + +三条案例均从 baseline 运行到 replay,并验证完整事件包。 + +### 9.5 安全测试 + +位置:`tests/safety/` + +覆盖: + +- 删除关键证据后降级或拒绝。 +- 伪造引用被 Grounding 拒绝。 +- 不可逆动作强制 L4。 +- 无授权动作止于提案。 +- 重复执行不产生重复副作用。 +- 执行中故障触发熔断和恢复。 +- 独立验证失败触发回滚或转人工。 +- 审计事件被篡改后完整性失败。 + +### 9.6 规模测试 + +位置:`tests/scale/` + +生成 `10^2`、`10^3`、`10^4` SQL 的确定性基准,记录: + +- 总耗时和 p50/p95 +- 峰值内存 +- 解析成功率 +- UNKNOWN / UNRESOLVED 数 +- 节点、边和产物体积 + +默认 CI 运行 `10^2`,定时或手工 workflow 运行 `10^3` 和 `10^4`,避免普通 PR 被长基准阻塞。 + +## 10. CI 与发布门禁 + +CI 阶段: + +1. Ruff 和基础类型检查。 +2. 分层依赖、随机 ID 和敏感产物扫描。 +3. 单元、契约、黄金、端到端和安全测试。 +4. 能力账本与书—代码映射一致性。 +5. 构建 wheel/sdist 并在全新虚拟环境运行十分钟 Quickstart。 +6. Explorer 与治理报告样式契约测试。 +7. 开源 guard、License 和包内容检查。 + +发布前再运行: + +- 完整 `10^3` / `10^4` 规模基准。 +- 桌面和 390x844 浏览器验证。 +- 三条案例的审计重放和恢复演练。 +- `git status` 白名单检查,确保没有本地生成物。 + +## 11. 完成标准 + +只有同时满足以下条件,功能分支才可推送: + +- 六条发布门槛全部有当前版本自动化证据。 +- 远端 `analyze`、`profile`、`serve`、`build`、`playground`、`demo` 回归通过。 +- 新增 `governance` 命令可运行三条案例。 +- 至少 20 组黄金 SQL 通过。 +- 安全测试证明四类硬拒绝和恢复路径成立。 +- 审计包能通过 hash-chain 完整性校验并按 task 重放。 +- README、能力账本、book map 和 limitations 与代码一致。 +- Explorer 与治理报告桌面/移动端无布局回归。 +- 分支基于最新 `origin/main`,提交不包含生成数据库、报告、截图或书稿。 + +推送目标为 `origin/feat/book-governance-reference`。不自动合并 `main`,由 GitHub Pull Request 和 CI 完成最终审查。 diff --git a/docs/theory-vs-implementation.md b/docs/theory-vs-implementation.md new file mode 100644 index 0000000..14d17d5 --- /dev/null +++ b/docs/theory-vs-implementation.md @@ -0,0 +1,18 @@ +# Theory vs. Reference Implementation + +SqlGraph treats the book's reliability rules as invariants while keeping +technology choices replaceable. + +| Invariant | Reference implementation | Replaceable part | +|---|---|---| +| Same effective input produces the same semantic identity | SHA256 baseline and graph identity | hash library and storage | +| Conclusions must cite inspectable evidence | `EvidenceBundle` and Grounding validation | graph database and retrieval engine | +| Evidence gaps stop or reduce action | coverage obligations and sufficiency decisions | expansion strategy | +| Irreversible actions do not run autonomously | reversibility veto and L4 decision | approval system | +| Execution must be idempotent and recoverable | file and DuckDB adapters | production scheduler adapter | +| Verification is independent of execution success | source rebuild and runtime observations | runtime query adapter | +| Governance history is replayable | JSONL hash chain | append-only database or object store | + +The repository does not claim that an LLM summary is reasoning evidence, that a +dry run proves reversibility, or that a graph path alone is a sufficient +evidence subgraph. diff --git a/examples/book_cases/caliber_consistency/query.sql b/examples/book_cases/caliber_consistency/query.sql new file mode 100644 index 0000000..ff0bcb1 --- /dev/null +++ b/examples/book_cases/caliber_consistency/query.sql @@ -0,0 +1,3 @@ +INSERT INTO ads_creative_report +SELECT clicks * 100.0 / impressions AS ctr +FROM dwd_ad_events; diff --git a/examples/book_cases/caliber_consistency/scenario.yaml b/examples/book_cases/caliber_consistency/scenario.yaml new file mode 100644 index 0000000..f1467c2 --- /dev/null +++ b/examples/book_cases/caliber_consistency/scenario.yaml @@ -0,0 +1,32 @@ +task_id: book-caliber-001 +intent: caliber_repair +dialect: duckdb +sql_file: query.sql +source_table: dwd_ad_events +target_table: ads_creative_report +parameters: {} +udf_manifest: {} +scheduler_manifest: {} +action: + type: sql_patch + adapter: sql_file_patch + authorization_scope: single_l3 + authorization_identity: book-demo-policy + blast_radius: 0.2 + object_risk: 0.3 + historical_reliability: 0.95 + roi: 0.9 + reversibility: + state_restorable: true + external_effects_controlled: true + rollback_verified: true + references: + - file-content-backup + after: | + INSERT INTO ads_creative_report + SELECT clicks / impressions AS ctr + FROM dwd_ad_events; +runtime: + checks_passed: true + reports_recomputed: true + new_alerts: 0 diff --git a/examples/book_cases/cold_table_retirement/query.sql b/examples/book_cases/cold_table_retirement/query.sql new file mode 100644 index 0000000..263c94e --- /dev/null +++ b/examples/book_cases/cold_table_retirement/query.sql @@ -0,0 +1,3 @@ +INSERT INTO cold_table_archive +SELECT event_id, event_date +FROM cold_event_log; diff --git a/examples/book_cases/cold_table_retirement/scenario.yaml b/examples/book_cases/cold_table_retirement/scenario.yaml new file mode 100644 index 0000000..e04492f --- /dev/null +++ b/examples/book_cases/cold_table_retirement/scenario.yaml @@ -0,0 +1,33 @@ +task_id: book-cold-table-001 +intent: cost_governance +dialect: duckdb +sql_file: query.sql +source_table: cold_event_log +target_table: cold_table_archive +parameters: {} +udf_manifest: {} +scheduler_manifest: {} +action: + type: quarantine_cold_table + adapter: sql_file_patch + authorization_scope: single_l3 + authorization_identity: book-demo-policy + blast_radius: 0.1 + object_risk: 0.2 + historical_reliability: 0.9 + roi: 0.7 + reversibility: + state_restorable: true + external_effects_controlled: true + rollback_verified: true + references: + - file-content-backup + after: | + INSERT INTO cold_table_archive + SELECT event_id, event_date + FROM cold_event_log + WHERE event_date >= DATE '2026-01-01'; +runtime: + checks_passed: true + reports_recomputed: true + new_alerts: 0 diff --git a/examples/book_cases/irreversible_drop/query.sql b/examples/book_cases/irreversible_drop/query.sql new file mode 100644 index 0000000..7ec630e --- /dev/null +++ b/examples/book_cases/irreversible_drop/query.sql @@ -0,0 +1,3 @@ +INSERT INTO critical_orders_archive +SELECT order_id, amount +FROM critical_orders; diff --git a/examples/book_cases/irreversible_drop/scenario.yaml b/examples/book_cases/irreversible_drop/scenario.yaml new file mode 100644 index 0000000..119e73c --- /dev/null +++ b/examples/book_cases/irreversible_drop/scenario.yaml @@ -0,0 +1,29 @@ +task_id: book-drop-001 +intent: destructive_governance +dialect: duckdb +sql_file: query.sql +source_table: critical_orders +target_table: critical_orders_archive +parameters: {} +udf_manifest: {} +scheduler_manifest: {} +action: + type: drop_table + adapter: sql_file_patch + authorization_scope: continuous_l5 + authorization_identity: book-demo-policy + blast_radius: 0.1 + object_risk: 0.1 + historical_reliability: 1.0 + roi: 1.0 + reversibility: + state_restorable: false + external_effects_controlled: false + rollback_verified: false + references: [] + after: | + DROP TABLE critical_orders; +runtime: + checks_passed: false + reports_recomputed: false + new_alerts: 0 diff --git a/examples/minimal/scenario.yaml b/examples/minimal/scenario.yaml new file mode 100644 index 0000000..8a2a42c --- /dev/null +++ b/examples/minimal/scenario.yaml @@ -0,0 +1,32 @@ +task_id: minimal-ctr-001 +intent: caliber_repair +dialect: duckdb +sql_file: sql/query.sql +source_table: src_events +target_table: dst_metric +parameters: {} +udf_manifest: {} +scheduler_manifest: {} +action: + type: sql_patch + adapter: sql_file_patch + authorization_scope: single_l3 + authorization_identity: quickstart-policy + blast_radius: 0.1 + object_risk: 0.1 + historical_reliability: 0.95 + roi: 0.8 + reversibility: + state_restorable: true + external_effects_controlled: true + rollback_verified: true + references: + - file-content-backup + after: | + INSERT INTO dst_metric + SELECT clicks / impressions AS ctr + FROM src_events; +runtime: + checks_passed: true + reports_recomputed: true + new_alerts: 0 diff --git a/examples/minimal/sql/query.sql b/examples/minimal/sql/query.sql new file mode 100644 index 0000000..4cb3a9c --- /dev/null +++ b/examples/minimal/sql/query.sql @@ -0,0 +1,3 @@ +INSERT INTO dst_metric +SELECT clicks * 100.0 / impressions AS ctr +FROM src_events; diff --git a/examples/video_commercial_warehouse/README.md b/examples/video_commercial_warehouse/README.md new file mode 100644 index 0000000..f98bcab --- /dev/null +++ b/examples/video_commercial_warehouse/README.md @@ -0,0 +1,81 @@ +# Video Commercial Warehouse + +一套可由 DuckDB 真实执行、可由 SqlGraph 构图和治理的视频平台商业化完整数仓。 + +## 规模 + +| 层级 | 表数 | 任务数 | +|---|---:|---:| +| ODS | 18 | 0 | +| STG | 18 | 18 | +| DIM | 14 | 0 | +| DWD | 18 | 18 | +| DWS | 16 | 16 | +| ADS | 10 | 10 | +| **合计** | **94** | **62** | + +覆盖内容、用户、推荐流量、广告资源、竞价投放、转化归因、预算计费、商业收入、 +实验策略和质量风控十个主题。 + +## 运行 + +从仓库根目录执行: + +```bash +# 重新从 catalog 生成 62 个 SQL 文件和 manifest +.devvenv/bin/python -m examples.video_commercial_warehouse.generate_sql + +# CI / 快速验证:5,000 推荐请求 +.devvenv/bin/python -m examples.video_commercial_warehouse.governance \ + --profile smoke \ + --output demo_output/video_commercial_warehouse/smoke \ + --no-open + +# 完整演示:200,000 推荐请求 +.devvenv/bin/python -m examples.video_commercial_warehouse.generate_sql +.devvenv/bin/python -m examples.video_commercial_warehouse.governance \ + --profile demo \ + --output demo_output/video_commercial_warehouse/demo +``` + +`generate_sql` 会把数仓恢复到包含 CTR 口径冲突的基线;`governance` 会: + +1. 用 DuckDB 生成 32 张基础表; +2. 真实执行 62 个 SQL 任务; +3. 用 SqlGraph 构建全仓血缘; +4. 识别 `dws_creative_performance_daily.ctr` 百分数/比率冲突; +5. 备份并修改真实 SQL; +6. 只重跑目标及其受影响下游; +7. 用 DuckDB 查询结果执行三层验证; +8. 生成结构化审计和离线交互报告。 + +## 恢复 + +```bash +.devvenv/bin/python -m examples.video_commercial_warehouse.governance \ + --output demo_output/video_commercial_warehouse/demo \ + --restore +``` + +恢复操作会同时恢复 SQL 和 DuckDB 快照,并将报告状态更新为 `restored`,避免旧报告 +继续显示为当前成功状态。 + +## 产物 + +```text +demo_output/video_commercial_warehouse/demo/ +├── before/ +│ └── sql/ # 62 个治理前 SQL 快照 +├── before.duckdb # 治理前完整数仓 +├── after.duckdb # 治理后增量重跑结果 +├── operation.json # 结构化治理审计 +└── warehouse_report.html # 离线交互报告 +``` + +## 设计纪律 + +- Catalog 是表、任务、依赖的唯一真相源。 +- SQL 使用 DuckDB 可直接执行且 SQLGlot 可解析的公共语法。 +- 所有种子数据由 `range()` 与确定性公式生成,不使用随机 ID。 +- 比率保持 `[0,1]`,展示层才转换成百分比。 +- 运行态验证必须来自 DuckDB 查询;没有查询结果不得判定闭环成功。 diff --git a/examples/video_commercial_warehouse/__init__.py b/examples/video_commercial_warehouse/__init__.py new file mode 100644 index 0000000..7e4fd04 --- /dev/null +++ b/examples/video_commercial_warehouse/__init__.py @@ -0,0 +1,19 @@ +"""可执行的视频平台商业化完整数仓示例。""" + +from examples.video_commercial_warehouse.catalog import ( + ALL_TABLES, + BASE_TABLES, + TASKS, + TaskSpec, + TableSpec, + validate_catalog, +) + +__all__ = [ + "ALL_TABLES", + "BASE_TABLES", + "TASKS", + "TaskSpec", + "TableSpec", + "validate_catalog", +] diff --git a/examples/video_commercial_warehouse/catalog.py b/examples/video_commercial_warehouse/catalog.py new file mode 100644 index 0000000..863af4a --- /dev/null +++ b/examples/video_commercial_warehouse/catalog.py @@ -0,0 +1,726 @@ +"""视频平台商业化数仓的权威表与任务目录。 + +Catalog 是 94 张表和 62 个任务的唯一真相源。SQL 文件、运行顺序、依赖图和 +可视化元数据均从这里派生,避免文件名、任务声明和实际 SQL 三套定义漂移。 +""" +from __future__ import annotations + +from collections import Counter +from dataclasses import asdict, dataclass + + +DOMAINS = { + "content", + "user", + "traffic", + "recommendation", + "ad_inventory", + "ad_delivery", + "attribution", + "billing", + "experiment", + "risk", +} + + +@dataclass(frozen=True) +class TableSpec: + name: str + layer: str + domain: str + columns: tuple[tuple[str, str], ...] + + def to_dict(self) -> dict: + data = asdict(self) + data["columns"] = [list(column) for column in self.columns] + return data + + +@dataclass(frozen=True) +class TaskSpec: + order: int + target: str + layer: str + domain: str + dependencies: tuple[str, ...] + select_sql: str + description: str + + @property + def sql(self) -> str: + return ( + f"-- {self.layer} / {self.domain}: {self.description}\n" + f"CREATE OR REPLACE TABLE {self.target} AS\n" + f"{self.select_sql.strip()};\n" + ) + + def to_dict(self) -> dict: + return { + "order": self.order, + "target": self.target, + "layer": self.layer, + "domain": self.domain, + "dependencies": list(self.dependencies), + "description": self.description, + "file": f"{self.order:03d}_{self.target}.sql", + } + + +def _columns(spec: str) -> tuple[tuple[str, str], ...]: + """把 ``name:TYPE,name:TYPE`` 转为不可变 schema 声明。""" + columns = [] + for item in spec.split(","): + name, separator, data_type = item.partition(":") + if not separator: + raise ValueError(f"非法字段声明: {item}") + columns.append((name, data_type)) + return tuple(columns) + + +def _base(name: str, layer: str, domain: str, columns: str) -> TableSpec: + return TableSpec(name, layer, domain, _columns(columns)) + + +_ODS = [ + _base("ods_recommend_request", "ODS", "recommendation", + "request_id:BIGINT,session_id:BIGINT,user_id:BIGINT,device_id:BIGINT," + "geo_id:BIGINT,channel_id:BIGINT,experiment_id:BIGINT,request_time:TIMESTAMP,event_date:DATE"), + _base("ods_video_exposure", "ODS", "traffic", + "exposure_id:BIGINT,request_id:BIGINT,user_id:BIGINT,video_id:BIGINT," + "position:INTEGER,exposure_time:TIMESTAMP,event_date:DATE"), + _base("ods_video_play", "ODS", "traffic", + "play_id:BIGINT,exposure_id:BIGINT,user_id:BIGINT,video_id:BIGINT," + "play_duration_sec:DOUBLE,completion_rate:DOUBLE,play_time:TIMESTAMP,event_date:DATE"), + _base("ods_user_interaction", "ODS", "content", + "interaction_id:BIGINT,play_id:BIGINT,user_id:BIGINT,video_id:BIGINT," + "interaction_type:VARCHAR,event_time:TIMESTAMP,event_date:DATE"), + _base("ods_search_event", "ODS", "traffic", + "search_id:BIGINT,session_id:BIGINT,user_id:BIGINT,query:VARCHAR," + "result_count:INTEGER,event_time:TIMESTAMP,event_date:DATE"), + _base("ods_user_session", "ODS", "user", + "session_id:BIGINT,user_id:BIGINT,device_id:BIGINT,channel_id:BIGINT," + "start_time:TIMESTAMP,end_time:TIMESTAMP,event_date:DATE"), + _base("ods_ad_request", "ODS", "ad_inventory", + "ad_request_id:BIGINT,request_id:BIGINT,user_id:BIGINT,ad_slot_id:BIGINT," + "audience_id:BIGINT,request_time:TIMESTAMP,event_date:DATE"), + _base("ods_ad_candidate", "ODS", "ad_inventory", + "candidate_id:BIGINT,ad_request_id:BIGINT,ad_creative_id:BIGINT," + "ad_group_id:BIGINT,campaign_id:BIGINT,advertiser_id:BIGINT," + "predicted_ctr:DOUBLE,event_date:DATE"), + _base("ods_ad_bid", "ODS", "ad_delivery", + "bid_id:BIGINT,candidate_id:BIGINT,bid_price:DOUBLE,is_winner:BOOLEAN,event_date:DATE"), + _base("ods_ad_impression", "ODS", "ad_delivery", + "ad_impression_id:BIGINT,ad_request_id:BIGINT,user_id:BIGINT," + "ad_creative_id:BIGINT,campaign_id:BIGINT,advertiser_id:BIGINT," + "ad_slot_id:BIGINT,impression_time:TIMESTAMP,cost:DOUBLE,event_date:DATE"), + _base("ods_ad_click", "ODS", "ad_delivery", + "ad_click_id:BIGINT,ad_impression_id:BIGINT,user_id:BIGINT," + "ad_creative_id:BIGINT,campaign_id:BIGINT,click_time:TIMESTAMP,event_date:DATE"), + _base("ods_ad_conversion", "ODS", "attribution", + "conversion_id:BIGINT,ad_click_id:BIGINT,user_id:BIGINT," + "ad_creative_id:BIGINT,campaign_id:BIGINT,conversion_type:VARCHAR," + "conversion_value:DOUBLE,conversion_time:TIMESTAMP,event_date:DATE"), + _base("ods_ad_charge", "ODS", "billing", + "charge_id:BIGINT,ad_impression_id:BIGINT,advertiser_id:BIGINT," + "campaign_id:BIGINT,amount:DOUBLE,charge_time:TIMESTAMP,event_date:DATE"), + _base("ods_ad_refund", "ODS", "billing", + "refund_id:BIGINT,charge_id:BIGINT,advertiser_id:BIGINT," + "amount:DOUBLE,reason:VARCHAR,refund_time:TIMESTAMP,event_date:DATE"), + _base("ods_budget_snapshot", "ODS", "billing", + "snapshot_id:BIGINT,campaign_id:BIGINT,budget:DOUBLE,spent:DOUBLE,snapshot_date:DATE"), + _base("ods_content_audit", "ODS", "risk", + "audit_id:BIGINT,video_id:BIGINT,audit_status:VARCHAR," + "risk_score:DOUBLE,audit_time:TIMESTAMP,event_date:DATE"), + _base("ods_ad_audit", "ODS", "risk", + "audit_id:BIGINT,ad_creative_id:BIGINT,audit_status:VARCHAR," + "risk_score:DOUBLE,audit_time:TIMESTAMP,event_date:DATE"), + _base("ods_invalid_traffic_signal", "ODS", "risk", + "signal_id:BIGINT,request_id:BIGINT,user_id:BIGINT,signal_type:VARCHAR," + "risk_score:DOUBLE,event_time:TIMESTAMP,event_date:DATE"), +] + +_DIM = [ + _base("dim_user", "DIM", "user", + "user_id:BIGINT,register_date:DATE,country_code:VARCHAR,age_bucket:VARCHAR,audience_id:BIGINT"), + _base("dim_creator", "DIM", "content", + "creator_id:BIGINT,user_id:BIGINT,creator_tier:VARCHAR"), + _base("dim_video", "DIM", "content", + "video_id:BIGINT,creator_id:BIGINT,category_id:BIGINT," + "duration_sec:INTEGER,publish_date:DATE,content_status:VARCHAR"), + _base("dim_content_category", "DIM", "content", + "category_id:BIGINT,category_name:VARCHAR"), + _base("dim_device", "DIM", "user", + "device_id:BIGINT,device_type:VARCHAR,os:VARCHAR"), + _base("dim_geo", "DIM", "user", + "geo_id:BIGINT,country_code:VARCHAR,region:VARCHAR"), + _base("dim_channel", "DIM", "traffic", + "channel_id:BIGINT,channel_name:VARCHAR"), + _base("dim_advertiser", "DIM", "billing", + "advertiser_id:BIGINT,advertiser_name:VARCHAR,industry:VARCHAR"), + _base("dim_campaign", "DIM", "ad_delivery", + "campaign_id:BIGINT,advertiser_id:BIGINT,budget:DOUBLE,objective:VARCHAR,status:VARCHAR"), + _base("dim_ad_group", "DIM", "ad_delivery", + "ad_group_id:BIGINT,campaign_id:BIGINT,audience_id:BIGINT,bid_type:VARCHAR,bid_value:DOUBLE"), + _base("dim_ad_creative", "DIM", "ad_delivery", + "ad_creative_id:BIGINT,ad_group_id:BIGINT,video_id:BIGINT," + "creative_format:VARCHAR,status:VARCHAR"), + _base("dim_ad_slot", "DIM", "ad_inventory", + "ad_slot_id:BIGINT,slot_name:VARCHAR,scene:VARCHAR,floor_price:DOUBLE"), + _base("dim_experiment", "DIM", "experiment", + "experiment_id:BIGINT,experiment_name:VARCHAR,variant:VARCHAR"), + _base("dim_audience", "DIM", "experiment", + "audience_id:BIGINT,audience_name:VARCHAR,strategy:VARCHAR"), +] + +BASE_TABLES: dict[str, TableSpec] = { + table.name: table for table in [*_ODS, *_DIM] +} + + +def _task( + order: int, + target: str, + layer: str, + domain: str, + dependencies: tuple[str, ...], + select_sql: str, + description: str, +) -> TaskSpec: + return TaskSpec(order, target, layer, domain, dependencies, select_sql, description) + + +def _stg_tasks() -> list[TaskSpec]: + tasks = [] + for index, source in enumerate(_ODS, start=1): + target = source.name.replace("ods_", "stg_", 1) + tasks.append(_task( + index, + target, + "STG", + source.domain, + (source.name,), + f"SELECT DISTINCT * FROM {source.name}", + f"标准化 {source.name} 并去除完全重复记录", + )) + return tasks + + +_DWD_TASKS = [ + _task(19, "dwd_recommend_request_fact", "DWD", "recommendation", + ("stg_recommend_request", "dim_user", "dim_device", "dim_geo", "dim_channel", "dim_experiment"), + """ + SELECT r.*, u.age_bucket, d.device_type, d.os, g.country_code, g.region, + c.channel_name, e.variant AS experiment_variant + FROM stg_recommend_request r + LEFT JOIN dim_user u ON r.user_id = u.user_id + LEFT JOIN dim_device d ON r.device_id = d.device_id + LEFT JOIN dim_geo g ON r.geo_id = g.geo_id + LEFT JOIN dim_channel c ON r.channel_id = c.channel_id + LEFT JOIN dim_experiment e ON r.experiment_id = e.experiment_id + """, "推荐请求明细宽表"), + _task(20, "dwd_video_exposure_fact", "DWD", "traffic", + ("stg_video_exposure", "dim_video", "dim_creator", "dim_content_category"), + """ + SELECT x.*, v.creator_id, v.category_id, v.duration_sec, + cr.creator_tier, cc.category_name + FROM stg_video_exposure x + LEFT JOIN dim_video v ON x.video_id = v.video_id + LEFT JOIN dim_creator cr ON v.creator_id = cr.creator_id + LEFT JOIN dim_content_category cc ON v.category_id = cc.category_id + """, "视频曝光明细宽表"), + _task(21, "dwd_video_play_fact", "DWD", "traffic", + ("stg_video_play", "dim_video"), + """ + SELECT p.*, v.creator_id, v.category_id, v.duration_sec, + CASE WHEN p.completion_rate >= 0.9 THEN 1 ELSE 0 END AS is_complete_play, + CASE WHEN p.play_duration_sec >= 5 THEN 1 ELSE 0 END AS is_valid_play + FROM stg_video_play p + LEFT JOIN dim_video v ON p.video_id = v.video_id + """, "视频播放及有效播放事实"), + _task(22, "dwd_user_interaction_fact", "DWD", "content", + ("stg_user_interaction", "dim_video"), + """ + SELECT i.*, v.creator_id, v.category_id, + CASE WHEN i.interaction_type = 'like' THEN 1 ELSE 0 END AS is_like, + CASE WHEN i.interaction_type = 'comment' THEN 1 ELSE 0 END AS is_comment, + CASE WHEN i.interaction_type = 'share' THEN 1 ELSE 0 END AS is_share + FROM stg_user_interaction i + LEFT JOIN dim_video v ON i.video_id = v.video_id + """, "用户互动明细"), + _task(23, "dwd_search_fact", "DWD", "traffic", + ("stg_search_event",), + """ + SELECT *, LENGTH(query) AS query_length, + CASE WHEN result_count > 0 THEN 1 ELSE 0 END AS has_result + FROM stg_search_event + """, "搜索行为事实"), + _task(24, "dwd_session_fact", "DWD", "user", + ("stg_user_session", "dim_device", "dim_channel"), + """ + SELECT s.*, d.device_type, d.os, c.channel_name, + DATE_DIFF('second', s.start_time, s.end_time) AS session_duration_sec + FROM stg_user_session s + LEFT JOIN dim_device d ON s.device_id = d.device_id + LEFT JOIN dim_channel c ON s.channel_id = c.channel_id + """, "用户会话事实"), + _task(25, "dwd_ad_opportunity_fact", "DWD", "ad_inventory", + ("stg_ad_request", "dim_ad_slot", "dim_audience"), + """ + SELECT r.*, s.slot_name, s.scene, s.floor_price, a.audience_name + FROM stg_ad_request r + LEFT JOIN dim_ad_slot s ON r.ad_slot_id = s.ad_slot_id + LEFT JOIN dim_audience a ON r.audience_id = a.audience_id + """, "广告机会明细"), + _task(26, "dwd_ad_candidate_fact", "DWD", "ad_inventory", + ("stg_ad_candidate", "dim_ad_creative", "dim_ad_group"), + """ + SELECT c.*, cr.video_id, cr.creative_format, g.bid_type, g.bid_value + FROM stg_ad_candidate c + LEFT JOIN dim_ad_creative cr ON c.ad_creative_id = cr.ad_creative_id + LEFT JOIN dim_ad_group g ON c.ad_group_id = g.ad_group_id + """, "广告候选召回明细"), + _task(27, "dwd_ad_auction_fact", "DWD", "ad_delivery", + ("stg_ad_bid", "dwd_ad_candidate_fact"), + """ + SELECT b.*, c.ad_request_id, c.ad_creative_id, c.ad_group_id, + c.campaign_id, c.advertiser_id, c.predicted_ctr, + b.bid_price * c.predicted_ctr AS rank_score + FROM stg_ad_bid b + JOIN dwd_ad_candidate_fact c ON b.candidate_id = c.candidate_id + """, "竞价与胜出事实"), + _task(28, "dwd_ad_delivery_fact", "DWD", "ad_delivery", + ("stg_ad_impression", "dim_campaign", "dim_ad_creative", "dim_ad_slot"), + """ + SELECT i.*, c.objective, cr.ad_group_id, cr.video_id, cr.creative_format, + s.scene, s.floor_price + FROM stg_ad_impression i + LEFT JOIN dim_campaign c ON i.campaign_id = c.campaign_id + LEFT JOIN dim_ad_creative cr ON i.ad_creative_id = cr.ad_creative_id + LEFT JOIN dim_ad_slot s ON i.ad_slot_id = s.ad_slot_id + """, "广告曝光投放事实"), + _task(29, "dwd_ad_click_fact", "DWD", "ad_delivery", + ("stg_ad_click", "dwd_ad_delivery_fact"), + """ + SELECT c.*, i.ad_request_id, i.advertiser_id, i.ad_slot_id, + i.cost, i.impression_time, + DATE_DIFF('second', i.impression_time, c.click_time) AS click_delay_sec + FROM stg_ad_click c + JOIN dwd_ad_delivery_fact i ON c.ad_impression_id = i.ad_impression_id + """, "广告点击事实"), + _task(30, "dwd_ad_conversion_fact", "DWD", "attribution", + ("stg_ad_conversion", "dwd_ad_click_fact"), + """ + SELECT c.*, k.ad_impression_id, k.advertiser_id, k.ad_slot_id, + DATE_DIFF('second', k.click_time, c.conversion_time) AS conversion_delay_sec + FROM stg_ad_conversion c + JOIN dwd_ad_click_fact k ON c.ad_click_id = k.ad_click_id + """, "广告转化事实"), + _task(31, "dwd_attribution_touch_fact", "DWD", "attribution", + ("dwd_ad_click_fact", "dwd_ad_conversion_fact"), + """ + SELECT c.ad_click_id AS touch_id, c.ad_impression_id, c.user_id, + c.ad_creative_id, c.campaign_id, c.advertiser_id, + v.conversion_id, COALESCE(v.conversion_value, 0) AS conversion_value, + CASE WHEN v.conversion_id IS NULL THEN 0 ELSE 1 END AS is_converted, + c.event_date + FROM dwd_ad_click_fact c + LEFT JOIN dwd_ad_conversion_fact v ON c.ad_click_id = v.ad_click_id + """, "点击到转化的归因触点"), + _task(32, "dwd_ad_attribution_wide", "DWD", "attribution", + ("dwd_ad_delivery_fact", "dwd_ad_click_fact", "dwd_ad_conversion_fact"), + """ + SELECT i.ad_impression_id, i.ad_request_id, i.user_id, i.ad_creative_id, + i.ad_group_id, i.campaign_id, i.advertiser_id, i.ad_slot_id, + i.cost, i.event_date, + CASE WHEN c.ad_click_id IS NULL THEN 0 ELSE 1 END AS is_clicked, + CASE WHEN v.conversion_id IS NULL THEN 0 ELSE 1 END AS is_converted, + COALESCE(v.conversion_value, 0) AS conversion_value + FROM dwd_ad_delivery_fact i + LEFT JOIN dwd_ad_click_fact c ON i.ad_impression_id = c.ad_impression_id + LEFT JOIN dwd_ad_conversion_fact v ON c.ad_click_id = v.ad_click_id + """, "曝光点击转化归因宽表"), + _task(33, "dwd_billing_fact", "DWD", "billing", + ("stg_ad_charge", "dim_advertiser", "dim_campaign"), + """ + SELECT c.*, a.industry, p.objective + FROM stg_ad_charge c + LEFT JOIN dim_advertiser a ON c.advertiser_id = a.advertiser_id + LEFT JOIN dim_campaign p ON c.campaign_id = p.campaign_id + """, "广告扣费事实"), + _task(34, "dwd_refund_fact", "DWD", "billing", + ("stg_ad_refund", "dwd_billing_fact"), + """ + SELECT r.*, b.campaign_id, b.event_date AS charge_date + FROM stg_ad_refund r + LEFT JOIN dwd_billing_fact b ON r.charge_id = b.charge_id + """, "广告退款事实"), + _task(35, "dwd_budget_fact", "DWD", "billing", + ("stg_budget_snapshot", "dim_campaign"), + """ + SELECT b.*, c.advertiser_id, c.objective, c.status, + b.spent / NULLIF(b.budget, 0) AS pacing_ratio + FROM stg_budget_snapshot b + LEFT JOIN dim_campaign c ON b.campaign_id = c.campaign_id + """, "预算与消耗节奏事实"), + _task(36, "dwd_risk_event_fact", "DWD", "risk", + ("stg_content_audit", "stg_ad_audit", "stg_invalid_traffic_signal"), + """ + SELECT audit_id AS risk_event_id, 'video' AS entity_type, video_id AS entity_id, + audit_status AS risk_type, risk_score, event_date + FROM stg_content_audit + UNION ALL + SELECT audit_id, 'ad_creative', ad_creative_id, audit_status, risk_score, event_date + FROM stg_ad_audit + UNION ALL + SELECT signal_id, 'request', request_id, signal_type, risk_score, event_date + FROM stg_invalid_traffic_signal + """, "内容、广告与无效流量统一风险事实"), +] + +_DWS_TASKS = [ + _task(37, "dws_video_traffic_daily", "DWS", "traffic", + ("dwd_video_exposure_fact", "dwd_video_play_fact", "dwd_user_interaction_fact"), + """ + WITH e AS ( + SELECT video_id, event_date, COUNT(*) AS exposure_count + FROM dwd_video_exposure_fact GROUP BY video_id, event_date + ), p AS ( + SELECT video_id, event_date, COUNT(*) AS play_count, + SUM(is_valid_play) AS valid_play_count, + AVG(completion_rate) AS avg_completion_rate + FROM dwd_video_play_fact GROUP BY video_id, event_date + ), i AS ( + SELECT video_id, event_date, COUNT(*) AS interaction_count + FROM dwd_user_interaction_fact GROUP BY video_id, event_date + ) + SELECT e.video_id, e.event_date, e.exposure_count, + COALESCE(p.play_count, 0) AS play_count, + COALESCE(p.valid_play_count, 0) AS valid_play_count, + COALESCE(p.avg_completion_rate, 0) AS avg_completion_rate, + COALESCE(i.interaction_count, 0) AS interaction_count + FROM e LEFT JOIN p USING(video_id, event_date) + LEFT JOIN i USING(video_id, event_date) + """, "视频流量效率日报"), + _task(38, "dws_creator_traffic_daily", "DWS", "content", + ("dws_video_traffic_daily", "dim_video"), + """ + SELECT v.creator_id, d.event_date, SUM(d.exposure_count) AS exposure_count, + SUM(d.play_count) AS play_count, SUM(d.interaction_count) AS interaction_count, + AVG(d.avg_completion_rate) AS avg_completion_rate + FROM dws_video_traffic_daily d + JOIN dim_video v ON d.video_id = v.video_id + GROUP BY v.creator_id, d.event_date + """, "作者流量日报"), + _task(39, "dws_user_engagement_daily", "DWS", "user", + ("dwd_video_play_fact", "dwd_user_interaction_fact"), + """ + WITH p AS ( + SELECT user_id, event_date, COUNT(*) AS play_count, + SUM(play_duration_sec) AS play_duration_sec + FROM dwd_video_play_fact GROUP BY user_id, event_date + ), i AS ( + SELECT user_id, event_date, COUNT(*) AS interaction_count + FROM dwd_user_interaction_fact GROUP BY user_id, event_date + ) + SELECT p.user_id, p.event_date, p.play_count, p.play_duration_sec, + COALESCE(i.interaction_count, 0) AS interaction_count + FROM p LEFT JOIN i USING(user_id, event_date) + """, "用户参与度日报"), + _task(40, "dws_channel_traffic_daily", "DWS", "traffic", + ("dwd_recommend_request_fact", "dwd_video_exposure_fact"), + """ + SELECT r.channel_id, r.channel_name, r.event_date, + COUNT(DISTINCT r.request_id) AS request_count, + COUNT(e.exposure_id) AS exposure_count + FROM dwd_recommend_request_fact r + LEFT JOIN dwd_video_exposure_fact e ON r.request_id = e.request_id + GROUP BY r.channel_id, r.channel_name, r.event_date + """, "渠道流量日报"), + _task(41, "dws_category_traffic_daily", "DWS", "content", + ("dwd_video_exposure_fact", "dwd_video_play_fact"), + """ + SELECT e.category_id, e.category_name, e.event_date, + COUNT(DISTINCT e.exposure_id) AS exposure_count, + COUNT(DISTINCT p.play_id) AS play_count, + AVG(p.completion_rate) AS avg_completion_rate + FROM dwd_video_exposure_fact e + LEFT JOIN dwd_video_play_fact p ON e.exposure_id = p.exposure_id + GROUP BY e.category_id, e.category_name, e.event_date + """, "内容分类流量日报"), + _task(42, "dws_ad_slot_funnel_daily", "DWS", "ad_inventory", + ("dwd_ad_opportunity_fact", "dwd_ad_delivery_fact", "dwd_ad_click_fact", "dwd_ad_conversion_fact"), + """ + WITH o AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS request_count + FROM dwd_ad_opportunity_fact GROUP BY ad_slot_id, event_date + ), i AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS impression_count + FROM dwd_ad_delivery_fact GROUP BY ad_slot_id, event_date + ), c AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS click_count + FROM dwd_ad_click_fact GROUP BY ad_slot_id, event_date + ), v AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS conversion_count + FROM dwd_ad_conversion_fact GROUP BY ad_slot_id, event_date + ) + SELECT o.ad_slot_id, o.event_date, o.request_count, + COALESCE(i.impression_count,0) AS impression_count, + COALESCE(c.click_count,0) AS click_count, + COALESCE(v.conversion_count,0) AS conversion_count, + COALESCE(i.impression_count,0)::DOUBLE / NULLIF(o.request_count,0) AS fill_rate + FROM o LEFT JOIN i USING(ad_slot_id,event_date) + LEFT JOIN c USING(ad_slot_id,event_date) + LEFT JOIN v USING(ad_slot_id,event_date) + """, "广告位请求到转化漏斗"), + _task(43, "dws_campaign_delivery_daily", "DWS", "ad_delivery", + ("dwd_ad_attribution_wide",), + """ + SELECT campaign_id, advertiser_id, event_date, + COUNT(*) AS impression_count, SUM(is_clicked) AS click_count, + SUM(is_converted) AS conversion_count, SUM(cost) AS spend, + SUM(conversion_value) AS conversion_value + FROM dwd_ad_attribution_wide + GROUP BY campaign_id, advertiser_id, event_date + """, "计划投放效果日报"), + _task(44, "dws_creative_performance_daily", "DWS", "ad_delivery", + ("dwd_ad_attribution_wide", "dim_ad_creative"), + """ + SELECT w.ad_creative_id, c.ad_group_id, w.campaign_id, w.advertiser_id, + w.event_date, COUNT(*) AS impression_count, + SUM(w.is_clicked) AS click_count, SUM(w.is_converted) AS conversion_count, + SUM(w.cost) AS spend, SUM(w.conversion_value) AS conversion_value, + ROUND(SUM(w.is_clicked) * 100.0 / NULLIF(COUNT(*),0), 6) AS ctr, + SUM(w.is_converted)::DOUBLE / NULLIF(SUM(w.is_clicked),0) AS cvr + FROM dwd_ad_attribution_wide w + LEFT JOIN dim_ad_creative c ON w.ad_creative_id = c.ad_creative_id + GROUP BY w.ad_creative_id, c.ad_group_id, w.campaign_id, w.advertiser_id, w.event_date + """, "素材效果日报(故意保留百分数 CTR 作为治理场景)"), + _task(45, "dws_advertiser_performance_daily", "DWS", "billing", + ("dws_campaign_delivery_daily",), + """ + SELECT advertiser_id, event_date, SUM(impression_count) AS impression_count, + SUM(click_count) AS click_count, SUM(conversion_count) AS conversion_count, + SUM(spend) AS spend, SUM(conversion_value) AS conversion_value + FROM dws_campaign_delivery_daily GROUP BY advertiser_id, event_date + """, "广告主效果日报"), + _task(46, "dws_audience_performance_daily", "DWS", "experiment", + ("dwd_ad_attribution_wide", "dim_user"), + """ + SELECT u.audience_id, w.event_date, COUNT(*) AS impression_count, + SUM(w.is_clicked) AS click_count, SUM(w.is_converted) AS conversion_count, + SUM(w.cost) AS spend + FROM dwd_ad_attribution_wide w + LEFT JOIN dim_user u ON w.user_id = u.user_id + GROUP BY u.audience_id, w.event_date + """, "受众效果日报"), + _task(47, "dws_experiment_daily", "DWS", "experiment", + ("dwd_recommend_request_fact", "dwd_video_exposure_fact", "dwd_ad_opportunity_fact"), + """ + SELECT r.experiment_id, r.experiment_variant, r.event_date, + COUNT(DISTINCT r.request_id) AS request_count, + COUNT(DISTINCT e.exposure_id) AS video_exposure_count, + COUNT(DISTINCT a.ad_request_id) AS ad_request_count + FROM dwd_recommend_request_fact r + LEFT JOIN dwd_video_exposure_fact e ON r.request_id = e.request_id + LEFT JOIN dwd_ad_opportunity_fact a ON r.request_id = a.request_id + GROUP BY r.experiment_id, r.experiment_variant, r.event_date + """, "实验流量与广告机会日报"), + _task(48, "dws_monetization_funnel_daily", "DWS", "ad_inventory", + ("dwd_ad_opportunity_fact", "dwd_ad_candidate_fact", "dwd_ad_auction_fact", + "dwd_ad_delivery_fact", "dwd_ad_click_fact", "dwd_ad_conversion_fact"), + """ + SELECT o.event_date, COUNT(DISTINCT o.ad_request_id) AS request_count, + COUNT(DISTINCT c.candidate_id) AS candidate_count, + COUNT(DISTINCT CASE WHEN a.is_winner THEN a.bid_id END) AS win_count, + COUNT(DISTINCT i.ad_impression_id) AS impression_count, + COUNT(DISTINCT k.ad_click_id) AS click_count, + COUNT(DISTINCT v.conversion_id) AS conversion_count + FROM dwd_ad_opportunity_fact o + LEFT JOIN dwd_ad_candidate_fact c ON o.ad_request_id = c.ad_request_id + LEFT JOIN dwd_ad_auction_fact a ON c.candidate_id = a.candidate_id + LEFT JOIN dwd_ad_delivery_fact i ON o.ad_request_id = i.ad_request_id + LEFT JOIN dwd_ad_click_fact k ON i.ad_impression_id = k.ad_impression_id + LEFT JOIN dwd_ad_conversion_fact v ON k.ad_click_id = v.ad_click_id + GROUP BY o.event_date + """, "商业化全漏斗日报"), + _task(49, "dws_revenue_daily", "DWS", "billing", + ("dwd_billing_fact", "dwd_refund_fact"), + """ + WITH c AS ( + SELECT advertiser_id, event_date, SUM(amount) AS gross_revenue + FROM dwd_billing_fact GROUP BY advertiser_id, event_date + ), r AS ( + SELECT advertiser_id, event_date, SUM(amount) AS refund_amount + FROM dwd_refund_fact GROUP BY advertiser_id, event_date + ) + SELECT c.advertiser_id, c.event_date, c.gross_revenue, + COALESCE(r.refund_amount,0) AS refund_amount, + c.gross_revenue - COALESCE(r.refund_amount,0) AS net_revenue + FROM c LEFT JOIN r USING(advertiser_id,event_date) + """, "平台广告收入日报"), + _task(50, "dws_roi_daily", "DWS", "attribution", + ("dws_campaign_delivery_daily",), + """ + SELECT campaign_id, advertiser_id, event_date, spend, conversion_value, + conversion_value / NULLIF(spend,0) AS roi + FROM dws_campaign_delivery_daily + """, "计划 ROI 日报"), + _task(51, "dws_budget_pacing_daily", "DWS", "billing", + ("dwd_budget_fact", "dws_campaign_delivery_daily"), + """ + SELECT b.campaign_id, b.advertiser_id, b.snapshot_date AS event_date, + b.budget, b.spent AS snapshot_spent, + COALESCE(d.spend,0) AS actual_spend, + COALESCE(d.spend,0) / NULLIF(b.budget,0) AS pacing_ratio + FROM dwd_budget_fact b + LEFT JOIN dws_campaign_delivery_daily d + ON b.campaign_id = d.campaign_id AND b.snapshot_date = d.event_date + """, "预算节奏日报"), + _task(52, "dws_risk_daily", "DWS", "risk", + ("dwd_risk_event_fact",), + """ + SELECT entity_type, event_date, COUNT(*) AS signal_count, + SUM(CASE WHEN risk_score >= 0.8 THEN 1 ELSE 0 END) AS high_risk_count, + AVG(risk_score) AS avg_risk_score + FROM dwd_risk_event_fact GROUP BY entity_type, event_date + """, "内容、广告与流量风险日报"), +] + +_ADS_TASKS = [ + _task(53, "ads_traffic_overview", "ADS", "traffic", + ("dws_channel_traffic_daily", "dws_video_traffic_daily"), + """ + SELECT c.event_date, SUM(c.request_count) AS request_count, + SUM(c.exposure_count) AS exposure_count, + SUM(v.play_count) AS play_count, + SUM(v.valid_play_count) AS valid_play_count + FROM dws_channel_traffic_daily c + LEFT JOIN dws_video_traffic_daily v ON c.event_date = v.event_date + GROUP BY c.event_date + """, "平台流量总览"), + _task(54, "ads_creator_dashboard", "ADS", "content", + ("dws_creator_traffic_daily", "dim_creator"), + """ + SELECT d.*, c.creator_tier, + d.interaction_count::DOUBLE / NULLIF(d.play_count,0) AS interaction_rate + FROM dws_creator_traffic_daily d + LEFT JOIN dim_creator c ON d.creator_id = c.creator_id + """, "作者流量与互动看板"), + _task(55, "ads_content_efficiency_report", "ADS", "content", + ("dws_category_traffic_daily",), + """ + SELECT *, play_count::DOUBLE / NULLIF(exposure_count,0) AS play_rate + FROM dws_category_traffic_daily + """, "内容分类效率报告"), + _task(56, "ads_ad_operation_dashboard", "ADS", "ad_inventory", + ("dws_monetization_funnel_daily", "dws_revenue_daily"), + """ + WITH revenue AS ( + SELECT event_date, SUM(net_revenue) AS net_revenue + FROM dws_revenue_daily GROUP BY event_date + ) + SELECT f.*, COALESCE(r.net_revenue,0) AS net_revenue, + f.impression_count::DOUBLE / NULLIF(f.request_count,0) AS fill_rate, + COALESCE(r.net_revenue,0) * 1000.0 / NULLIF(f.impression_count,0) AS ecpm + FROM dws_monetization_funnel_daily f + LEFT JOIN revenue r ON f.event_date = r.event_date + """, "广告运营总览"), + _task(57, "ads_advertiser_report", "ADS", "billing", + ("dws_advertiser_performance_daily", "dws_revenue_daily", "dim_advertiser"), + """ + SELECT p.*, a.advertiser_name, a.industry, r.net_revenue, + p.conversion_value / NULLIF(p.spend,0) AS roi + FROM dws_advertiser_performance_daily p + LEFT JOIN dws_revenue_daily r USING(advertiser_id,event_date) + LEFT JOIN dim_advertiser a USING(advertiser_id) + """, "广告主经营报告"), + _task(58, "ads_campaign_monitor", "ADS", "ad_delivery", + ("dws_campaign_delivery_daily", "dws_budget_pacing_daily", "dim_campaign"), + """ + SELECT d.*, c.objective, c.status, b.budget, b.pacing_ratio, + d.click_count::DOUBLE / NULLIF(d.impression_count,0) AS ctr, + d.conversion_count::DOUBLE / NULLIF(d.click_count,0) AS cvr + FROM dws_campaign_delivery_daily d + LEFT JOIN dws_budget_pacing_daily b USING(campaign_id,advertiser_id,event_date) + LEFT JOIN dim_campaign c USING(campaign_id,advertiser_id) + """, "计划投放与预算监控"), + _task(59, "ads_creative_report", "ADS", "ad_delivery", + ("dws_creative_performance_daily", "dim_ad_creative"), + """ + SELECT p.*, c.creative_format, + CASE + WHEN p.ctr >= 0.05 THEN 'A_excellent' + WHEN p.ctr >= 0.03 THEN 'B_good' + WHEN p.ctr >= 0.01 THEN 'C_normal' + ELSE 'D_poor' + END AS ctr_level, + RANK() OVER ( + PARTITION BY p.advertiser_id, p.event_date ORDER BY p.ctr DESC + ) AS ctr_rank + FROM dws_creative_performance_daily p + LEFT JOIN dim_ad_creative c USING(ad_creative_id,ad_group_id) + """, "素材效果与 CTR 分级报告"), + _task(60, "ads_revenue_dashboard", "ADS", "billing", + ("dws_revenue_daily", "dws_roi_daily"), + """ + SELECT r.event_date, SUM(r.gross_revenue) AS gross_revenue, + SUM(r.refund_amount) AS refund_amount, SUM(r.net_revenue) AS net_revenue, + AVG(i.roi) AS avg_roi + FROM dws_revenue_daily r + LEFT JOIN dws_roi_daily i ON r.event_date = i.event_date + GROUP BY r.event_date + """, "商业收入与 ROI 驾驶舱"), + _task(61, "ads_experiment_report", "ADS", "experiment", + ("dws_experiment_daily",), + """ + SELECT *, ad_request_count::DOUBLE / NULLIF(request_count,0) AS ad_request_rate, + video_exposure_count::DOUBLE / NULLIF(request_count,0) AS exposure_per_request + FROM dws_experiment_daily + """, "实验流量与商业化效果报告"), + _task(62, "ads_risk_dashboard", "ADS", "risk", + ("dws_risk_daily",), + """ + SELECT *, high_risk_count::DOUBLE / NULLIF(signal_count,0) AS high_risk_rate, + CASE WHEN avg_risk_score >= 0.8 THEN 'critical' + WHEN avg_risk_score >= 0.5 THEN 'warning' + ELSE 'normal' END AS risk_level + FROM dws_risk_daily + """, "商业化质量与风险驾驶舱"), +] + +TASKS: tuple[TaskSpec, ...] = tuple([ + *_stg_tasks(), + *_DWD_TASKS, + *_DWS_TASKS, + *_ADS_TASKS, +]) + +ALL_TABLES: tuple[str, ...] = tuple([*BASE_TABLES, *(task.target for task in TASKS)]) + + +def layer_counts() -> dict[str, int]: + counts = Counter(table.layer for table in BASE_TABLES.values()) + counts.update(task.layer for task in TASKS) + return dict(counts) + + +def validate_catalog() -> None: + """验证规模、唯一性、依赖存在性和拓扑顺序。""" + if len(BASE_TABLES) != 32 or len(TASKS) != 62 or len(ALL_TABLES) != 94: + raise ValueError("catalog 规模必须为 32 基础表 + 62 派生表 = 94 表") + if len(set(ALL_TABLES)) != len(ALL_TABLES): + raise ValueError("catalog 存在重复表名") + if set(layer_counts()) != {"ODS", "STG", "DIM", "DWD", "DWS", "ADS"}: + raise ValueError("catalog 层级不完整") + if {table.domain for table in BASE_TABLES.values()} | {task.domain for task in TASKS} != DOMAINS: + raise ValueError("catalog 业务主题不完整") + + available = set(BASE_TABLES) + for task in TASKS: + missing = set(task.dependencies) - available + if missing: + raise ValueError(f"{task.target} 依赖未定义或顺序错误: {sorted(missing)}") + available.add(task.target) + + +validate_catalog() diff --git a/examples/video_commercial_warehouse/generate_sql.py b/examples/video_commercial_warehouse/generate_sql.py new file mode 100644 index 0000000..27d9048 --- /dev/null +++ b/examples/video_commercial_warehouse/generate_sql.py @@ -0,0 +1,83 @@ +"""从权威 catalog 确定性生成 62 个 SQL 文件与 manifest。""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from examples.video_commercial_warehouse.catalog import ( + ALL_TABLES, + BASE_TABLES, + TASKS, + layer_counts, + validate_catalog, +) + + +ROOT = Path(__file__).resolve().parent + + +def generate_sql(output_dir: Path) -> list[Path]: + """物化 SQL 与 manifest;同一 catalog 重复生成时字节级一致。""" + validate_catalog() + output_dir = Path(output_dir) + sql_dir = output_dir / "sql" + sql_dir.mkdir(parents=True, exist_ok=True) + + expected_names = set() + paths: list[Path] = [] + for task in TASKS: + filename = f"{task.order:03d}_{task.target}.sql" + expected_names.add(filename) + path = sql_dir / filename + path.write_text(task.sql, encoding="utf-8") + paths.append(path) + + # 清理 catalog 已删除/重命名后遗留的生成 SQL,避免执行幽灵任务。 + for stale in sql_dir.glob("*.sql"): + if stale.name not in expected_names: + stale.unlink() + + derived = {task.target: task for task in TASKS} + tables = [] + for name in ALL_TABLES: + if name in BASE_TABLES: + tables.append(BASE_TABLES[name].to_dict()) + else: + task = derived[name] + tables.append({ + "name": task.target, + "layer": task.layer, + "domain": task.domain, + "columns": [], + }) + + manifest = { + "name": "video_commercial_warehouse", + "description": "视频平台流量与广告商业化完整数仓", + "table_count": len(ALL_TABLES), + "base_table_count": len(BASE_TABLES), + "task_count": len(TASKS), + "layer_counts": dict(sorted(layer_counts().items())), + "tables": tables, + "tasks": [task.to_dict() for task in TASKS], + } + (output_dir / "manifest.json").write_text( + json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return paths + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="生成视频平台商业化数仓 SQL") + parser.add_argument("--output", type=Path, default=ROOT) + args = parser.parse_args(argv) + paths = generate_sql(args.output) + print(f"已生成 {len(paths)} 个 SQL 任务: {args.output.resolve() / 'sql'}") + print(f"manifest: {args.output.resolve() / 'manifest.json'}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/examples/video_commercial_warehouse/governance.py b/examples/video_commercial_warehouse/governance.py new file mode 100644 index 0000000..52c0982 --- /dev/null +++ b/examples/video_commercial_warehouse/governance.py @@ -0,0 +1,521 @@ +"""视频商业化数仓的真实 DuckDB CTR 治理闭环。""" +from __future__ import annotations + +import argparse +import json +import re +import shutil +from dataclasses import asdict +from pathlib import Path +from typing import Any + +import duckdb + +from examples.video_commercial_warehouse.catalog import ( + ALL_TABLES, + BASE_TABLES, + TASKS, +) +from examples.video_commercial_warehouse.generate_sql import ROOT, generate_sql +from examples.video_commercial_warehouse.runtime import ( + RunResult, + build_warehouse, + collect_table_rows, + rerun_tasks, +) +from sqlgraph.agent import GovernanceLoop +from sqlgraph.api import build_graph +from sqlgraph.autonomy import ( + AuthorizationScope, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) +from sqlgraph.contract import BusinessContract, GovernanceIssue +from sqlgraph.evidence import build_evidence_subgraph +from sqlgraph.lineage import drilldown +from sqlgraph.metrics import blast_score, bridge_score, with_residual_risk + + +TASK_ID = "VCW-CTR-001" +TARGET_TABLE = "dws_creative_performance_daily" +TARGET_FILE = "044_dws_creative_performance_daily.sql" +DOWNSTREAM_TABLE = "ads_creative_report" +EXPECTED_THRESHOLDS = (0.05, 0.03, 0.01) + +BEFORE_CTR_EXPR = ( + "ROUND(SUM(w.is_clicked) * 100.0 / NULLIF(COUNT(*),0), 6) AS ctr" +) +AFTER_CTR_EXPR = ( + "ROUND(SUM(w.is_clicked) / NULLIF(COUNT(*),0), 6) AS ctr" +) + + +class GovernanceError(RuntimeError): + """治理流程无法安全继续。""" + + +def _ensure_generated(root: Path) -> Path: + sql_dir = root / "sql" + if len(list(sql_dir.glob("*.sql"))) == 62: + return sql_dir + generate_sql(root) + return sql_dir + + +def _snapshot_sql(sql_dir: Path, output_dir: Path) -> Path: + before = output_dir / "before" + before_sql = before / "sql" + if not before_sql.exists(): + shutil.copytree(sql_dir, before_sql) + return before_sql + + +def _apply_fix(target: Path, output_dir: Path) -> dict[str, Any]: + source = target.read_text(encoding="utf-8") + before_count = source.count(BEFORE_CTR_EXPR) + after_count = source.count(AFTER_CTR_EXPR) + backup = output_dir / "before" / target.name + backup.parent.mkdir(parents=True, exist_ok=True) + + if before_count == 0 and after_count == 1: + return { + "changed": False, + "status": "already_compliant", + "target": str(target), + "backup": str(backup) if backup.exists() else None, + } + if before_count != 1 or after_count != 0: + raise GovernanceError( + "CTR 表达式无法唯一识别:" + f"before={before_count}, after={after_count}" + ) + if not backup.exists(): + backup.write_text(source, encoding="utf-8") + target.write_text(source.replace(BEFORE_CTR_EXPR, AFTER_CTR_EXPR), encoding="utf-8") + return { + "changed": True, + "status": "fixed", + "target": str(target), + "backup": str(backup), + } + + +def _thresholds(sql_path: Path) -> tuple[float, ...]: + text = sql_path.read_text(encoding="utf-8") + return tuple(float(value) for value in re.findall(r"p\.ctr\s*>=\s*([0-9.]+)", text)) + + +def _affected_tasks(target: str) -> list[str]: + """从 catalog 依赖计算目标及其全部派生下游,保持拓扑顺序。""" + affected = {target} + changed = True + while changed: + changed = False + for task in TASKS: + if task.target not in affected and set(task.dependencies) & affected: + affected.add(task.target) + changed = True + return [task.target for task in TASKS if task.target in affected] + + +def _query_runtime(database: Path) -> dict[str, Any]: + with duckdb.connect(str(database), read_only=True) as con: + summary = con.execute(""" + SELECT + COUNT(*) AS row_count, + COUNT(*) FILTER (WHERE ctr < 0 OR ctr > 1) AS invalid_ctr_rows, + COUNT(DISTINCT ctr_level) AS grade_count, + MIN(ctr) AS min_ctr, + MAX(ctr) AS max_ctr, + COUNT(*) FILTER ( + WHERE ctr_level <> + CASE WHEN ctr >= 0.05 THEN 'A_excellent' + WHEN ctr >= 0.03 THEN 'B_good' + WHEN ctr >= 0.01 THEN 'C_normal' + ELSE 'D_poor' END + ) AS grade_mismatch_rows + FROM ads_creative_report + """).fetchone() + samples = con.execute(""" + SELECT ad_creative_id, event_date, impression_count, click_count, + ROUND(ctr, 6) AS ctr, ctr_level + FROM ads_creative_report + ORDER BY ctr DESC, ad_creative_id, event_date + LIMIT 12 + """).fetchall() + return { + "row_count": int(summary[0]), + "invalid_ctr_rows": int(summary[1]), + "grade_count": int(summary[2]), + "min_ctr": float(summary[3] or 0), + "max_ctr": float(summary[4] or 0), + "grade_mismatch_rows": int(summary[5]), + "samples": [ + { + "ad_creative_id": int(row[0]), + "event_date": str(row[1]), + "impression_count": int(row[2]), + "click_count": int(row[3]), + "ctr": float(row[4]), + "ctr_level": row[5], + } + for row in samples + ], + } + + +def _graph_diff(before_graph, after_graph) -> dict[str, Any]: + before_nodes = {node.id for node in before_graph.nodes} + after_nodes = {node.id for node in after_graph.nodes} + before_edges = {edge.id for edge in before_graph.edges} + after_edges = {edge.id for edge in after_graph.edges} + added = sorted((after_nodes | after_edges) - (before_nodes | before_edges)) + removed = sorted((before_nodes | before_edges) - (after_nodes | after_edges)) + return { + "added": added, + "removed": removed, + "added_count": len(added), + "removed_count": len(removed), + } + + +def _evidence_payload(evidence) -> dict[str, Any]: + return { + "task_id": evidence.task_id, + "version_id": evidence.version_id, + "node_ids": sorted(evidence.node_ids), + "edge_ids": sorted(evidence.edge_ids), + "coverage": evidence.coverage, + "gaps": evidence.gaps, + } + + +def _run_payload(result: RunResult) -> dict[str, Any]: + return result.to_dict() + + +def _warehouse_payload(database: Path, before_run: RunResult) -> dict[str, Any]: + """汇总报告需要的 94 表、62 任务及真实行数。""" + with duckdb.connect(str(database), read_only=True) as con: + table_rows = collect_table_rows(con) + task_by_target = {task.target: task for task in TASKS} + run_by_target = {run.target: run for run in before_run.task_runs} + downstream: dict[str, list[str]] = {name: [] for name in ALL_TABLES} + for task in TASKS: + for dependency in task.dependencies: + downstream.setdefault(dependency, []).append(task.target) + + tables = [] + for name in ALL_TABLES: + if name in BASE_TABLES: + spec = BASE_TABLES[name] + layer, domain, columns = spec.layer, spec.domain, [ + {"name": column, "type": data_type} + for column, data_type in spec.columns + ] + upstream = [] + sql_file = None + else: + task = task_by_target[name] + layer, domain, columns = task.layer, task.domain, [] + upstream = list(task.dependencies) + sql_file = f"{task.order:03d}_{task.target}.sql" + tables.append({ + "name": name, + "layer": layer, + "domain": domain, + "row_count": table_rows[name], + "upstream": upstream, + "downstream": sorted(downstream.get(name, [])), + "columns": columns, + "sql_file": sql_file, + }) + + return { + "tables": tables, + "tasks": [ + { + **task.to_dict(), + "row_count": table_rows[task.target], + "elapsed_ms": ( + run_by_target[task.target].elapsed_ms + if task.target in run_by_target else None + ), + } + for task in TASKS + ], + } + + +def run_ctr_governance( + root: Path, + output_dir: Path, + *, + profile: str = "smoke", +) -> dict[str, Any]: + """执行全量基线、真实 SQL 修复、影响 DAG 重跑和真实运行态验证。""" + root = Path(root).resolve() + output_dir = Path(output_dir).resolve() + output_dir.mkdir(parents=True, exist_ok=True) + sql_dir = _ensure_generated(root) + target = sql_dir / TARGET_FILE + report_sql = sql_dir / "059_ads_creative_report.sql" + before_sql_dir = _snapshot_sql(sql_dir, output_dir) + before_db = output_dir / "before.duckdb" + after_db = output_dir / "after.duckdb" + + current = target.read_text(encoding="utf-8") + is_conflict = BEFORE_CTR_EXPR in current and AFTER_CTR_EXPR not in current + is_compliant = AFTER_CTR_EXPR in current and BEFORE_CTR_EXPR not in current + if not (is_conflict or is_compliant): + raise GovernanceError("目标 CTR SQL 不处于可识别的冲突或合规状态") + + if is_conflict or not before_db.exists(): + before_run = build_warehouse(before_db, profile=profile, sql_dir=sql_dir) + else: + # 幂等重跑沿用第一次保存的真实冲突前数据库。 + before_run = RunResult(str(before_db), profile, tuple(), {}) + before_runtime = _query_runtime(before_db) + before_graph = build_graph(str(before_sql_dir), dialect="duckdb") + + change = _apply_fix(target, output_dir) + affected_tasks = _affected_tasks(TARGET_TABLE) + try: + if change["changed"]: + shutil.copy2(before_db, after_db) + reruns = rerun_tasks(after_db, affected_tasks, sql_dir=sql_dir) + after_run = { + "mode": "incremental", + "success_count": len(reruns), + "task_runs": [run.to_dict() for run in reruns], + } + else: + rebuilt = build_warehouse(after_db, profile=profile, sql_dir=sql_dir) + after_run = {"mode": "no_op_validation", **_run_payload(rebuilt)} + except Exception as exc: + backup = output_dir / "before" / TARGET_FILE + if change["changed"] and backup.exists(): + shutil.copy2(backup, target) + raise GovernanceError(f"增量重跑失败,SQL 已恢复: {exc}") from exc + + after_runtime = _query_runtime(after_db) + after_graph = build_graph(str(sql_dir), dialect="duckdb") + thresholds = _thresholds(report_sql) + threshold_consistent = thresholds == EXPECTED_THRESHOLDS + runtime_ok = ( + after_runtime["invalid_ctr_rows"] == 0 + and after_runtime["grade_mismatch_rows"] == 0 + and threshold_consistent + ) + + blast = blast_score(before_graph, TARGET_TABLE) + bridge = bridge_score(before_graph, TARGET_TABLE) + lineage = drilldown(after_graph, TARGET_TABLE, DOWNSTREAM_TABLE) + evidence = build_evidence_subgraph( + after_graph, + TASK_ID, + [TARGET_TABLE, DOWNSTREAM_TABLE], + intent="caliber_repair", + ) + contract = BusinessContract( + metric="creative_ctr", + version="v2", + definition="click_count / impression_count,范围 [0,1]", + data_type="DOUBLE", + unit="ratio", + null_behavior="impression_count=0 时为 NULL", + owner="commercial-data-governance", + ) + issue = None + if not threshold_consistent: + issue = GovernanceIssue( + metric=contract.metric, + contract_version=contract.version, + implemented_semantics=f"ctr thresholds={thresholds}", + expected_semantics=f"ctr thresholds={EXPECTED_THRESHOLDS}", + ) + + action = GovernanceAction( + action_type="fix_creative_ctr_ratio", + evidence_version=evidence.version_id, + evidence_grounded=bool(lineage.get("found")), + reversibility=ReversibilityEvidence( + state_restorable=True, + external_effects_controlled=True, + rollback_verified=True, + references=(str(change.get("backup") or "existing-backup"),), + ), + blast_radius=min(float(blast["score"]) / 10.0, 1.0), + object_risk=0.4, + authorization_scope=AuthorizationScope.SINGLE_L3, + historical_reliability=0.95, + roi=0.9, + ) + decision = decide_autonomy(action) + trail = GovernanceLoop(after_graph).run( + TASK_ID, + TARGET_TABLE, + DOWNSTREAM_TABLE, + action, + governance_issue=issue, + runtime_observed={ + "value": after_runtime["max_ctr"], + "reports_recomputed": runtime_ok, + "new_alerts": 0 if runtime_ok else 1, + }, + runtime_target=None, + ) + audit = trail.to_dict() + if not change["changed"]: + execute = next(step for step in audit["steps"] if step["step"] == "execute") + execute["detail"] = { + "executed": False, + "mode": "no_op/already_compliant", + } + execute["transition"] = "already_compliant->verify" + + result = { + "catalog": { + "table_count": 94, + "task_count": 62, + "profile": profile, + }, + "warehouse": _warehouse_payload(after_db, before_run), + "scenario": { + "task_id": TASK_ID, + "title": "视频商业化素材 CTR 口径治理", + "target_table": TARGET_TABLE, + "target_file": str(target), + "downstream_table": DOWNSTREAM_TABLE, + "status": "fixed" if change["changed"] else "already_compliant", + }, + "change": { + **change, + "before_expression": BEFORE_CTR_EXPR, + "after_expression": AFTER_CTR_EXPR, + }, + "database": { + "before": str(before_db), + "after": str(after_db), + "before_build": _run_payload(before_run), + "after_build": after_run, + }, + "graph": { + "before_stats": before_graph.stats(), + "after_stats": after_graph.stats(), + "coverage": after_graph.metadata.get("coverage", {}), + "environment": after_graph.metadata.get("environment", {}), + }, + "diff": _graph_diff(before_graph, after_graph), + "impact": { + "reachable": blast["reachable"], + "blast": blast, + "bridge": bridge, + "executable_tasks": affected_tasks, + }, + "rerun": { + "tasks": affected_tasks if change["changed"] else [], + "success_count": after_run["success_count"], + "mode": after_run["mode"], + }, + "lineage_evidence": lineage, + "evidence": _evidence_payload(evidence), + "contract": asdict(contract), + "governance_issue": issue.to_dict() if issue else None, + "decision": decision.to_dict(), + "runtime": { + "source": "duckdb_query", + "thresholds": list(thresholds), + "expected_thresholds": list(EXPECTED_THRESHOLDS), + "threshold_consistent": threshold_consistent, + "before": before_runtime, + "after": after_runtime, + }, + "verify": trail.verify_report, + "audit": audit, + "residual": with_residual_risk({ + "verdict": "真实 DuckDB 结果已验证,生产环境仍需监控实际数据漂移", + }), + } + (output_dir / "operation.json").write_text( + json.dumps(result, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return result + + +def restore_governance(root: Path, output_dir: Path) -> dict[str, Any]: + """恢复治理前 SQL 与 DuckDB,并把当前报告显式更新为 restored。""" + root = Path(root).resolve() + output_dir = Path(output_dir).resolve() + target = root / "sql" / TARGET_FILE + sql_backup = output_dir / "before" / TARGET_FILE + database_backup = output_dir / "before.duckdb" + database_current = output_dir / "after.duckdb" + operation_path = output_dir / "operation.json" + if not sql_backup.is_file() or not database_backup.is_file(): + raise GovernanceError("缺少 SQL 或 DuckDB 治理前快照,无法恢复") + if not operation_path.is_file(): + raise GovernanceError("缺少 operation.json,无法生成恢复审计") + + shutil.copy2(sql_backup, target) + shutil.copy2(database_backup, database_current) + result = json.loads(operation_path.read_text(encoding="utf-8")) + result["scenario"]["status"] = "restored" + result["change"]["changed"] = True + result["change"]["status"] = "restored" + result["rerun"] = {"tasks": [], "success_count": 0, "mode": "restore_snapshot"} + result["runtime"]["after"] = _query_runtime(database_current) + result["verify"]["closed_loop_status"] = "incomplete" + result["audit"]["outcome"] = "restored" + result["audit"]["steps"].append({ + "step": "restore", + "evidence_version": result["evidence"]["version_id"], + "detail": { + "sql_restored": True, + "database_restored": True, + }, + "transition": "closed->restored", + }) + operation_path.write_text( + json.dumps(result, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="执行完整视频商业化数仓 CTR 治理") + parser.add_argument("--root", type=Path, default=ROOT) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--profile", choices=("smoke", "demo"), default="smoke") + parser.add_argument("--restore", action="store_true", help="恢复治理前 SQL 与 DuckDB 快照") + parser.add_argument("--no-open", action="store_true", help="生成报告后不打开浏览器") + args = parser.parse_args(argv) + try: + if args.restore: + result = restore_governance(args.root, args.output) + else: + result = run_ctr_governance(args.root, args.output, profile=args.profile) + except GovernanceError as exc: + print(f"治理失败: {exc}") + return 1 + from examples.video_commercial_warehouse.report import render_report + + report_path = render_report(result, args.output / "warehouse_report.html") + print( + f"完成: {result['catalog']['table_count']} 表 / " + f"{result['catalog']['task_count']} 任务 / " + f"verify={result['verify']['closed_loop_status']}" + ) + print(f"审计: {args.output.resolve() / 'operation.json'}") + print(f"报告: file://{report_path}") + if not args.no_open: + import webbrowser + webbrowser.open(report_path.as_uri()) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/examples/video_commercial_warehouse/manifest.json b/examples/video_commercial_warehouse/manifest.json new file mode 100644 index 0000000..924a7d9 --- /dev/null +++ b/examples/video_commercial_warehouse/manifest.json @@ -0,0 +1,2081 @@ +{ + "base_table_count": 32, + "description": "视频平台流量与广告商业化完整数仓", + "layer_counts": { + "ADS": 10, + "DIM": 14, + "DWD": 18, + "DWS": 16, + "ODS": 18, + "STG": 18 + }, + "name": "video_commercial_warehouse", + "table_count": 94, + "tables": [ + { + "columns": [ + [ + "request_id", + "BIGINT" + ], + [ + "session_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "device_id", + "BIGINT" + ], + [ + "geo_id", + "BIGINT" + ], + [ + "channel_id", + "BIGINT" + ], + [ + "experiment_id", + "BIGINT" + ], + [ + "request_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "recommendation", + "layer": "ODS", + "name": "ods_recommend_request" + }, + { + "columns": [ + [ + "exposure_id", + "BIGINT" + ], + [ + "request_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "video_id", + "BIGINT" + ], + [ + "position", + "INTEGER" + ], + [ + "exposure_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "traffic", + "layer": "ODS", + "name": "ods_video_exposure" + }, + { + "columns": [ + [ + "play_id", + "BIGINT" + ], + [ + "exposure_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "video_id", + "BIGINT" + ], + [ + "play_duration_sec", + "DOUBLE" + ], + [ + "completion_rate", + "DOUBLE" + ], + [ + "play_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "traffic", + "layer": "ODS", + "name": "ods_video_play" + }, + { + "columns": [ + [ + "interaction_id", + "BIGINT" + ], + [ + "play_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "video_id", + "BIGINT" + ], + [ + "interaction_type", + "VARCHAR" + ], + [ + "event_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "content", + "layer": "ODS", + "name": "ods_user_interaction" + }, + { + "columns": [ + [ + "search_id", + "BIGINT" + ], + [ + "session_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "query", + "VARCHAR" + ], + [ + "result_count", + "INTEGER" + ], + [ + "event_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "traffic", + "layer": "ODS", + "name": "ods_search_event" + }, + { + "columns": [ + [ + "session_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "device_id", + "BIGINT" + ], + [ + "channel_id", + "BIGINT" + ], + [ + "start_time", + "TIMESTAMP" + ], + [ + "end_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "user", + "layer": "ODS", + "name": "ods_user_session" + }, + { + "columns": [ + [ + "ad_request_id", + "BIGINT" + ], + [ + "request_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "ad_slot_id", + "BIGINT" + ], + [ + "audience_id", + "BIGINT" + ], + [ + "request_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "ad_inventory", + "layer": "ODS", + "name": "ods_ad_request" + }, + { + "columns": [ + [ + "candidate_id", + "BIGINT" + ], + [ + "ad_request_id", + "BIGINT" + ], + [ + "ad_creative_id", + "BIGINT" + ], + [ + "ad_group_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "advertiser_id", + "BIGINT" + ], + [ + "predicted_ctr", + "DOUBLE" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "ad_inventory", + "layer": "ODS", + "name": "ods_ad_candidate" + }, + { + "columns": [ + [ + "bid_id", + "BIGINT" + ], + [ + "candidate_id", + "BIGINT" + ], + [ + "bid_price", + "DOUBLE" + ], + [ + "is_winner", + "BOOLEAN" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "ad_delivery", + "layer": "ODS", + "name": "ods_ad_bid" + }, + { + "columns": [ + [ + "ad_impression_id", + "BIGINT" + ], + [ + "ad_request_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "ad_creative_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "advertiser_id", + "BIGINT" + ], + [ + "ad_slot_id", + "BIGINT" + ], + [ + "impression_time", + "TIMESTAMP" + ], + [ + "cost", + "DOUBLE" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "ad_delivery", + "layer": "ODS", + "name": "ods_ad_impression" + }, + { + "columns": [ + [ + "ad_click_id", + "BIGINT" + ], + [ + "ad_impression_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "ad_creative_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "click_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "ad_delivery", + "layer": "ODS", + "name": "ods_ad_click" + }, + { + "columns": [ + [ + "conversion_id", + "BIGINT" + ], + [ + "ad_click_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "ad_creative_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "conversion_type", + "VARCHAR" + ], + [ + "conversion_value", + "DOUBLE" + ], + [ + "conversion_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "attribution", + "layer": "ODS", + "name": "ods_ad_conversion" + }, + { + "columns": [ + [ + "charge_id", + "BIGINT" + ], + [ + "ad_impression_id", + "BIGINT" + ], + [ + "advertiser_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "amount", + "DOUBLE" + ], + [ + "charge_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "billing", + "layer": "ODS", + "name": "ods_ad_charge" + }, + { + "columns": [ + [ + "refund_id", + "BIGINT" + ], + [ + "charge_id", + "BIGINT" + ], + [ + "advertiser_id", + "BIGINT" + ], + [ + "amount", + "DOUBLE" + ], + [ + "reason", + "VARCHAR" + ], + [ + "refund_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "billing", + "layer": "ODS", + "name": "ods_ad_refund" + }, + { + "columns": [ + [ + "snapshot_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "budget", + "DOUBLE" + ], + [ + "spent", + "DOUBLE" + ], + [ + "snapshot_date", + "DATE" + ] + ], + "domain": "billing", + "layer": "ODS", + "name": "ods_budget_snapshot" + }, + { + "columns": [ + [ + "audit_id", + "BIGINT" + ], + [ + "video_id", + "BIGINT" + ], + [ + "audit_status", + "VARCHAR" + ], + [ + "risk_score", + "DOUBLE" + ], + [ + "audit_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "risk", + "layer": "ODS", + "name": "ods_content_audit" + }, + { + "columns": [ + [ + "audit_id", + "BIGINT" + ], + [ + "ad_creative_id", + "BIGINT" + ], + [ + "audit_status", + "VARCHAR" + ], + [ + "risk_score", + "DOUBLE" + ], + [ + "audit_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "risk", + "layer": "ODS", + "name": "ods_ad_audit" + }, + { + "columns": [ + [ + "signal_id", + "BIGINT" + ], + [ + "request_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "signal_type", + "VARCHAR" + ], + [ + "risk_score", + "DOUBLE" + ], + [ + "event_time", + "TIMESTAMP" + ], + [ + "event_date", + "DATE" + ] + ], + "domain": "risk", + "layer": "ODS", + "name": "ods_invalid_traffic_signal" + }, + { + "columns": [ + [ + "user_id", + "BIGINT" + ], + [ + "register_date", + "DATE" + ], + [ + "country_code", + "VARCHAR" + ], + [ + "age_bucket", + "VARCHAR" + ], + [ + "audience_id", + "BIGINT" + ] + ], + "domain": "user", + "layer": "DIM", + "name": "dim_user" + }, + { + "columns": [ + [ + "creator_id", + "BIGINT" + ], + [ + "user_id", + "BIGINT" + ], + [ + "creator_tier", + "VARCHAR" + ] + ], + "domain": "content", + "layer": "DIM", + "name": "dim_creator" + }, + { + "columns": [ + [ + "video_id", + "BIGINT" + ], + [ + "creator_id", + "BIGINT" + ], + [ + "category_id", + "BIGINT" + ], + [ + "duration_sec", + "INTEGER" + ], + [ + "publish_date", + "DATE" + ], + [ + "content_status", + "VARCHAR" + ] + ], + "domain": "content", + "layer": "DIM", + "name": "dim_video" + }, + { + "columns": [ + [ + "category_id", + "BIGINT" + ], + [ + "category_name", + "VARCHAR" + ] + ], + "domain": "content", + "layer": "DIM", + "name": "dim_content_category" + }, + { + "columns": [ + [ + "device_id", + "BIGINT" + ], + [ + "device_type", + "VARCHAR" + ], + [ + "os", + "VARCHAR" + ] + ], + "domain": "user", + "layer": "DIM", + "name": "dim_device" + }, + { + "columns": [ + [ + "geo_id", + "BIGINT" + ], + [ + "country_code", + "VARCHAR" + ], + [ + "region", + "VARCHAR" + ] + ], + "domain": "user", + "layer": "DIM", + "name": "dim_geo" + }, + { + "columns": [ + [ + "channel_id", + "BIGINT" + ], + [ + "channel_name", + "VARCHAR" + ] + ], + "domain": "traffic", + "layer": "DIM", + "name": "dim_channel" + }, + { + "columns": [ + [ + "advertiser_id", + "BIGINT" + ], + [ + "advertiser_name", + "VARCHAR" + ], + [ + "industry", + "VARCHAR" + ] + ], + "domain": "billing", + "layer": "DIM", + "name": "dim_advertiser" + }, + { + "columns": [ + [ + "campaign_id", + "BIGINT" + ], + [ + "advertiser_id", + "BIGINT" + ], + [ + "budget", + "DOUBLE" + ], + [ + "objective", + "VARCHAR" + ], + [ + "status", + "VARCHAR" + ] + ], + "domain": "ad_delivery", + "layer": "DIM", + "name": "dim_campaign" + }, + { + "columns": [ + [ + "ad_group_id", + "BIGINT" + ], + [ + "campaign_id", + "BIGINT" + ], + [ + "audience_id", + "BIGINT" + ], + [ + "bid_type", + "VARCHAR" + ], + [ + "bid_value", + "DOUBLE" + ] + ], + "domain": "ad_delivery", + "layer": "DIM", + "name": "dim_ad_group" + }, + { + "columns": [ + [ + "ad_creative_id", + "BIGINT" + ], + [ + "ad_group_id", + "BIGINT" + ], + [ + "video_id", + "BIGINT" + ], + [ + "creative_format", + "VARCHAR" + ], + [ + "status", + "VARCHAR" + ] + ], + "domain": "ad_delivery", + "layer": "DIM", + "name": "dim_ad_creative" + }, + { + "columns": [ + [ + "ad_slot_id", + "BIGINT" + ], + [ + "slot_name", + "VARCHAR" + ], + [ + "scene", + "VARCHAR" + ], + [ + "floor_price", + "DOUBLE" + ] + ], + "domain": "ad_inventory", + "layer": "DIM", + "name": "dim_ad_slot" + }, + { + "columns": [ + [ + "experiment_id", + "BIGINT" + ], + [ + "experiment_name", + "VARCHAR" + ], + [ + "variant", + "VARCHAR" + ] + ], + "domain": "experiment", + "layer": "DIM", + "name": "dim_experiment" + }, + { + "columns": [ + [ + "audience_id", + "BIGINT" + ], + [ + "audience_name", + "VARCHAR" + ], + [ + "strategy", + "VARCHAR" + ] + ], + "domain": "experiment", + "layer": "DIM", + "name": "dim_audience" + }, + { + "columns": [], + "domain": "recommendation", + "layer": "STG", + "name": "stg_recommend_request" + }, + { + "columns": [], + "domain": "traffic", + "layer": "STG", + "name": "stg_video_exposure" + }, + { + "columns": [], + "domain": "traffic", + "layer": "STG", + "name": "stg_video_play" + }, + { + "columns": [], + "domain": "content", + "layer": "STG", + "name": "stg_user_interaction" + }, + { + "columns": [], + "domain": "traffic", + "layer": "STG", + "name": "stg_search_event" + }, + { + "columns": [], + "domain": "user", + "layer": "STG", + "name": "stg_user_session" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "STG", + "name": "stg_ad_request" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "STG", + "name": "stg_ad_candidate" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "STG", + "name": "stg_ad_bid" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "STG", + "name": "stg_ad_impression" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "STG", + "name": "stg_ad_click" + }, + { + "columns": [], + "domain": "attribution", + "layer": "STG", + "name": "stg_ad_conversion" + }, + { + "columns": [], + "domain": "billing", + "layer": "STG", + "name": "stg_ad_charge" + }, + { + "columns": [], + "domain": "billing", + "layer": "STG", + "name": "stg_ad_refund" + }, + { + "columns": [], + "domain": "billing", + "layer": "STG", + "name": "stg_budget_snapshot" + }, + { + "columns": [], + "domain": "risk", + "layer": "STG", + "name": "stg_content_audit" + }, + { + "columns": [], + "domain": "risk", + "layer": "STG", + "name": "stg_ad_audit" + }, + { + "columns": [], + "domain": "risk", + "layer": "STG", + "name": "stg_invalid_traffic_signal" + }, + { + "columns": [], + "domain": "recommendation", + "layer": "DWD", + "name": "dwd_recommend_request_fact" + }, + { + "columns": [], + "domain": "traffic", + "layer": "DWD", + "name": "dwd_video_exposure_fact" + }, + { + "columns": [], + "domain": "traffic", + "layer": "DWD", + "name": "dwd_video_play_fact" + }, + { + "columns": [], + "domain": "content", + "layer": "DWD", + "name": "dwd_user_interaction_fact" + }, + { + "columns": [], + "domain": "traffic", + "layer": "DWD", + "name": "dwd_search_fact" + }, + { + "columns": [], + "domain": "user", + "layer": "DWD", + "name": "dwd_session_fact" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "DWD", + "name": "dwd_ad_opportunity_fact" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "DWD", + "name": "dwd_ad_candidate_fact" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "DWD", + "name": "dwd_ad_auction_fact" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "DWD", + "name": "dwd_ad_delivery_fact" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "DWD", + "name": "dwd_ad_click_fact" + }, + { + "columns": [], + "domain": "attribution", + "layer": "DWD", + "name": "dwd_ad_conversion_fact" + }, + { + "columns": [], + "domain": "attribution", + "layer": "DWD", + "name": "dwd_attribution_touch_fact" + }, + { + "columns": [], + "domain": "attribution", + "layer": "DWD", + "name": "dwd_ad_attribution_wide" + }, + { + "columns": [], + "domain": "billing", + "layer": "DWD", + "name": "dwd_billing_fact" + }, + { + "columns": [], + "domain": "billing", + "layer": "DWD", + "name": "dwd_refund_fact" + }, + { + "columns": [], + "domain": "billing", + "layer": "DWD", + "name": "dwd_budget_fact" + }, + { + "columns": [], + "domain": "risk", + "layer": "DWD", + "name": "dwd_risk_event_fact" + }, + { + "columns": [], + "domain": "traffic", + "layer": "DWS", + "name": "dws_video_traffic_daily" + }, + { + "columns": [], + "domain": "content", + "layer": "DWS", + "name": "dws_creator_traffic_daily" + }, + { + "columns": [], + "domain": "user", + "layer": "DWS", + "name": "dws_user_engagement_daily" + }, + { + "columns": [], + "domain": "traffic", + "layer": "DWS", + "name": "dws_channel_traffic_daily" + }, + { + "columns": [], + "domain": "content", + "layer": "DWS", + "name": "dws_category_traffic_daily" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "DWS", + "name": "dws_ad_slot_funnel_daily" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "DWS", + "name": "dws_campaign_delivery_daily" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "DWS", + "name": "dws_creative_performance_daily" + }, + { + "columns": [], + "domain": "billing", + "layer": "DWS", + "name": "dws_advertiser_performance_daily" + }, + { + "columns": [], + "domain": "experiment", + "layer": "DWS", + "name": "dws_audience_performance_daily" + }, + { + "columns": [], + "domain": "experiment", + "layer": "DWS", + "name": "dws_experiment_daily" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "DWS", + "name": "dws_monetization_funnel_daily" + }, + { + "columns": [], + "domain": "billing", + "layer": "DWS", + "name": "dws_revenue_daily" + }, + { + "columns": [], + "domain": "attribution", + "layer": "DWS", + "name": "dws_roi_daily" + }, + { + "columns": [], + "domain": "billing", + "layer": "DWS", + "name": "dws_budget_pacing_daily" + }, + { + "columns": [], + "domain": "risk", + "layer": "DWS", + "name": "dws_risk_daily" + }, + { + "columns": [], + "domain": "traffic", + "layer": "ADS", + "name": "ads_traffic_overview" + }, + { + "columns": [], + "domain": "content", + "layer": "ADS", + "name": "ads_creator_dashboard" + }, + { + "columns": [], + "domain": "content", + "layer": "ADS", + "name": "ads_content_efficiency_report" + }, + { + "columns": [], + "domain": "ad_inventory", + "layer": "ADS", + "name": "ads_ad_operation_dashboard" + }, + { + "columns": [], + "domain": "billing", + "layer": "ADS", + "name": "ads_advertiser_report" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "ADS", + "name": "ads_campaign_monitor" + }, + { + "columns": [], + "domain": "ad_delivery", + "layer": "ADS", + "name": "ads_creative_report" + }, + { + "columns": [], + "domain": "billing", + "layer": "ADS", + "name": "ads_revenue_dashboard" + }, + { + "columns": [], + "domain": "experiment", + "layer": "ADS", + "name": "ads_experiment_report" + }, + { + "columns": [], + "domain": "risk", + "layer": "ADS", + "name": "ads_risk_dashboard" + } + ], + "task_count": 62, + "tasks": [ + { + "dependencies": [ + "ods_recommend_request" + ], + "description": "标准化 ods_recommend_request 并去除完全重复记录", + "domain": "recommendation", + "file": "001_stg_recommend_request.sql", + "layer": "STG", + "order": 1, + "target": "stg_recommend_request" + }, + { + "dependencies": [ + "ods_video_exposure" + ], + "description": "标准化 ods_video_exposure 并去除完全重复记录", + "domain": "traffic", + "file": "002_stg_video_exposure.sql", + "layer": "STG", + "order": 2, + "target": "stg_video_exposure" + }, + { + "dependencies": [ + "ods_video_play" + ], + "description": "标准化 ods_video_play 并去除完全重复记录", + "domain": "traffic", + "file": "003_stg_video_play.sql", + "layer": "STG", + "order": 3, + "target": "stg_video_play" + }, + { + "dependencies": [ + "ods_user_interaction" + ], + "description": "标准化 ods_user_interaction 并去除完全重复记录", + "domain": "content", + "file": "004_stg_user_interaction.sql", + "layer": "STG", + "order": 4, + "target": "stg_user_interaction" + }, + { + "dependencies": [ + "ods_search_event" + ], + "description": "标准化 ods_search_event 并去除完全重复记录", + "domain": "traffic", + "file": "005_stg_search_event.sql", + "layer": "STG", + "order": 5, + "target": "stg_search_event" + }, + { + "dependencies": [ + "ods_user_session" + ], + "description": "标准化 ods_user_session 并去除完全重复记录", + "domain": "user", + "file": "006_stg_user_session.sql", + "layer": "STG", + "order": 6, + "target": "stg_user_session" + }, + { + "dependencies": [ + "ods_ad_request" + ], + "description": "标准化 ods_ad_request 并去除完全重复记录", + "domain": "ad_inventory", + "file": "007_stg_ad_request.sql", + "layer": "STG", + "order": 7, + "target": "stg_ad_request" + }, + { + "dependencies": [ + "ods_ad_candidate" + ], + "description": "标准化 ods_ad_candidate 并去除完全重复记录", + "domain": "ad_inventory", + "file": "008_stg_ad_candidate.sql", + "layer": "STG", + "order": 8, + "target": "stg_ad_candidate" + }, + { + "dependencies": [ + "ods_ad_bid" + ], + "description": "标准化 ods_ad_bid 并去除完全重复记录", + "domain": "ad_delivery", + "file": "009_stg_ad_bid.sql", + "layer": "STG", + "order": 9, + "target": "stg_ad_bid" + }, + { + "dependencies": [ + "ods_ad_impression" + ], + "description": "标准化 ods_ad_impression 并去除完全重复记录", + "domain": "ad_delivery", + "file": "010_stg_ad_impression.sql", + "layer": "STG", + "order": 10, + "target": "stg_ad_impression" + }, + { + "dependencies": [ + "ods_ad_click" + ], + "description": "标准化 ods_ad_click 并去除完全重复记录", + "domain": "ad_delivery", + "file": "011_stg_ad_click.sql", + "layer": "STG", + "order": 11, + "target": "stg_ad_click" + }, + { + "dependencies": [ + "ods_ad_conversion" + ], + "description": "标准化 ods_ad_conversion 并去除完全重复记录", + "domain": "attribution", + "file": "012_stg_ad_conversion.sql", + "layer": "STG", + "order": 12, + "target": "stg_ad_conversion" + }, + { + "dependencies": [ + "ods_ad_charge" + ], + "description": "标准化 ods_ad_charge 并去除完全重复记录", + "domain": "billing", + "file": "013_stg_ad_charge.sql", + "layer": "STG", + "order": 13, + "target": "stg_ad_charge" + }, + { + "dependencies": [ + "ods_ad_refund" + ], + "description": "标准化 ods_ad_refund 并去除完全重复记录", + "domain": "billing", + "file": "014_stg_ad_refund.sql", + "layer": "STG", + "order": 14, + "target": "stg_ad_refund" + }, + { + "dependencies": [ + "ods_budget_snapshot" + ], + "description": "标准化 ods_budget_snapshot 并去除完全重复记录", + "domain": "billing", + "file": "015_stg_budget_snapshot.sql", + "layer": "STG", + "order": 15, + "target": "stg_budget_snapshot" + }, + { + "dependencies": [ + "ods_content_audit" + ], + "description": "标准化 ods_content_audit 并去除完全重复记录", + "domain": "risk", + "file": "016_stg_content_audit.sql", + "layer": "STG", + "order": 16, + "target": "stg_content_audit" + }, + { + "dependencies": [ + "ods_ad_audit" + ], + "description": "标准化 ods_ad_audit 并去除完全重复记录", + "domain": "risk", + "file": "017_stg_ad_audit.sql", + "layer": "STG", + "order": 17, + "target": "stg_ad_audit" + }, + { + "dependencies": [ + "ods_invalid_traffic_signal" + ], + "description": "标准化 ods_invalid_traffic_signal 并去除完全重复记录", + "domain": "risk", + "file": "018_stg_invalid_traffic_signal.sql", + "layer": "STG", + "order": 18, + "target": "stg_invalid_traffic_signal" + }, + { + "dependencies": [ + "stg_recommend_request", + "dim_user", + "dim_device", + "dim_geo", + "dim_channel", + "dim_experiment" + ], + "description": "推荐请求明细宽表", + "domain": "recommendation", + "file": "019_dwd_recommend_request_fact.sql", + "layer": "DWD", + "order": 19, + "target": "dwd_recommend_request_fact" + }, + { + "dependencies": [ + "stg_video_exposure", + "dim_video", + "dim_creator", + "dim_content_category" + ], + "description": "视频曝光明细宽表", + "domain": "traffic", + "file": "020_dwd_video_exposure_fact.sql", + "layer": "DWD", + "order": 20, + "target": "dwd_video_exposure_fact" + }, + { + "dependencies": [ + "stg_video_play", + "dim_video" + ], + "description": "视频播放及有效播放事实", + "domain": "traffic", + "file": "021_dwd_video_play_fact.sql", + "layer": "DWD", + "order": 21, + "target": "dwd_video_play_fact" + }, + { + "dependencies": [ + "stg_user_interaction", + "dim_video" + ], + "description": "用户互动明细", + "domain": "content", + "file": "022_dwd_user_interaction_fact.sql", + "layer": "DWD", + "order": 22, + "target": "dwd_user_interaction_fact" + }, + { + "dependencies": [ + "stg_search_event" + ], + "description": "搜索行为事实", + "domain": "traffic", + "file": "023_dwd_search_fact.sql", + "layer": "DWD", + "order": 23, + "target": "dwd_search_fact" + }, + { + "dependencies": [ + "stg_user_session", + "dim_device", + "dim_channel" + ], + "description": "用户会话事实", + "domain": "user", + "file": "024_dwd_session_fact.sql", + "layer": "DWD", + "order": 24, + "target": "dwd_session_fact" + }, + { + "dependencies": [ + "stg_ad_request", + "dim_ad_slot", + "dim_audience" + ], + "description": "广告机会明细", + "domain": "ad_inventory", + "file": "025_dwd_ad_opportunity_fact.sql", + "layer": "DWD", + "order": 25, + "target": "dwd_ad_opportunity_fact" + }, + { + "dependencies": [ + "stg_ad_candidate", + "dim_ad_creative", + "dim_ad_group" + ], + "description": "广告候选召回明细", + "domain": "ad_inventory", + "file": "026_dwd_ad_candidate_fact.sql", + "layer": "DWD", + "order": 26, + "target": "dwd_ad_candidate_fact" + }, + { + "dependencies": [ + "stg_ad_bid", + "dwd_ad_candidate_fact" + ], + "description": "竞价与胜出事实", + "domain": "ad_delivery", + "file": "027_dwd_ad_auction_fact.sql", + "layer": "DWD", + "order": 27, + "target": "dwd_ad_auction_fact" + }, + { + "dependencies": [ + "stg_ad_impression", + "dim_campaign", + "dim_ad_creative", + "dim_ad_slot" + ], + "description": "广告曝光投放事实", + "domain": "ad_delivery", + "file": "028_dwd_ad_delivery_fact.sql", + "layer": "DWD", + "order": 28, + "target": "dwd_ad_delivery_fact" + }, + { + "dependencies": [ + "stg_ad_click", + "dwd_ad_delivery_fact" + ], + "description": "广告点击事实", + "domain": "ad_delivery", + "file": "029_dwd_ad_click_fact.sql", + "layer": "DWD", + "order": 29, + "target": "dwd_ad_click_fact" + }, + { + "dependencies": [ + "stg_ad_conversion", + "dwd_ad_click_fact" + ], + "description": "广告转化事实", + "domain": "attribution", + "file": "030_dwd_ad_conversion_fact.sql", + "layer": "DWD", + "order": 30, + "target": "dwd_ad_conversion_fact" + }, + { + "dependencies": [ + "dwd_ad_click_fact", + "dwd_ad_conversion_fact" + ], + "description": "点击到转化的归因触点", + "domain": "attribution", + "file": "031_dwd_attribution_touch_fact.sql", + "layer": "DWD", + "order": 31, + "target": "dwd_attribution_touch_fact" + }, + { + "dependencies": [ + "dwd_ad_delivery_fact", + "dwd_ad_click_fact", + "dwd_ad_conversion_fact" + ], + "description": "曝光点击转化归因宽表", + "domain": "attribution", + "file": "032_dwd_ad_attribution_wide.sql", + "layer": "DWD", + "order": 32, + "target": "dwd_ad_attribution_wide" + }, + { + "dependencies": [ + "stg_ad_charge", + "dim_advertiser", + "dim_campaign" + ], + "description": "广告扣费事实", + "domain": "billing", + "file": "033_dwd_billing_fact.sql", + "layer": "DWD", + "order": 33, + "target": "dwd_billing_fact" + }, + { + "dependencies": [ + "stg_ad_refund", + "dwd_billing_fact" + ], + "description": "广告退款事实", + "domain": "billing", + "file": "034_dwd_refund_fact.sql", + "layer": "DWD", + "order": 34, + "target": "dwd_refund_fact" + }, + { + "dependencies": [ + "stg_budget_snapshot", + "dim_campaign" + ], + "description": "预算与消耗节奏事实", + "domain": "billing", + "file": "035_dwd_budget_fact.sql", + "layer": "DWD", + "order": 35, + "target": "dwd_budget_fact" + }, + { + "dependencies": [ + "stg_content_audit", + "stg_ad_audit", + "stg_invalid_traffic_signal" + ], + "description": "内容、广告与无效流量统一风险事实", + "domain": "risk", + "file": "036_dwd_risk_event_fact.sql", + "layer": "DWD", + "order": 36, + "target": "dwd_risk_event_fact" + }, + { + "dependencies": [ + "dwd_video_exposure_fact", + "dwd_video_play_fact", + "dwd_user_interaction_fact" + ], + "description": "视频流量效率日报", + "domain": "traffic", + "file": "037_dws_video_traffic_daily.sql", + "layer": "DWS", + "order": 37, + "target": "dws_video_traffic_daily" + }, + { + "dependencies": [ + "dws_video_traffic_daily", + "dim_video" + ], + "description": "作者流量日报", + "domain": "content", + "file": "038_dws_creator_traffic_daily.sql", + "layer": "DWS", + "order": 38, + "target": "dws_creator_traffic_daily" + }, + { + "dependencies": [ + "dwd_video_play_fact", + "dwd_user_interaction_fact" + ], + "description": "用户参与度日报", + "domain": "user", + "file": "039_dws_user_engagement_daily.sql", + "layer": "DWS", + "order": 39, + "target": "dws_user_engagement_daily" + }, + { + "dependencies": [ + "dwd_recommend_request_fact", + "dwd_video_exposure_fact" + ], + "description": "渠道流量日报", + "domain": "traffic", + "file": "040_dws_channel_traffic_daily.sql", + "layer": "DWS", + "order": 40, + "target": "dws_channel_traffic_daily" + }, + { + "dependencies": [ + "dwd_video_exposure_fact", + "dwd_video_play_fact" + ], + "description": "内容分类流量日报", + "domain": "content", + "file": "041_dws_category_traffic_daily.sql", + "layer": "DWS", + "order": 41, + "target": "dws_category_traffic_daily" + }, + { + "dependencies": [ + "dwd_ad_opportunity_fact", + "dwd_ad_delivery_fact", + "dwd_ad_click_fact", + "dwd_ad_conversion_fact" + ], + "description": "广告位请求到转化漏斗", + "domain": "ad_inventory", + "file": "042_dws_ad_slot_funnel_daily.sql", + "layer": "DWS", + "order": 42, + "target": "dws_ad_slot_funnel_daily" + }, + { + "dependencies": [ + "dwd_ad_attribution_wide" + ], + "description": "计划投放效果日报", + "domain": "ad_delivery", + "file": "043_dws_campaign_delivery_daily.sql", + "layer": "DWS", + "order": 43, + "target": "dws_campaign_delivery_daily" + }, + { + "dependencies": [ + "dwd_ad_attribution_wide", + "dim_ad_creative" + ], + "description": "素材效果日报(故意保留百分数 CTR 作为治理场景)", + "domain": "ad_delivery", + "file": "044_dws_creative_performance_daily.sql", + "layer": "DWS", + "order": 44, + "target": "dws_creative_performance_daily" + }, + { + "dependencies": [ + "dws_campaign_delivery_daily" + ], + "description": "广告主效果日报", + "domain": "billing", + "file": "045_dws_advertiser_performance_daily.sql", + "layer": "DWS", + "order": 45, + "target": "dws_advertiser_performance_daily" + }, + { + "dependencies": [ + "dwd_ad_attribution_wide", + "dim_user" + ], + "description": "受众效果日报", + "domain": "experiment", + "file": "046_dws_audience_performance_daily.sql", + "layer": "DWS", + "order": 46, + "target": "dws_audience_performance_daily" + }, + { + "dependencies": [ + "dwd_recommend_request_fact", + "dwd_video_exposure_fact", + "dwd_ad_opportunity_fact" + ], + "description": "实验流量与广告机会日报", + "domain": "experiment", + "file": "047_dws_experiment_daily.sql", + "layer": "DWS", + "order": 47, + "target": "dws_experiment_daily" + }, + { + "dependencies": [ + "dwd_ad_opportunity_fact", + "dwd_ad_candidate_fact", + "dwd_ad_auction_fact", + "dwd_ad_delivery_fact", + "dwd_ad_click_fact", + "dwd_ad_conversion_fact" + ], + "description": "商业化全漏斗日报", + "domain": "ad_inventory", + "file": "048_dws_monetization_funnel_daily.sql", + "layer": "DWS", + "order": 48, + "target": "dws_monetization_funnel_daily" + }, + { + "dependencies": [ + "dwd_billing_fact", + "dwd_refund_fact" + ], + "description": "平台广告收入日报", + "domain": "billing", + "file": "049_dws_revenue_daily.sql", + "layer": "DWS", + "order": 49, + "target": "dws_revenue_daily" + }, + { + "dependencies": [ + "dws_campaign_delivery_daily" + ], + "description": "计划 ROI 日报", + "domain": "attribution", + "file": "050_dws_roi_daily.sql", + "layer": "DWS", + "order": 50, + "target": "dws_roi_daily" + }, + { + "dependencies": [ + "dwd_budget_fact", + "dws_campaign_delivery_daily" + ], + "description": "预算节奏日报", + "domain": "billing", + "file": "051_dws_budget_pacing_daily.sql", + "layer": "DWS", + "order": 51, + "target": "dws_budget_pacing_daily" + }, + { + "dependencies": [ + "dwd_risk_event_fact" + ], + "description": "内容、广告与流量风险日报", + "domain": "risk", + "file": "052_dws_risk_daily.sql", + "layer": "DWS", + "order": 52, + "target": "dws_risk_daily" + }, + { + "dependencies": [ + "dws_channel_traffic_daily", + "dws_video_traffic_daily" + ], + "description": "平台流量总览", + "domain": "traffic", + "file": "053_ads_traffic_overview.sql", + "layer": "ADS", + "order": 53, + "target": "ads_traffic_overview" + }, + { + "dependencies": [ + "dws_creator_traffic_daily", + "dim_creator" + ], + "description": "作者流量与互动看板", + "domain": "content", + "file": "054_ads_creator_dashboard.sql", + "layer": "ADS", + "order": 54, + "target": "ads_creator_dashboard" + }, + { + "dependencies": [ + "dws_category_traffic_daily" + ], + "description": "内容分类效率报告", + "domain": "content", + "file": "055_ads_content_efficiency_report.sql", + "layer": "ADS", + "order": 55, + "target": "ads_content_efficiency_report" + }, + { + "dependencies": [ + "dws_monetization_funnel_daily", + "dws_revenue_daily" + ], + "description": "广告运营总览", + "domain": "ad_inventory", + "file": "056_ads_ad_operation_dashboard.sql", + "layer": "ADS", + "order": 56, + "target": "ads_ad_operation_dashboard" + }, + { + "dependencies": [ + "dws_advertiser_performance_daily", + "dws_revenue_daily", + "dim_advertiser" + ], + "description": "广告主经营报告", + "domain": "billing", + "file": "057_ads_advertiser_report.sql", + "layer": "ADS", + "order": 57, + "target": "ads_advertiser_report" + }, + { + "dependencies": [ + "dws_campaign_delivery_daily", + "dws_budget_pacing_daily", + "dim_campaign" + ], + "description": "计划投放与预算监控", + "domain": "ad_delivery", + "file": "058_ads_campaign_monitor.sql", + "layer": "ADS", + "order": 58, + "target": "ads_campaign_monitor" + }, + { + "dependencies": [ + "dws_creative_performance_daily", + "dim_ad_creative" + ], + "description": "素材效果与 CTR 分级报告", + "domain": "ad_delivery", + "file": "059_ads_creative_report.sql", + "layer": "ADS", + "order": 59, + "target": "ads_creative_report" + }, + { + "dependencies": [ + "dws_revenue_daily", + "dws_roi_daily" + ], + "description": "商业收入与 ROI 驾驶舱", + "domain": "billing", + "file": "060_ads_revenue_dashboard.sql", + "layer": "ADS", + "order": 60, + "target": "ads_revenue_dashboard" + }, + { + "dependencies": [ + "dws_experiment_daily" + ], + "description": "实验流量与商业化效果报告", + "domain": "experiment", + "file": "061_ads_experiment_report.sql", + "layer": "ADS", + "order": 61, + "target": "ads_experiment_report" + }, + { + "dependencies": [ + "dws_risk_daily" + ], + "description": "商业化质量与风险驾驶舱", + "domain": "risk", + "file": "062_ads_risk_dashboard.sql", + "layer": "ADS", + "order": 62, + "target": "ads_risk_dashboard" + } + ] +} diff --git a/examples/video_commercial_warehouse/report.py b/examples/video_commercial_warehouse/report.py new file mode 100644 index 0000000..a16973d --- /dev/null +++ b/examples/video_commercial_warehouse/report.py @@ -0,0 +1,32 @@ +"""视频商业化数仓治理工作台报告生成器。""" +from __future__ import annotations + +import html +import json +from pathlib import Path +from typing import Any + +from sqlgraph.serve.theme import load_explorer_css + + +TEMPLATE_PATH = Path(__file__).resolve().parent / "templates" / "workbench.html" + + +def _safe_json(payload: dict[str, Any]) -> str: + return json.dumps(payload, ensure_ascii=False, separators=(",", ":")).replace( + " Path: + """将治理结果渲染为无外部依赖的单文件工作台。""" + output_path = Path(output_path).resolve() + output_path.parent.mkdir(parents=True, exist_ok=True) + template = TEMPLATE_PATH.read_text(encoding="utf-8") + document = template.replace( + "__TITLE__", html.escape(payload["scenario"]["title"]) + ).replace( + "__EXPLORER_CSS__", load_explorer_css() + ).replace("__PAYLOAD__", _safe_json(payload)) + output_path.write_text(document, encoding="utf-8") + return output_path diff --git a/examples/video_commercial_warehouse/runtime.py b/examples/video_commercial_warehouse/runtime.py new file mode 100644 index 0000000..d2bfdd7 --- /dev/null +++ b/examples/video_commercial_warehouse/runtime.py @@ -0,0 +1,173 @@ +"""DuckDB 数仓运行时:种子生成、全量执行与增量任务执行。""" +from __future__ import annotations + +import time +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Iterable + +import duckdb + +from examples.video_commercial_warehouse.catalog import ( + ALL_TABLES, + TASKS, + TaskSpec, + validate_catalog, +) +from examples.video_commercial_warehouse.generate_sql import ROOT, generate_sql +from examples.video_commercial_warehouse.seed import seed_base_tables + + +class WarehouseExecutionError(RuntimeError): + """某个数仓任务执行失败。""" + + +@dataclass(frozen=True) +class TaskRun: + order: int + target: str + status: str + row_count: int + elapsed_ms: float + error: str = "" + + def to_dict(self) -> dict: + return asdict(self) + + +@dataclass(frozen=True) +class RunResult: + database_path: str + profile: str + task_runs: tuple[TaskRun, ...] + table_rows: dict[str, int] + + @property + def success_count(self) -> int: + return sum(run.status == "success" for run in self.task_runs) + + @property + def failed(self) -> list[TaskRun]: + return [run for run in self.task_runs if run.status != "success"] + + @property + def elapsed_ms(self) -> float: + return round(sum(run.elapsed_ms for run in self.task_runs), 3) + + def to_dict(self) -> dict: + return { + "database_path": self.database_path, + "profile": self.profile, + "success_count": self.success_count, + "failed": [run.to_dict() for run in self.failed], + "elapsed_ms": self.elapsed_ms, + "task_runs": [run.to_dict() for run in self.task_runs], + "table_rows": self.table_rows, + } + + +def _sql_path(sql_dir: Path, task: TaskSpec) -> Path: + return sql_dir / f"{task.order:03d}_{task.target}.sql" + + +def execute_tasks( + con: duckdb.DuckDBPyConnection, + tasks: Iterable[TaskSpec], + sql_dir: Path, +) -> tuple[TaskRun, ...]: + """按给定顺序执行任务;首个失败立即中断并携带任务上下文。""" + records: list[TaskRun] = [] + for task in tasks: + path = _sql_path(sql_dir, task) + if not path.is_file(): + raise WarehouseExecutionError(f"SQL 文件不存在: {path}") + sql = path.read_text(encoding="utf-8") + started = time.perf_counter() + try: + con.execute(sql) + row_count = int( + con.execute(f'SELECT COUNT(*) FROM "{task.target}"').fetchone()[0] + ) + except Exception as exc: + elapsed = round((time.perf_counter() - started) * 1000, 3) + records.append(TaskRun(task.order, task.target, "failed", 0, elapsed, str(exc))) + raise WarehouseExecutionError( + f"任务 {task.order:03d} {task.target} 执行失败: {exc}" + ) from exc + elapsed = round((time.perf_counter() - started) * 1000, 3) + records.append(TaskRun(task.order, task.target, "success", row_count, elapsed)) + return tuple(records) + + +def collect_table_rows(con: duckdb.DuckDBPyConnection) -> dict[str, int]: + """查询 catalog 中全部 94 张表的行数。""" + rows: dict[str, int] = {} + for table in ALL_TABLES: + exists = con.execute( + "SELECT COUNT(*) FROM information_schema.tables " + "WHERE table_schema='main' AND table_name=?", + [table], + ).fetchone()[0] + if not exists: + raise WarehouseExecutionError(f"目标表未生成: {table}") + rows[table] = int(con.execute(f'SELECT COUNT(*) FROM "{table}"').fetchone()[0]) + return rows + + +def build_warehouse( + database_path: Path, + *, + profile: str = "smoke", + sql_dir: Path | None = None, +) -> RunResult: + """从空 DuckDB 开始生成基础数据并执行完整 62 任务。""" + validate_catalog() + database_path = Path(database_path).resolve() + database_path.parent.mkdir(parents=True, exist_ok=True) + if database_path.exists(): + database_path.unlink() + + if sql_dir is None: + resolved_sql_dir = ROOT / "sql" + # 默认运行不得覆盖用户或治理流程已经修改过的真实 SQL;仅在产物缺失时生成。 + if len(list(resolved_sql_dir.glob("*.sql"))) != 62: + paths = generate_sql(ROOT) + resolved_sql_dir = paths[0].parent + else: + resolved_sql_dir = Path(sql_dir).resolve() + if len(list(resolved_sql_dir.glob("*.sql"))) != 62: + raise WarehouseExecutionError( + f"SQL 目录必须包含 62 个任务文件: {resolved_sql_dir}" + ) + + con = duckdb.connect(str(database_path)) + try: + seed_base_tables(con, profile) + records = execute_tasks(con, TASKS, resolved_sql_dir) + table_rows = collect_table_rows(con) + con.execute("CHECKPOINT") + finally: + con.close() + return RunResult(str(database_path), profile, records, table_rows) + + +def rerun_tasks( + database_path: Path, + task_names: Iterable[str], + *, + sql_dir: Path | None = None, +) -> tuple[TaskRun, ...]: + """在已有数据库上按 catalog 顺序重跑指定任务集合。""" + wanted = set(task_names) + tasks = [task for task in TASKS if task.target in wanted] + missing = wanted - {task.target for task in tasks} + if missing: + raise WarehouseExecutionError(f"增量任务不在 catalog: {sorted(missing)}") + resolved_sql_dir = Path(sql_dir) if sql_dir else ROOT / "sql" + con = duckdb.connect(str(Path(database_path).resolve())) + try: + records = execute_tasks(con, tasks, resolved_sql_dir) + con.execute("CHECKPOINT") + finally: + con.close() + return records diff --git a/examples/video_commercial_warehouse/seed.py b/examples/video_commercial_warehouse/seed.py new file mode 100644 index 0000000..ef28fe8 --- /dev/null +++ b/examples/video_commercial_warehouse/seed.py @@ -0,0 +1,338 @@ +"""DuckDB 确定性基础数据生成器。 + +所有大表都通过 DuckDB `range()` 和模运算集合生成,不依赖随机数,也不在 Python +中逐行插入。相同 profile 的表行数和内容可复算。 +""" +from __future__ import annotations + +from dataclasses import dataclass + +import duckdb + + +@dataclass(frozen=True) +class SeedProfile: + name: str + days: int + users: int + creators: int + videos: int + advertisers: int + campaigns: int + ad_groups: int + ad_creatives: int + recommend_requests: int + + +PROFILES = { + "smoke": SeedProfile("smoke", 3, 200, 30, 100, 8, 20, 40, 80, 5_000), + "demo": SeedProfile("demo", 30, 5_000, 300, 2_000, 50, 200, 500, 1_000, 200_000), +} + + +def _run(con: duckdb.DuckDBPyConnection, sql: str) -> None: + con.execute(sql) + + +def seed_base_tables(con: duckdb.DuckDBPyConnection, profile: str = "smoke") -> None: + """创建 18 张 ODS 和 14 张 DIM 表。""" + if profile not in PROFILES: + raise ValueError(f"未知数据 profile: {profile}; 可选 {sorted(PROFILES)}") + p = PROFILES[profile] + sessions = max(p.recommend_requests // 5, 1) + exposures = p.recommend_requests * 2 + plays = exposures * 4 // 5 + interactions = max(plays // 5, 1) + searches = max(p.recommend_requests // 20, 1) + ad_requests = max(p.recommend_requests // 3, 1) + candidates = ad_requests * 3 + impressions = ad_requests * 9 // 10 + clicks = max(impressions // 20, 1) + conversions = max(clicks // 5, 1) + refunds = max(impressions // 100, 1) + devices = max(p.users // 2, 20) + categories = 12 + geos = 12 + channels = 6 + slots = 6 + experiments = 4 + audiences = 8 + + # --------------------------- DIM ------------------------------------- + _run(con, f""" + CREATE OR REPLACE TABLE dim_user AS + SELECT i + 1 AS user_id, + DATE '2025-01-01' + CAST(i % 365 AS INTEGER) AS register_date, + CASE i % 4 WHEN 0 THEN 'CN' WHEN 1 THEN 'US' WHEN 2 THEN 'BR' ELSE 'ID' END AS country_code, + CASE i % 4 WHEN 0 THEN '18-24' WHEN 1 THEN '25-34' WHEN 2 THEN '35-44' ELSE '45+' END AS age_bucket, + (i % {audiences}) + 1 AS audience_id + FROM range({p.users}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_creator AS + SELECT i + 1 AS creator_id, (i % {p.users}) + 1 AS user_id, + CASE i % 3 WHEN 0 THEN 'head' WHEN 1 THEN 'growth' ELSE 'long_tail' END AS creator_tier + FROM range({p.creators}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_content_category AS + SELECT i + 1 AS category_id, 'category_' || CAST(i + 1 AS VARCHAR) AS category_name + FROM range({categories}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_video AS + SELECT i + 1 AS video_id, (i % {p.creators}) + 1 AS creator_id, + (i % {categories}) + 1 AS category_id, 15 + CAST(i % 286 AS INTEGER) AS duration_sec, + DATE '2026-01-01' + CAST(i % 180 AS INTEGER) AS publish_date, + CASE WHEN i % 29 = 0 THEN 'restricted' ELSE 'active' END AS content_status + FROM range({p.videos}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_device AS + SELECT i + 1 AS device_id, + CASE i % 3 WHEN 0 THEN 'phone' WHEN 1 THEN 'tablet' ELSE 'desktop' END AS device_type, + CASE i % 3 WHEN 0 THEN 'android' WHEN 1 THEN 'ios' ELSE 'web' END AS os + FROM range({devices}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_geo AS + SELECT i + 1 AS geo_id, + CASE i % 4 WHEN 0 THEN 'CN' WHEN 1 THEN 'US' WHEN 2 THEN 'BR' ELSE 'ID' END AS country_code, + 'region_' || CAST(i + 1 AS VARCHAR) AS region + FROM range({geos}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_channel AS + SELECT i + 1 AS channel_id, + CASE i % 6 WHEN 0 THEN 'organic' WHEN 1 THEN 'push' WHEN 2 THEN 'search' + WHEN 3 THEN 'social' WHEN 4 THEN 'partner' ELSE 'direct' END AS channel_name + FROM range({channels}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_advertiser AS + SELECT i + 1 AS advertiser_id, 'advertiser_' || CAST(i + 1 AS VARCHAR) AS advertiser_name, + CASE i % 4 WHEN 0 THEN 'game' WHEN 1 THEN 'retail' + WHEN 2 THEN 'finance' ELSE 'education' END AS industry + FROM range({p.advertisers}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_campaign AS + SELECT i + 1 AS campaign_id, (i % {p.advertisers}) + 1 AS advertiser_id, + 5000.0 + (i % 20) * 500.0 AS budget, + CASE i % 3 WHEN 0 THEN 'conversion' WHEN 1 THEN 'traffic' ELSE 'reach' END AS objective, + 'active' AS status + FROM range({p.campaigns}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_audience AS + SELECT i + 1 AS audience_id, 'audience_' || CAST(i + 1 AS VARCHAR) AS audience_name, + CASE i % 3 WHEN 0 THEN 'interest' WHEN 1 THEN 'lookalike' ELSE 'retarget' END AS strategy + FROM range({audiences}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_ad_group AS + SELECT i + 1 AS ad_group_id, (i % {p.campaigns}) + 1 AS campaign_id, + (i % {audiences}) + 1 AS audience_id, + CASE i % 3 WHEN 0 THEN 'cpm' WHEN 1 THEN 'cpc' ELSE 'ocpc' END AS bid_type, + 0.5 + (i % 20) * 0.1 AS bid_value + FROM range({p.ad_groups}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_ad_creative AS + SELECT i + 1 AS ad_creative_id, (i % {p.ad_groups}) + 1 AS ad_group_id, + (i % {p.videos}) + 1 AS video_id, + CASE i % 3 WHEN 0 THEN 'native_video' WHEN 1 THEN 'feed_card' ELSE 'post_roll' END AS creative_format, + CASE WHEN i % 31 = 0 THEN 'paused' ELSE 'active' END AS status + FROM range({p.ad_creatives}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_ad_slot AS + SELECT i + 1 AS ad_slot_id, + CASE i % 3 WHEN 0 THEN 'feed' WHEN 1 THEN 'detail' ELSE 'post_roll' END AS slot_name, + CASE i % 2 WHEN 0 THEN 'recommend' ELSE 'content' END AS scene, + 0.01 + (i % 6) * 0.005 AS floor_price + FROM range({slots}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE dim_experiment AS + SELECT i + 1 AS experiment_id, 'monetization_exp_' || CAST(i + 1 AS VARCHAR) AS experiment_name, + CASE i % 2 WHEN 0 THEN 'control' ELSE 'treatment' END AS variant + FROM range({experiments}) t(i) + """) + + # --------------------------- ODS traffic ----------------------------- + _run(con, f""" + CREATE OR REPLACE TABLE ods_user_session AS + SELECT i + 1 AS session_id, (i % {p.users}) + 1 AS user_id, + (i % {devices}) + 1 AS device_id, (i % {channels}) + 1 AS channel_id, + TIMESTAMP '2026-08-01 00:00:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS start_time, + TIMESTAMP '2026-08-01 00:05:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS end_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({sessions}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_recommend_request AS + SELECT i + 1 AS request_id, (i % {sessions}) + 1 AS session_id, + (i % {p.users}) + 1 AS user_id, (i % {devices}) + 1 AS device_id, + (i % {geos}) + 1 AS geo_id, (i % {channels}) + 1 AS channel_id, + (i % {experiments}) + 1 AS experiment_id, + TIMESTAMP '2026-08-01 00:00:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS request_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({p.recommend_requests}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_video_exposure AS + SELECT i + 1 AS exposure_id, CAST(FLOOR(i / 2) AS BIGINT) + 1 AS request_id, + (CAST(FLOOR(i / 2) AS BIGINT) % {p.users}) + 1 AS user_id, + (i % {p.videos}) + 1 AS video_id, CAST(i % 20 AS INTEGER) + 1 AS position, + TIMESTAMP '2026-08-01 00:00:01' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS exposure_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({exposures}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_video_play AS + SELECT i + 1 AS play_id, i + 1 AS exposure_id, (i % {p.users}) + 1 AS user_id, + (i % {p.videos}) + 1 AS video_id, 3.0 + (i % 120) AS play_duration_sec, + LEAST(1.0, (3.0 + (i % 120)) / (15.0 + (i % 286))) AS completion_rate, + TIMESTAMP '2026-08-01 00:00:02' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS play_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({plays}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_user_interaction AS + SELECT i + 1 AS interaction_id, (i * 5) + 1 AS play_id, + (i % {p.users}) + 1 AS user_id, (i % {p.videos}) + 1 AS video_id, + CASE i % 3 WHEN 0 THEN 'like' WHEN 1 THEN 'comment' ELSE 'share' END AS interaction_type, + TIMESTAMP '2026-08-01 00:00:05' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS event_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({interactions}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_search_event AS + SELECT i + 1 AS search_id, (i % {sessions}) + 1 AS session_id, + (i % {p.users}) + 1 AS user_id, 'query_' || CAST(i % 50 AS VARCHAR) AS query, + CAST((i * 7) % 100 AS INTEGER) AS result_count, + TIMESTAMP '2026-08-01 00:01:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS event_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({searches}) t(i) + """) + + # --------------------------- ODS ads --------------------------------- + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_request AS + SELECT i + 1 AS ad_request_id, i * 3 + 1 AS request_id, + (i % {p.users}) + 1 AS user_id, (i % {slots}) + 1 AS ad_slot_id, + (i % {audiences}) + 1 AS audience_id, + TIMESTAMP '2026-08-01 00:00:03' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS request_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({ad_requests}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_candidate AS + SELECT i + 1 AS candidate_id, CAST(FLOOR(i / 3) AS BIGINT) + 1 AS ad_request_id, + (i % {p.ad_creatives}) + 1 AS ad_creative_id, + (i % {p.ad_groups}) + 1 AS ad_group_id, + (i % {p.campaigns}) + 1 AS campaign_id, + (i % {p.advertisers}) + 1 AS advertiser_id, + 0.005 + (i % 100) / 1000.0 AS predicted_ctr, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({candidates}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_bid AS + SELECT i + 1 AS bid_id, i + 1 AS candidate_id, + 0.5 + (i % 30) * 0.1 AS bid_price, (i % 3 = 0) AS is_winner, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({candidates}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_impression AS + SELECT i + 1 AS ad_impression_id, i + 1 AS ad_request_id, + (i % {p.users}) + 1 AS user_id, (i % {p.ad_creatives}) + 1 AS ad_creative_id, + (i % {p.campaigns}) + 1 AS campaign_id, + (i % {p.advertisers}) + 1 AS advertiser_id, + (i % {slots}) + 1 AS ad_slot_id, + TIMESTAMP '2026-08-01 00:00:04' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS impression_time, + 0.01 + (i % 25) * 0.002 AS cost, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({impressions}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_click AS + SELECT i + 1 AS ad_click_id, i * 20 + 1 AS ad_impression_id, + ((i * 20) % {p.users}) + 1 AS user_id, + ((i * 20) % {p.ad_creatives}) + 1 AS ad_creative_id, + ((i * 20) % {p.campaigns}) + 1 AS campaign_id, + TIMESTAMP '2026-08-01 00:00:10' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS click_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({clicks}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_conversion AS + SELECT i + 1 AS conversion_id, i * 5 + 1 AS ad_click_id, + ((i * 100) % {p.users}) + 1 AS user_id, + ((i * 100) % {p.ad_creatives}) + 1 AS ad_creative_id, + ((i * 100) % {p.campaigns}) + 1 AS campaign_id, + CASE i % 3 WHEN 0 THEN 'purchase' WHEN 1 THEN 'signup' ELSE 'install' END AS conversion_type, + 10.0 + (i % 80) AS conversion_value, + TIMESTAMP '2026-08-01 00:03:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS conversion_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({conversions}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_charge AS + SELECT i + 1 AS charge_id, i + 1 AS ad_impression_id, + (i % {p.advertisers}) + 1 AS advertiser_id, + (i % {p.campaigns}) + 1 AS campaign_id, + 0.01 + (i % 25) * 0.002 AS amount, + TIMESTAMP '2026-08-01 00:00:05' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS charge_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({impressions}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_refund AS + SELECT i + 1 AS refund_id, i * 100 + 1 AS charge_id, + ((i * 100) % {p.advertisers}) + 1 AS advertiser_id, + 0.01 AS amount, 'invalid_traffic' AS reason, + TIMESTAMP '2026-08-01 01:00:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS refund_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({refunds}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_budget_snapshot AS + SELECT i * {p.days} + d + 1 AS snapshot_id, i + 1 AS campaign_id, + 5000.0 + (i % 20) * 500.0 AS budget, + (d + 1) * (100.0 + (i % 30)) AS spent, + DATE '2026-08-01' + CAST(d AS INTEGER) AS snapshot_date + FROM range({p.campaigns}) c(i), range({p.days}) x(d) + """) + + # --------------------------- ODS governance -------------------------- + _run(con, f""" + CREATE OR REPLACE TABLE ods_content_audit AS + SELECT i + 1 AS audit_id, i + 1 AS video_id, + CASE WHEN i % 29 = 0 THEN 'rejected' ELSE 'approved' END AS audit_status, + CASE WHEN i % 29 = 0 THEN 0.9 ELSE 0.1 END AS risk_score, + TIMESTAMP '2026-08-01 02:00:00' + (i % 86400) * INTERVAL 1 SECOND AS audit_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({p.videos}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_ad_audit AS + SELECT i + 1 AS audit_id, i + 1 AS ad_creative_id, + CASE WHEN i % 31 = 0 THEN 'rejected' ELSE 'approved' END AS audit_status, + CASE WHEN i % 31 = 0 THEN 0.95 ELSE 0.08 END AS risk_score, + TIMESTAMP '2026-08-01 02:30:00' + (i % 86400) * INTERVAL 1 SECOND AS audit_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({p.ad_creatives}) t(i) + """) + _run(con, f""" + CREATE OR REPLACE TABLE ods_invalid_traffic_signal AS + SELECT i + 1 AS signal_id, i * 50 + 1 AS request_id, + ((i * 50) % {p.users}) + 1 AS user_id, + CASE i % 3 WHEN 0 THEN 'rapid_click' WHEN 1 THEN 'device_farm' ELSE 'proxy' END AS signal_type, + 0.7 + (i % 30) / 100.0 AS risk_score, + TIMESTAMP '2026-08-01 03:00:00' + (i % ({p.days} * 86400)) * INTERVAL 1 SECOND AS event_time, + DATE '2026-08-01' + CAST(i % {p.days} AS INTEGER) AS event_date + FROM range({max(p.recommend_requests // 50, 1)}) t(i) + """) diff --git a/examples/video_commercial_warehouse/sql/001_stg_recommend_request.sql b/examples/video_commercial_warehouse/sql/001_stg_recommend_request.sql new file mode 100644 index 0000000..35a3c87 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/001_stg_recommend_request.sql @@ -0,0 +1,3 @@ +-- STG / recommendation: 标准化 ods_recommend_request 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_recommend_request AS +SELECT DISTINCT * FROM ods_recommend_request; diff --git a/examples/video_commercial_warehouse/sql/002_stg_video_exposure.sql b/examples/video_commercial_warehouse/sql/002_stg_video_exposure.sql new file mode 100644 index 0000000..3f84186 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/002_stg_video_exposure.sql @@ -0,0 +1,3 @@ +-- STG / traffic: 标准化 ods_video_exposure 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_video_exposure AS +SELECT DISTINCT * FROM ods_video_exposure; diff --git a/examples/video_commercial_warehouse/sql/003_stg_video_play.sql b/examples/video_commercial_warehouse/sql/003_stg_video_play.sql new file mode 100644 index 0000000..5847907 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/003_stg_video_play.sql @@ -0,0 +1,3 @@ +-- STG / traffic: 标准化 ods_video_play 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_video_play AS +SELECT DISTINCT * FROM ods_video_play; diff --git a/examples/video_commercial_warehouse/sql/004_stg_user_interaction.sql b/examples/video_commercial_warehouse/sql/004_stg_user_interaction.sql new file mode 100644 index 0000000..49e27d2 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/004_stg_user_interaction.sql @@ -0,0 +1,3 @@ +-- STG / content: 标准化 ods_user_interaction 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_user_interaction AS +SELECT DISTINCT * FROM ods_user_interaction; diff --git a/examples/video_commercial_warehouse/sql/005_stg_search_event.sql b/examples/video_commercial_warehouse/sql/005_stg_search_event.sql new file mode 100644 index 0000000..70ff794 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/005_stg_search_event.sql @@ -0,0 +1,3 @@ +-- STG / traffic: 标准化 ods_search_event 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_search_event AS +SELECT DISTINCT * FROM ods_search_event; diff --git a/examples/video_commercial_warehouse/sql/006_stg_user_session.sql b/examples/video_commercial_warehouse/sql/006_stg_user_session.sql new file mode 100644 index 0000000..7e7e7e2 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/006_stg_user_session.sql @@ -0,0 +1,3 @@ +-- STG / user: 标准化 ods_user_session 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_user_session AS +SELECT DISTINCT * FROM ods_user_session; diff --git a/examples/video_commercial_warehouse/sql/007_stg_ad_request.sql b/examples/video_commercial_warehouse/sql/007_stg_ad_request.sql new file mode 100644 index 0000000..23796d3 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/007_stg_ad_request.sql @@ -0,0 +1,3 @@ +-- STG / ad_inventory: 标准化 ods_ad_request 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_request AS +SELECT DISTINCT * FROM ods_ad_request; diff --git a/examples/video_commercial_warehouse/sql/008_stg_ad_candidate.sql b/examples/video_commercial_warehouse/sql/008_stg_ad_candidate.sql new file mode 100644 index 0000000..730f0b7 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/008_stg_ad_candidate.sql @@ -0,0 +1,3 @@ +-- STG / ad_inventory: 标准化 ods_ad_candidate 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_candidate AS +SELECT DISTINCT * FROM ods_ad_candidate; diff --git a/examples/video_commercial_warehouse/sql/009_stg_ad_bid.sql b/examples/video_commercial_warehouse/sql/009_stg_ad_bid.sql new file mode 100644 index 0000000..15bac78 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/009_stg_ad_bid.sql @@ -0,0 +1,3 @@ +-- STG / ad_delivery: 标准化 ods_ad_bid 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_bid AS +SELECT DISTINCT * FROM ods_ad_bid; diff --git a/examples/video_commercial_warehouse/sql/010_stg_ad_impression.sql b/examples/video_commercial_warehouse/sql/010_stg_ad_impression.sql new file mode 100644 index 0000000..a15c925 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/010_stg_ad_impression.sql @@ -0,0 +1,3 @@ +-- STG / ad_delivery: 标准化 ods_ad_impression 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_impression AS +SELECT DISTINCT * FROM ods_ad_impression; diff --git a/examples/video_commercial_warehouse/sql/011_stg_ad_click.sql b/examples/video_commercial_warehouse/sql/011_stg_ad_click.sql new file mode 100644 index 0000000..86516ad --- /dev/null +++ b/examples/video_commercial_warehouse/sql/011_stg_ad_click.sql @@ -0,0 +1,3 @@ +-- STG / ad_delivery: 标准化 ods_ad_click 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_click AS +SELECT DISTINCT * FROM ods_ad_click; diff --git a/examples/video_commercial_warehouse/sql/012_stg_ad_conversion.sql b/examples/video_commercial_warehouse/sql/012_stg_ad_conversion.sql new file mode 100644 index 0000000..254aba4 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/012_stg_ad_conversion.sql @@ -0,0 +1,3 @@ +-- STG / attribution: 标准化 ods_ad_conversion 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_conversion AS +SELECT DISTINCT * FROM ods_ad_conversion; diff --git a/examples/video_commercial_warehouse/sql/013_stg_ad_charge.sql b/examples/video_commercial_warehouse/sql/013_stg_ad_charge.sql new file mode 100644 index 0000000..77ffb9a --- /dev/null +++ b/examples/video_commercial_warehouse/sql/013_stg_ad_charge.sql @@ -0,0 +1,3 @@ +-- STG / billing: 标准化 ods_ad_charge 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_charge AS +SELECT DISTINCT * FROM ods_ad_charge; diff --git a/examples/video_commercial_warehouse/sql/014_stg_ad_refund.sql b/examples/video_commercial_warehouse/sql/014_stg_ad_refund.sql new file mode 100644 index 0000000..052b408 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/014_stg_ad_refund.sql @@ -0,0 +1,3 @@ +-- STG / billing: 标准化 ods_ad_refund 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_refund AS +SELECT DISTINCT * FROM ods_ad_refund; diff --git a/examples/video_commercial_warehouse/sql/015_stg_budget_snapshot.sql b/examples/video_commercial_warehouse/sql/015_stg_budget_snapshot.sql new file mode 100644 index 0000000..0c2d084 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/015_stg_budget_snapshot.sql @@ -0,0 +1,3 @@ +-- STG / billing: 标准化 ods_budget_snapshot 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_budget_snapshot AS +SELECT DISTINCT * FROM ods_budget_snapshot; diff --git a/examples/video_commercial_warehouse/sql/016_stg_content_audit.sql b/examples/video_commercial_warehouse/sql/016_stg_content_audit.sql new file mode 100644 index 0000000..7bcef93 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/016_stg_content_audit.sql @@ -0,0 +1,3 @@ +-- STG / risk: 标准化 ods_content_audit 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_content_audit AS +SELECT DISTINCT * FROM ods_content_audit; diff --git a/examples/video_commercial_warehouse/sql/017_stg_ad_audit.sql b/examples/video_commercial_warehouse/sql/017_stg_ad_audit.sql new file mode 100644 index 0000000..f281702 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/017_stg_ad_audit.sql @@ -0,0 +1,3 @@ +-- STG / risk: 标准化 ods_ad_audit 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_ad_audit AS +SELECT DISTINCT * FROM ods_ad_audit; diff --git a/examples/video_commercial_warehouse/sql/018_stg_invalid_traffic_signal.sql b/examples/video_commercial_warehouse/sql/018_stg_invalid_traffic_signal.sql new file mode 100644 index 0000000..db5040f --- /dev/null +++ b/examples/video_commercial_warehouse/sql/018_stg_invalid_traffic_signal.sql @@ -0,0 +1,3 @@ +-- STG / risk: 标准化 ods_invalid_traffic_signal 并去除完全重复记录 +CREATE OR REPLACE TABLE stg_invalid_traffic_signal AS +SELECT DISTINCT * FROM ods_invalid_traffic_signal; diff --git a/examples/video_commercial_warehouse/sql/019_dwd_recommend_request_fact.sql b/examples/video_commercial_warehouse/sql/019_dwd_recommend_request_fact.sql new file mode 100644 index 0000000..3fd4768 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/019_dwd_recommend_request_fact.sql @@ -0,0 +1,10 @@ +-- DWD / recommendation: 推荐请求明细宽表 +CREATE OR REPLACE TABLE dwd_recommend_request_fact AS +SELECT r.*, u.age_bucket, d.device_type, d.os, g.country_code, g.region, + c.channel_name, e.variant AS experiment_variant + FROM stg_recommend_request r + LEFT JOIN dim_user u ON r.user_id = u.user_id + LEFT JOIN dim_device d ON r.device_id = d.device_id + LEFT JOIN dim_geo g ON r.geo_id = g.geo_id + LEFT JOIN dim_channel c ON r.channel_id = c.channel_id + LEFT JOIN dim_experiment e ON r.experiment_id = e.experiment_id; diff --git a/examples/video_commercial_warehouse/sql/020_dwd_video_exposure_fact.sql b/examples/video_commercial_warehouse/sql/020_dwd_video_exposure_fact.sql new file mode 100644 index 0000000..fd58c71 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/020_dwd_video_exposure_fact.sql @@ -0,0 +1,8 @@ +-- DWD / traffic: 视频曝光明细宽表 +CREATE OR REPLACE TABLE dwd_video_exposure_fact AS +SELECT x.*, v.creator_id, v.category_id, v.duration_sec, + cr.creator_tier, cc.category_name + FROM stg_video_exposure x + LEFT JOIN dim_video v ON x.video_id = v.video_id + LEFT JOIN dim_creator cr ON v.creator_id = cr.creator_id + LEFT JOIN dim_content_category cc ON v.category_id = cc.category_id; diff --git a/examples/video_commercial_warehouse/sql/021_dwd_video_play_fact.sql b/examples/video_commercial_warehouse/sql/021_dwd_video_play_fact.sql new file mode 100644 index 0000000..8d6c162 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/021_dwd_video_play_fact.sql @@ -0,0 +1,7 @@ +-- DWD / traffic: 视频播放及有效播放事实 +CREATE OR REPLACE TABLE dwd_video_play_fact AS +SELECT p.*, v.creator_id, v.category_id, v.duration_sec, + CASE WHEN p.completion_rate >= 0.9 THEN 1 ELSE 0 END AS is_complete_play, + CASE WHEN p.play_duration_sec >= 5 THEN 1 ELSE 0 END AS is_valid_play + FROM stg_video_play p + LEFT JOIN dim_video v ON p.video_id = v.video_id; diff --git a/examples/video_commercial_warehouse/sql/022_dwd_user_interaction_fact.sql b/examples/video_commercial_warehouse/sql/022_dwd_user_interaction_fact.sql new file mode 100644 index 0000000..625c949 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/022_dwd_user_interaction_fact.sql @@ -0,0 +1,8 @@ +-- DWD / content: 用户互动明细 +CREATE OR REPLACE TABLE dwd_user_interaction_fact AS +SELECT i.*, v.creator_id, v.category_id, + CASE WHEN i.interaction_type = 'like' THEN 1 ELSE 0 END AS is_like, + CASE WHEN i.interaction_type = 'comment' THEN 1 ELSE 0 END AS is_comment, + CASE WHEN i.interaction_type = 'share' THEN 1 ELSE 0 END AS is_share + FROM stg_user_interaction i + LEFT JOIN dim_video v ON i.video_id = v.video_id; diff --git a/examples/video_commercial_warehouse/sql/023_dwd_search_fact.sql b/examples/video_commercial_warehouse/sql/023_dwd_search_fact.sql new file mode 100644 index 0000000..d65a019 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/023_dwd_search_fact.sql @@ -0,0 +1,5 @@ +-- DWD / traffic: 搜索行为事实 +CREATE OR REPLACE TABLE dwd_search_fact AS +SELECT *, LENGTH(query) AS query_length, + CASE WHEN result_count > 0 THEN 1 ELSE 0 END AS has_result + FROM stg_search_event; diff --git a/examples/video_commercial_warehouse/sql/024_dwd_session_fact.sql b/examples/video_commercial_warehouse/sql/024_dwd_session_fact.sql new file mode 100644 index 0000000..dd83a88 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/024_dwd_session_fact.sql @@ -0,0 +1,7 @@ +-- DWD / user: 用户会话事实 +CREATE OR REPLACE TABLE dwd_session_fact AS +SELECT s.*, d.device_type, d.os, c.channel_name, + DATE_DIFF('second', s.start_time, s.end_time) AS session_duration_sec + FROM stg_user_session s + LEFT JOIN dim_device d ON s.device_id = d.device_id + LEFT JOIN dim_channel c ON s.channel_id = c.channel_id; diff --git a/examples/video_commercial_warehouse/sql/025_dwd_ad_opportunity_fact.sql b/examples/video_commercial_warehouse/sql/025_dwd_ad_opportunity_fact.sql new file mode 100644 index 0000000..9b880e5 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/025_dwd_ad_opportunity_fact.sql @@ -0,0 +1,6 @@ +-- DWD / ad_inventory: 广告机会明细 +CREATE OR REPLACE TABLE dwd_ad_opportunity_fact AS +SELECT r.*, s.slot_name, s.scene, s.floor_price, a.audience_name + FROM stg_ad_request r + LEFT JOIN dim_ad_slot s ON r.ad_slot_id = s.ad_slot_id + LEFT JOIN dim_audience a ON r.audience_id = a.audience_id; diff --git a/examples/video_commercial_warehouse/sql/026_dwd_ad_candidate_fact.sql b/examples/video_commercial_warehouse/sql/026_dwd_ad_candidate_fact.sql new file mode 100644 index 0000000..767f312 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/026_dwd_ad_candidate_fact.sql @@ -0,0 +1,6 @@ +-- DWD / ad_inventory: 广告候选召回明细 +CREATE OR REPLACE TABLE dwd_ad_candidate_fact AS +SELECT c.*, cr.video_id, cr.creative_format, g.bid_type, g.bid_value + FROM stg_ad_candidate c + LEFT JOIN dim_ad_creative cr ON c.ad_creative_id = cr.ad_creative_id + LEFT JOIN dim_ad_group g ON c.ad_group_id = g.ad_group_id; diff --git a/examples/video_commercial_warehouse/sql/027_dwd_ad_auction_fact.sql b/examples/video_commercial_warehouse/sql/027_dwd_ad_auction_fact.sql new file mode 100644 index 0000000..96862ba --- /dev/null +++ b/examples/video_commercial_warehouse/sql/027_dwd_ad_auction_fact.sql @@ -0,0 +1,7 @@ +-- DWD / ad_delivery: 竞价与胜出事实 +CREATE OR REPLACE TABLE dwd_ad_auction_fact AS +SELECT b.*, c.ad_request_id, c.ad_creative_id, c.ad_group_id, + c.campaign_id, c.advertiser_id, c.predicted_ctr, + b.bid_price * c.predicted_ctr AS rank_score + FROM stg_ad_bid b + JOIN dwd_ad_candidate_fact c ON b.candidate_id = c.candidate_id; diff --git a/examples/video_commercial_warehouse/sql/028_dwd_ad_delivery_fact.sql b/examples/video_commercial_warehouse/sql/028_dwd_ad_delivery_fact.sql new file mode 100644 index 0000000..b44118a --- /dev/null +++ b/examples/video_commercial_warehouse/sql/028_dwd_ad_delivery_fact.sql @@ -0,0 +1,8 @@ +-- DWD / ad_delivery: 广告曝光投放事实 +CREATE OR REPLACE TABLE dwd_ad_delivery_fact AS +SELECT i.*, c.objective, cr.ad_group_id, cr.video_id, cr.creative_format, + s.scene, s.floor_price + FROM stg_ad_impression i + LEFT JOIN dim_campaign c ON i.campaign_id = c.campaign_id + LEFT JOIN dim_ad_creative cr ON i.ad_creative_id = cr.ad_creative_id + LEFT JOIN dim_ad_slot s ON i.ad_slot_id = s.ad_slot_id; diff --git a/examples/video_commercial_warehouse/sql/029_dwd_ad_click_fact.sql b/examples/video_commercial_warehouse/sql/029_dwd_ad_click_fact.sql new file mode 100644 index 0000000..12a526d --- /dev/null +++ b/examples/video_commercial_warehouse/sql/029_dwd_ad_click_fact.sql @@ -0,0 +1,7 @@ +-- DWD / ad_delivery: 广告点击事实 +CREATE OR REPLACE TABLE dwd_ad_click_fact AS +SELECT c.*, i.ad_request_id, i.advertiser_id, i.ad_slot_id, + i.cost, i.impression_time, + DATE_DIFF('second', i.impression_time, c.click_time) AS click_delay_sec + FROM stg_ad_click c + JOIN dwd_ad_delivery_fact i ON c.ad_impression_id = i.ad_impression_id; diff --git a/examples/video_commercial_warehouse/sql/030_dwd_ad_conversion_fact.sql b/examples/video_commercial_warehouse/sql/030_dwd_ad_conversion_fact.sql new file mode 100644 index 0000000..4dabca4 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/030_dwd_ad_conversion_fact.sql @@ -0,0 +1,6 @@ +-- DWD / attribution: 广告转化事实 +CREATE OR REPLACE TABLE dwd_ad_conversion_fact AS +SELECT c.*, k.ad_impression_id, k.advertiser_id, k.ad_slot_id, + DATE_DIFF('second', k.click_time, c.conversion_time) AS conversion_delay_sec + FROM stg_ad_conversion c + JOIN dwd_ad_click_fact k ON c.ad_click_id = k.ad_click_id; diff --git a/examples/video_commercial_warehouse/sql/031_dwd_attribution_touch_fact.sql b/examples/video_commercial_warehouse/sql/031_dwd_attribution_touch_fact.sql new file mode 100644 index 0000000..0913ff9 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/031_dwd_attribution_touch_fact.sql @@ -0,0 +1,9 @@ +-- DWD / attribution: 点击到转化的归因触点 +CREATE OR REPLACE TABLE dwd_attribution_touch_fact AS +SELECT c.ad_click_id AS touch_id, c.ad_impression_id, c.user_id, + c.ad_creative_id, c.campaign_id, c.advertiser_id, + v.conversion_id, COALESCE(v.conversion_value, 0) AS conversion_value, + CASE WHEN v.conversion_id IS NULL THEN 0 ELSE 1 END AS is_converted, + c.event_date + FROM dwd_ad_click_fact c + LEFT JOIN dwd_ad_conversion_fact v ON c.ad_click_id = v.ad_click_id; diff --git a/examples/video_commercial_warehouse/sql/032_dwd_ad_attribution_wide.sql b/examples/video_commercial_warehouse/sql/032_dwd_ad_attribution_wide.sql new file mode 100644 index 0000000..fc28a33 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/032_dwd_ad_attribution_wide.sql @@ -0,0 +1,11 @@ +-- DWD / attribution: 曝光点击转化归因宽表 +CREATE OR REPLACE TABLE dwd_ad_attribution_wide AS +SELECT i.ad_impression_id, i.ad_request_id, i.user_id, i.ad_creative_id, + i.ad_group_id, i.campaign_id, i.advertiser_id, i.ad_slot_id, + i.cost, i.event_date, + CASE WHEN c.ad_click_id IS NULL THEN 0 ELSE 1 END AS is_clicked, + CASE WHEN v.conversion_id IS NULL THEN 0 ELSE 1 END AS is_converted, + COALESCE(v.conversion_value, 0) AS conversion_value + FROM dwd_ad_delivery_fact i + LEFT JOIN dwd_ad_click_fact c ON i.ad_impression_id = c.ad_impression_id + LEFT JOIN dwd_ad_conversion_fact v ON c.ad_click_id = v.ad_click_id; diff --git a/examples/video_commercial_warehouse/sql/033_dwd_billing_fact.sql b/examples/video_commercial_warehouse/sql/033_dwd_billing_fact.sql new file mode 100644 index 0000000..8c6d6c0 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/033_dwd_billing_fact.sql @@ -0,0 +1,6 @@ +-- DWD / billing: 广告扣费事实 +CREATE OR REPLACE TABLE dwd_billing_fact AS +SELECT c.*, a.industry, p.objective + FROM stg_ad_charge c + LEFT JOIN dim_advertiser a ON c.advertiser_id = a.advertiser_id + LEFT JOIN dim_campaign p ON c.campaign_id = p.campaign_id; diff --git a/examples/video_commercial_warehouse/sql/034_dwd_refund_fact.sql b/examples/video_commercial_warehouse/sql/034_dwd_refund_fact.sql new file mode 100644 index 0000000..8fbaec3 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/034_dwd_refund_fact.sql @@ -0,0 +1,5 @@ +-- DWD / billing: 广告退款事实 +CREATE OR REPLACE TABLE dwd_refund_fact AS +SELECT r.*, b.campaign_id, b.event_date AS charge_date + FROM stg_ad_refund r + LEFT JOIN dwd_billing_fact b ON r.charge_id = b.charge_id; diff --git a/examples/video_commercial_warehouse/sql/035_dwd_budget_fact.sql b/examples/video_commercial_warehouse/sql/035_dwd_budget_fact.sql new file mode 100644 index 0000000..3707020 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/035_dwd_budget_fact.sql @@ -0,0 +1,6 @@ +-- DWD / billing: 预算与消耗节奏事实 +CREATE OR REPLACE TABLE dwd_budget_fact AS +SELECT b.*, c.advertiser_id, c.objective, c.status, + b.spent / NULLIF(b.budget, 0) AS pacing_ratio + FROM stg_budget_snapshot b + LEFT JOIN dim_campaign c ON b.campaign_id = c.campaign_id; diff --git a/examples/video_commercial_warehouse/sql/036_dwd_risk_event_fact.sql b/examples/video_commercial_warehouse/sql/036_dwd_risk_event_fact.sql new file mode 100644 index 0000000..26f3ab2 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/036_dwd_risk_event_fact.sql @@ -0,0 +1,11 @@ +-- DWD / risk: 内容、广告与无效流量统一风险事实 +CREATE OR REPLACE TABLE dwd_risk_event_fact AS +SELECT audit_id AS risk_event_id, 'video' AS entity_type, video_id AS entity_id, + audit_status AS risk_type, risk_score, event_date + FROM stg_content_audit + UNION ALL + SELECT audit_id, 'ad_creative', ad_creative_id, audit_status, risk_score, event_date + FROM stg_ad_audit + UNION ALL + SELECT signal_id, 'request', request_id, signal_type, risk_score, event_date + FROM stg_invalid_traffic_signal; diff --git a/examples/video_commercial_warehouse/sql/037_dws_video_traffic_daily.sql b/examples/video_commercial_warehouse/sql/037_dws_video_traffic_daily.sql new file mode 100644 index 0000000..04e77df --- /dev/null +++ b/examples/video_commercial_warehouse/sql/037_dws_video_traffic_daily.sql @@ -0,0 +1,21 @@ +-- DWS / traffic: 视频流量效率日报 +CREATE OR REPLACE TABLE dws_video_traffic_daily AS +WITH e AS ( + SELECT video_id, event_date, COUNT(*) AS exposure_count + FROM dwd_video_exposure_fact GROUP BY video_id, event_date + ), p AS ( + SELECT video_id, event_date, COUNT(*) AS play_count, + SUM(is_valid_play) AS valid_play_count, + AVG(completion_rate) AS avg_completion_rate + FROM dwd_video_play_fact GROUP BY video_id, event_date + ), i AS ( + SELECT video_id, event_date, COUNT(*) AS interaction_count + FROM dwd_user_interaction_fact GROUP BY video_id, event_date + ) + SELECT e.video_id, e.event_date, e.exposure_count, + COALESCE(p.play_count, 0) AS play_count, + COALESCE(p.valid_play_count, 0) AS valid_play_count, + COALESCE(p.avg_completion_rate, 0) AS avg_completion_rate, + COALESCE(i.interaction_count, 0) AS interaction_count + FROM e LEFT JOIN p USING(video_id, event_date) + LEFT JOIN i USING(video_id, event_date); diff --git a/examples/video_commercial_warehouse/sql/038_dws_creator_traffic_daily.sql b/examples/video_commercial_warehouse/sql/038_dws_creator_traffic_daily.sql new file mode 100644 index 0000000..80d70c2 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/038_dws_creator_traffic_daily.sql @@ -0,0 +1,8 @@ +-- DWS / content: 作者流量日报 +CREATE OR REPLACE TABLE dws_creator_traffic_daily AS +SELECT v.creator_id, d.event_date, SUM(d.exposure_count) AS exposure_count, + SUM(d.play_count) AS play_count, SUM(d.interaction_count) AS interaction_count, + AVG(d.avg_completion_rate) AS avg_completion_rate + FROM dws_video_traffic_daily d + JOIN dim_video v ON d.video_id = v.video_id + GROUP BY v.creator_id, d.event_date; diff --git a/examples/video_commercial_warehouse/sql/039_dws_user_engagement_daily.sql b/examples/video_commercial_warehouse/sql/039_dws_user_engagement_daily.sql new file mode 100644 index 0000000..0d9b7d0 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/039_dws_user_engagement_daily.sql @@ -0,0 +1,13 @@ +-- DWS / user: 用户参与度日报 +CREATE OR REPLACE TABLE dws_user_engagement_daily AS +WITH p AS ( + SELECT user_id, event_date, COUNT(*) AS play_count, + SUM(play_duration_sec) AS play_duration_sec + FROM dwd_video_play_fact GROUP BY user_id, event_date + ), i AS ( + SELECT user_id, event_date, COUNT(*) AS interaction_count + FROM dwd_user_interaction_fact GROUP BY user_id, event_date + ) + SELECT p.user_id, p.event_date, p.play_count, p.play_duration_sec, + COALESCE(i.interaction_count, 0) AS interaction_count + FROM p LEFT JOIN i USING(user_id, event_date); diff --git a/examples/video_commercial_warehouse/sql/040_dws_channel_traffic_daily.sql b/examples/video_commercial_warehouse/sql/040_dws_channel_traffic_daily.sql new file mode 100644 index 0000000..b94239d --- /dev/null +++ b/examples/video_commercial_warehouse/sql/040_dws_channel_traffic_daily.sql @@ -0,0 +1,8 @@ +-- DWS / traffic: 渠道流量日报 +CREATE OR REPLACE TABLE dws_channel_traffic_daily AS +SELECT r.channel_id, r.channel_name, r.event_date, + COUNT(DISTINCT r.request_id) AS request_count, + COUNT(e.exposure_id) AS exposure_count + FROM dwd_recommend_request_fact r + LEFT JOIN dwd_video_exposure_fact e ON r.request_id = e.request_id + GROUP BY r.channel_id, r.channel_name, r.event_date; diff --git a/examples/video_commercial_warehouse/sql/041_dws_category_traffic_daily.sql b/examples/video_commercial_warehouse/sql/041_dws_category_traffic_daily.sql new file mode 100644 index 0000000..21832a6 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/041_dws_category_traffic_daily.sql @@ -0,0 +1,9 @@ +-- DWS / content: 内容分类流量日报 +CREATE OR REPLACE TABLE dws_category_traffic_daily AS +SELECT e.category_id, e.category_name, e.event_date, + COUNT(DISTINCT e.exposure_id) AS exposure_count, + COUNT(DISTINCT p.play_id) AS play_count, + AVG(p.completion_rate) AS avg_completion_rate + FROM dwd_video_exposure_fact e + LEFT JOIN dwd_video_play_fact p ON e.exposure_id = p.exposure_id + GROUP BY e.category_id, e.category_name, e.event_date; diff --git a/examples/video_commercial_warehouse/sql/042_dws_ad_slot_funnel_daily.sql b/examples/video_commercial_warehouse/sql/042_dws_ad_slot_funnel_daily.sql new file mode 100644 index 0000000..81360fa --- /dev/null +++ b/examples/video_commercial_warehouse/sql/042_dws_ad_slot_funnel_daily.sql @@ -0,0 +1,23 @@ +-- DWS / ad_inventory: 广告位请求到转化漏斗 +CREATE OR REPLACE TABLE dws_ad_slot_funnel_daily AS +WITH o AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS request_count + FROM dwd_ad_opportunity_fact GROUP BY ad_slot_id, event_date + ), i AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS impression_count + FROM dwd_ad_delivery_fact GROUP BY ad_slot_id, event_date + ), c AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS click_count + FROM dwd_ad_click_fact GROUP BY ad_slot_id, event_date + ), v AS ( + SELECT ad_slot_id, event_date, COUNT(*) AS conversion_count + FROM dwd_ad_conversion_fact GROUP BY ad_slot_id, event_date + ) + SELECT o.ad_slot_id, o.event_date, o.request_count, + COALESCE(i.impression_count,0) AS impression_count, + COALESCE(c.click_count,0) AS click_count, + COALESCE(v.conversion_count,0) AS conversion_count, + COALESCE(i.impression_count,0)::DOUBLE / NULLIF(o.request_count,0) AS fill_rate + FROM o LEFT JOIN i USING(ad_slot_id,event_date) + LEFT JOIN c USING(ad_slot_id,event_date) + LEFT JOIN v USING(ad_slot_id,event_date); diff --git a/examples/video_commercial_warehouse/sql/043_dws_campaign_delivery_daily.sql b/examples/video_commercial_warehouse/sql/043_dws_campaign_delivery_daily.sql new file mode 100644 index 0000000..3d36a89 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/043_dws_campaign_delivery_daily.sql @@ -0,0 +1,8 @@ +-- DWS / ad_delivery: 计划投放效果日报 +CREATE OR REPLACE TABLE dws_campaign_delivery_daily AS +SELECT campaign_id, advertiser_id, event_date, + COUNT(*) AS impression_count, SUM(is_clicked) AS click_count, + SUM(is_converted) AS conversion_count, SUM(cost) AS spend, + SUM(conversion_value) AS conversion_value + FROM dwd_ad_attribution_wide + GROUP BY campaign_id, advertiser_id, event_date; diff --git a/examples/video_commercial_warehouse/sql/044_dws_creative_performance_daily.sql b/examples/video_commercial_warehouse/sql/044_dws_creative_performance_daily.sql new file mode 100644 index 0000000..08f1574 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/044_dws_creative_performance_daily.sql @@ -0,0 +1,11 @@ +-- DWS / ad_delivery: 素材效果日报(故意保留百分数 CTR 作为治理场景) +CREATE OR REPLACE TABLE dws_creative_performance_daily AS +SELECT w.ad_creative_id, c.ad_group_id, w.campaign_id, w.advertiser_id, + w.event_date, COUNT(*) AS impression_count, + SUM(w.is_clicked) AS click_count, SUM(w.is_converted) AS conversion_count, + SUM(w.cost) AS spend, SUM(w.conversion_value) AS conversion_value, + ROUND(SUM(w.is_clicked) / NULLIF(COUNT(*),0), 6) AS ctr, + SUM(w.is_converted)::DOUBLE / NULLIF(SUM(w.is_clicked),0) AS cvr + FROM dwd_ad_attribution_wide w + LEFT JOIN dim_ad_creative c ON w.ad_creative_id = c.ad_creative_id + GROUP BY w.ad_creative_id, c.ad_group_id, w.campaign_id, w.advertiser_id, w.event_date; diff --git a/examples/video_commercial_warehouse/sql/045_dws_advertiser_performance_daily.sql b/examples/video_commercial_warehouse/sql/045_dws_advertiser_performance_daily.sql new file mode 100644 index 0000000..16740f8 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/045_dws_advertiser_performance_daily.sql @@ -0,0 +1,6 @@ +-- DWS / billing: 广告主效果日报 +CREATE OR REPLACE TABLE dws_advertiser_performance_daily AS +SELECT advertiser_id, event_date, SUM(impression_count) AS impression_count, + SUM(click_count) AS click_count, SUM(conversion_count) AS conversion_count, + SUM(spend) AS spend, SUM(conversion_value) AS conversion_value + FROM dws_campaign_delivery_daily GROUP BY advertiser_id, event_date; diff --git a/examples/video_commercial_warehouse/sql/046_dws_audience_performance_daily.sql b/examples/video_commercial_warehouse/sql/046_dws_audience_performance_daily.sql new file mode 100644 index 0000000..4343420 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/046_dws_audience_performance_daily.sql @@ -0,0 +1,8 @@ +-- DWS / experiment: 受众效果日报 +CREATE OR REPLACE TABLE dws_audience_performance_daily AS +SELECT u.audience_id, w.event_date, COUNT(*) AS impression_count, + SUM(w.is_clicked) AS click_count, SUM(w.is_converted) AS conversion_count, + SUM(w.cost) AS spend + FROM dwd_ad_attribution_wide w + LEFT JOIN dim_user u ON w.user_id = u.user_id + GROUP BY u.audience_id, w.event_date; diff --git a/examples/video_commercial_warehouse/sql/047_dws_experiment_daily.sql b/examples/video_commercial_warehouse/sql/047_dws_experiment_daily.sql new file mode 100644 index 0000000..f2d94a0 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/047_dws_experiment_daily.sql @@ -0,0 +1,10 @@ +-- DWS / experiment: 实验流量与广告机会日报 +CREATE OR REPLACE TABLE dws_experiment_daily AS +SELECT r.experiment_id, r.experiment_variant, r.event_date, + COUNT(DISTINCT r.request_id) AS request_count, + COUNT(DISTINCT e.exposure_id) AS video_exposure_count, + COUNT(DISTINCT a.ad_request_id) AS ad_request_count + FROM dwd_recommend_request_fact r + LEFT JOIN dwd_video_exposure_fact e ON r.request_id = e.request_id + LEFT JOIN dwd_ad_opportunity_fact a ON r.request_id = a.request_id + GROUP BY r.experiment_id, r.experiment_variant, r.event_date; diff --git a/examples/video_commercial_warehouse/sql/048_dws_monetization_funnel_daily.sql b/examples/video_commercial_warehouse/sql/048_dws_monetization_funnel_daily.sql new file mode 100644 index 0000000..26891c3 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/048_dws_monetization_funnel_daily.sql @@ -0,0 +1,15 @@ +-- DWS / ad_inventory: 商业化全漏斗日报 +CREATE OR REPLACE TABLE dws_monetization_funnel_daily AS +SELECT o.event_date, COUNT(DISTINCT o.ad_request_id) AS request_count, + COUNT(DISTINCT c.candidate_id) AS candidate_count, + COUNT(DISTINCT CASE WHEN a.is_winner THEN a.bid_id END) AS win_count, + COUNT(DISTINCT i.ad_impression_id) AS impression_count, + COUNT(DISTINCT k.ad_click_id) AS click_count, + COUNT(DISTINCT v.conversion_id) AS conversion_count + FROM dwd_ad_opportunity_fact o + LEFT JOIN dwd_ad_candidate_fact c ON o.ad_request_id = c.ad_request_id + LEFT JOIN dwd_ad_auction_fact a ON c.candidate_id = a.candidate_id + LEFT JOIN dwd_ad_delivery_fact i ON o.ad_request_id = i.ad_request_id + LEFT JOIN dwd_ad_click_fact k ON i.ad_impression_id = k.ad_impression_id + LEFT JOIN dwd_ad_conversion_fact v ON k.ad_click_id = v.ad_click_id + GROUP BY o.event_date; diff --git a/examples/video_commercial_warehouse/sql/049_dws_revenue_daily.sql b/examples/video_commercial_warehouse/sql/049_dws_revenue_daily.sql new file mode 100644 index 0000000..b9c9695 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/049_dws_revenue_daily.sql @@ -0,0 +1,13 @@ +-- DWS / billing: 平台广告收入日报 +CREATE OR REPLACE TABLE dws_revenue_daily AS +WITH c AS ( + SELECT advertiser_id, event_date, SUM(amount) AS gross_revenue + FROM dwd_billing_fact GROUP BY advertiser_id, event_date + ), r AS ( + SELECT advertiser_id, event_date, SUM(amount) AS refund_amount + FROM dwd_refund_fact GROUP BY advertiser_id, event_date + ) + SELECT c.advertiser_id, c.event_date, c.gross_revenue, + COALESCE(r.refund_amount,0) AS refund_amount, + c.gross_revenue - COALESCE(r.refund_amount,0) AS net_revenue + FROM c LEFT JOIN r USING(advertiser_id,event_date); diff --git a/examples/video_commercial_warehouse/sql/050_dws_roi_daily.sql b/examples/video_commercial_warehouse/sql/050_dws_roi_daily.sql new file mode 100644 index 0000000..b241173 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/050_dws_roi_daily.sql @@ -0,0 +1,5 @@ +-- DWS / attribution: 计划 ROI 日报 +CREATE OR REPLACE TABLE dws_roi_daily AS +SELECT campaign_id, advertiser_id, event_date, spend, conversion_value, + conversion_value / NULLIF(spend,0) AS roi + FROM dws_campaign_delivery_daily; diff --git a/examples/video_commercial_warehouse/sql/051_dws_budget_pacing_daily.sql b/examples/video_commercial_warehouse/sql/051_dws_budget_pacing_daily.sql new file mode 100644 index 0000000..017e28f --- /dev/null +++ b/examples/video_commercial_warehouse/sql/051_dws_budget_pacing_daily.sql @@ -0,0 +1,9 @@ +-- DWS / billing: 预算节奏日报 +CREATE OR REPLACE TABLE dws_budget_pacing_daily AS +SELECT b.campaign_id, b.advertiser_id, b.snapshot_date AS event_date, + b.budget, b.spent AS snapshot_spent, + COALESCE(d.spend,0) AS actual_spend, + COALESCE(d.spend,0) / NULLIF(b.budget,0) AS pacing_ratio + FROM dwd_budget_fact b + LEFT JOIN dws_campaign_delivery_daily d + ON b.campaign_id = d.campaign_id AND b.snapshot_date = d.event_date; diff --git a/examples/video_commercial_warehouse/sql/052_dws_risk_daily.sql b/examples/video_commercial_warehouse/sql/052_dws_risk_daily.sql new file mode 100644 index 0000000..5ada06c --- /dev/null +++ b/examples/video_commercial_warehouse/sql/052_dws_risk_daily.sql @@ -0,0 +1,6 @@ +-- DWS / risk: 内容、广告与流量风险日报 +CREATE OR REPLACE TABLE dws_risk_daily AS +SELECT entity_type, event_date, COUNT(*) AS signal_count, + SUM(CASE WHEN risk_score >= 0.8 THEN 1 ELSE 0 END) AS high_risk_count, + AVG(risk_score) AS avg_risk_score + FROM dwd_risk_event_fact GROUP BY entity_type, event_date; diff --git a/examples/video_commercial_warehouse/sql/053_ads_traffic_overview.sql b/examples/video_commercial_warehouse/sql/053_ads_traffic_overview.sql new file mode 100644 index 0000000..93cf73c --- /dev/null +++ b/examples/video_commercial_warehouse/sql/053_ads_traffic_overview.sql @@ -0,0 +1,9 @@ +-- ADS / traffic: 平台流量总览 +CREATE OR REPLACE TABLE ads_traffic_overview AS +SELECT c.event_date, SUM(c.request_count) AS request_count, + SUM(c.exposure_count) AS exposure_count, + SUM(v.play_count) AS play_count, + SUM(v.valid_play_count) AS valid_play_count + FROM dws_channel_traffic_daily c + LEFT JOIN dws_video_traffic_daily v ON c.event_date = v.event_date + GROUP BY c.event_date; diff --git a/examples/video_commercial_warehouse/sql/054_ads_creator_dashboard.sql b/examples/video_commercial_warehouse/sql/054_ads_creator_dashboard.sql new file mode 100644 index 0000000..018cd1a --- /dev/null +++ b/examples/video_commercial_warehouse/sql/054_ads_creator_dashboard.sql @@ -0,0 +1,6 @@ +-- ADS / content: 作者流量与互动看板 +CREATE OR REPLACE TABLE ads_creator_dashboard AS +SELECT d.*, c.creator_tier, + d.interaction_count::DOUBLE / NULLIF(d.play_count,0) AS interaction_rate + FROM dws_creator_traffic_daily d + LEFT JOIN dim_creator c ON d.creator_id = c.creator_id; diff --git a/examples/video_commercial_warehouse/sql/055_ads_content_efficiency_report.sql b/examples/video_commercial_warehouse/sql/055_ads_content_efficiency_report.sql new file mode 100644 index 0000000..d798c42 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/055_ads_content_efficiency_report.sql @@ -0,0 +1,4 @@ +-- ADS / content: 内容分类效率报告 +CREATE OR REPLACE TABLE ads_content_efficiency_report AS +SELECT *, play_count::DOUBLE / NULLIF(exposure_count,0) AS play_rate + FROM dws_category_traffic_daily; diff --git a/examples/video_commercial_warehouse/sql/056_ads_ad_operation_dashboard.sql b/examples/video_commercial_warehouse/sql/056_ads_ad_operation_dashboard.sql new file mode 100644 index 0000000..19f1416 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/056_ads_ad_operation_dashboard.sql @@ -0,0 +1,11 @@ +-- ADS / ad_inventory: 广告运营总览 +CREATE OR REPLACE TABLE ads_ad_operation_dashboard AS +WITH revenue AS ( + SELECT event_date, SUM(net_revenue) AS net_revenue + FROM dws_revenue_daily GROUP BY event_date + ) + SELECT f.*, COALESCE(r.net_revenue,0) AS net_revenue, + f.impression_count::DOUBLE / NULLIF(f.request_count,0) AS fill_rate, + COALESCE(r.net_revenue,0) * 1000.0 / NULLIF(f.impression_count,0) AS ecpm + FROM dws_monetization_funnel_daily f + LEFT JOIN revenue r ON f.event_date = r.event_date; diff --git a/examples/video_commercial_warehouse/sql/057_ads_advertiser_report.sql b/examples/video_commercial_warehouse/sql/057_ads_advertiser_report.sql new file mode 100644 index 0000000..d2bfd41 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/057_ads_advertiser_report.sql @@ -0,0 +1,7 @@ +-- ADS / billing: 广告主经营报告 +CREATE OR REPLACE TABLE ads_advertiser_report AS +SELECT p.*, a.advertiser_name, a.industry, r.net_revenue, + p.conversion_value / NULLIF(p.spend,0) AS roi + FROM dws_advertiser_performance_daily p + LEFT JOIN dws_revenue_daily r USING(advertiser_id,event_date) + LEFT JOIN dim_advertiser a USING(advertiser_id); diff --git a/examples/video_commercial_warehouse/sql/058_ads_campaign_monitor.sql b/examples/video_commercial_warehouse/sql/058_ads_campaign_monitor.sql new file mode 100644 index 0000000..0739ac2 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/058_ads_campaign_monitor.sql @@ -0,0 +1,8 @@ +-- ADS / ad_delivery: 计划投放与预算监控 +CREATE OR REPLACE TABLE ads_campaign_monitor AS +SELECT d.*, c.objective, c.status, b.budget, b.pacing_ratio, + d.click_count::DOUBLE / NULLIF(d.impression_count,0) AS ctr, + d.conversion_count::DOUBLE / NULLIF(d.click_count,0) AS cvr + FROM dws_campaign_delivery_daily d + LEFT JOIN dws_budget_pacing_daily b USING(campaign_id,advertiser_id,event_date) + LEFT JOIN dim_campaign c USING(campaign_id,advertiser_id); diff --git a/examples/video_commercial_warehouse/sql/059_ads_creative_report.sql b/examples/video_commercial_warehouse/sql/059_ads_creative_report.sql new file mode 100644 index 0000000..57eea7a --- /dev/null +++ b/examples/video_commercial_warehouse/sql/059_ads_creative_report.sql @@ -0,0 +1,14 @@ +-- ADS / ad_delivery: 素材效果与 CTR 分级报告 +CREATE OR REPLACE TABLE ads_creative_report AS +SELECT p.*, c.creative_format, + CASE + WHEN p.ctr >= 0.05 THEN 'A_excellent' + WHEN p.ctr >= 0.03 THEN 'B_good' + WHEN p.ctr >= 0.01 THEN 'C_normal' + ELSE 'D_poor' + END AS ctr_level, + RANK() OVER ( + PARTITION BY p.advertiser_id, p.event_date ORDER BY p.ctr DESC + ) AS ctr_rank + FROM dws_creative_performance_daily p + LEFT JOIN dim_ad_creative c USING(ad_creative_id,ad_group_id); diff --git a/examples/video_commercial_warehouse/sql/060_ads_revenue_dashboard.sql b/examples/video_commercial_warehouse/sql/060_ads_revenue_dashboard.sql new file mode 100644 index 0000000..fc204d5 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/060_ads_revenue_dashboard.sql @@ -0,0 +1,8 @@ +-- ADS / billing: 商业收入与 ROI 驾驶舱 +CREATE OR REPLACE TABLE ads_revenue_dashboard AS +SELECT r.event_date, SUM(r.gross_revenue) AS gross_revenue, + SUM(r.refund_amount) AS refund_amount, SUM(r.net_revenue) AS net_revenue, + AVG(i.roi) AS avg_roi + FROM dws_revenue_daily r + LEFT JOIN dws_roi_daily i ON r.event_date = i.event_date + GROUP BY r.event_date; diff --git a/examples/video_commercial_warehouse/sql/061_ads_experiment_report.sql b/examples/video_commercial_warehouse/sql/061_ads_experiment_report.sql new file mode 100644 index 0000000..81c813e --- /dev/null +++ b/examples/video_commercial_warehouse/sql/061_ads_experiment_report.sql @@ -0,0 +1,5 @@ +-- ADS / experiment: 实验流量与商业化效果报告 +CREATE OR REPLACE TABLE ads_experiment_report AS +SELECT *, ad_request_count::DOUBLE / NULLIF(request_count,0) AS ad_request_rate, + video_exposure_count::DOUBLE / NULLIF(request_count,0) AS exposure_per_request + FROM dws_experiment_daily; diff --git a/examples/video_commercial_warehouse/sql/062_ads_risk_dashboard.sql b/examples/video_commercial_warehouse/sql/062_ads_risk_dashboard.sql new file mode 100644 index 0000000..60e5e06 --- /dev/null +++ b/examples/video_commercial_warehouse/sql/062_ads_risk_dashboard.sql @@ -0,0 +1,7 @@ +-- ADS / risk: 商业化质量与风险驾驶舱 +CREATE OR REPLACE TABLE ads_risk_dashboard AS +SELECT *, high_risk_count::DOUBLE / NULLIF(signal_count,0) AS high_risk_rate, + CASE WHEN avg_risk_score >= 0.8 THEN 'critical' + WHEN avg_risk_score >= 0.5 THEN 'warning' + ELSE 'normal' END AS risk_level + FROM dws_risk_daily; diff --git a/examples/video_commercial_warehouse/templates/workbench.html b/examples/video_commercial_warehouse/templates/workbench.html new file mode 100644 index 0000000..698d41b --- /dev/null +++ b/examples/video_commercial_warehouse/templates/workbench.html @@ -0,0 +1,443 @@ + + + + + + +__TITLE__ · SqlGraph + + + + + +
+
+
SqlGraph Explorer
视频商业化数仓
+ +

当前任务

+
+
+
+
+
当前治理任务

+ + +
+

发生了什么

业务结论先于技术细节

+
口径不一致

上游输出百分数,下游按比率解释

素材 CTR 原值 被当作比率使用,导致低点击率素材被错误评级。治理动作统一为 [0,1] 比率,并只重跑受影响任务。

+
修复前越界
影响任务
修复后越界
自治等级
+
+

影响路径

只展示本次决策所需证据

+
DWDDWSADS +
+
结构影响
增量重跑
证据基线
可逆性可恢复
+
+

执行与验证

修改内容、执行任务与验证结论

+
+ + +单位:percentage → ratio +范围:不确定 → [0,1]
+
+
+
+ +
+

血缘分析

,按任务上下文逐步缩小范围

+
+
+
+
+
+
+
+
+

没有匹配结果

调整关键词、层级或业务主题。

+ +
+
+
+ +
+

执行计划

先过安全闸门,再执行受影响 DAG

+

SQL 修改

+
+

自治决策

动作等级
人工审批
可逆性否决
授权范围单次 L3
+

增量执行顺序

+

恢复资产

SQL 快照before/sql/数据库快照before.duckdb当前数据库after.duckdb
+
+
+ +
+

验证结果

三层独立报告,运行态来自 DuckDB 实际查询

+

DuckDB 真实查询对比

指标修复前修复后单位
+

残余风险

验证成功不等于零风险

+
+ +
+

审计摘要

七步共享同一证据版本,原始记录默认折叠

查看原始审计记录
+
+
+
+
+ + + + diff --git a/pyproject.toml b/pyproject.toml index e5ddf7c..3a53703 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,9 +11,14 @@ documentation = "https://github.com/Liyurun/SqlGraph#readme" keywords = ["sql", "lineage", "data-lineage", "knowledge-graph", "data-warehouse", "sqlglot", "graphrag", "data-governance"] packages = [{include = "sqlgraph"}] include = [ + { path = "CAPABILITIES.yaml", format = ["sdist", "wheel"] }, { path = "assets/*.png", format = "sdist" }, - { path = "docs/architecture.md", format = "sdist" }, + { path = "docs/*.md", format = "sdist" }, + { path = "schemas/*.json", format = ["sdist", "wheel"] }, { path = "examples/ads_pipeline/**/*", format = ["sdist", "wheel"] }, + { path = "examples/minimal/**/*", format = ["sdist", "wheel"] }, + { path = "examples/book_cases/**/*", format = ["sdist", "wheel"] }, + { path = "examples/video_commercial_warehouse/**/*", format = "sdist" }, { path = "examples/df_sample.csv", format = ["sdist", "wheel"] }, { path = "examples/table_ddl_sample.csv", format = ["sdist", "wheel"] }, { path = "sqlgraph/serve/web/**/*.j2", format = ["sdist", "wheel"] }, @@ -26,7 +31,6 @@ classifiers = [ "Intended Audience :: Information Technology", "License :: OSI Approved :: MIT License", "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.9", "Programming Language :: Python :: 3.10", "Programming Language :: Python :: 3.11", "Programming Language :: Python :: 3.12", @@ -38,12 +42,15 @@ classifiers = [ sqlgraph = "sqlgraph.cli:main" [tool.poetry.dependencies] -python = ">=3.9,<3.13" +python = ">=3.10,<3.13" sqlglot = ">=25.0.0" pandas = ">=2.0.0" typer = ">=0.9.0" jinja2 = ">=3.1.0" rich = ">=13.0.0" +duckdb = ">=1.0.0" +jsonschema = ">=4.18.0" +pyyaml = ">=6.0.0" networkx = {version = ">=3.0", optional = true} ipython = {version = ">=8.0", optional = true} node2vec = {version = ">=0.5.0", optional = true} @@ -52,6 +59,9 @@ scikit-learn = {version = ">=1.3.0", optional = true} [tool.poetry.group.dev.dependencies] pytest = ">=7.0.0" pytest-cov = ">=4.0.0" +ruff = ">=0.5.0" +mypy = ">=1.8.0" +build = ">=1.2.0" [tool.poetry.extras] analysis = [] @@ -77,3 +87,31 @@ testpaths = ["tests"] python_files = "test_*.py" python_classes = "Test*" python_functions = "test_*" + +[tool.coverage.run] +source = ["sqlgraph"] +branch = false + +[tool.coverage.report] +show_missing = true + +[tool.ruff] +line-length = 120 +target-version = "py310" +extend-exclude = ["docs"] + +[tool.ruff.lint] +select = ["F", "B", "E7"] +ignore = ["B008", "B904", "E702", "F401", "F541", "F841"] + +[tool.ruff.lint.per-file-ignores] +"**/__init__.py" = ["F401"] +"tests/**" = ["F401", "B"] + +[tool.mypy] +python_version = "3.10" +ignore_missing_imports = true +warn_unused_ignores = false +disallow_untyped_defs = false +check_untyped_defs = false +exclude = "(^|/)(docs|tests)/" diff --git a/schemas/action-plan-v1.schema.json b/schemas/action-plan-v1.schema.json new file mode 100644 index 0000000..a4453f7 --- /dev/null +++ b/schemas/action-plan-v1.schema.json @@ -0,0 +1,29 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/action-plan-v1.schema.json", + "title": "SqlGraph Action Plan v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "task_id", + "baseline_id", + "evidence_version", + "decision_id", + "adapter", + "operations", + "rollback_plan", + "idempotency_key" + ], + "properties": { + "schema_version": {"const": "action-plan-v1"}, + "task_id": {"type": "string", "minLength": 1}, + "baseline_id": {"type": "string", "minLength": 1}, + "evidence_version": {"type": "string", "minLength": 1}, + "decision_id": {"type": "string", "minLength": 1}, + "adapter": {"type": "string", "minLength": 1}, + "operations": {"type": "array", "minItems": 1, "items": {"type": "object"}}, + "rollback_plan": {"type": "array", "items": {"type": "object"}}, + "idempotency_key": {"type": "string", "pattern": "^idem_[0-9a-f]{32}$"} + } +} diff --git a/schemas/audit-event-v1.schema.json b/schemas/audit-event-v1.schema.json new file mode 100644 index 0000000..7247a82 --- /dev/null +++ b/schemas/audit-event-v1.schema.json @@ -0,0 +1,43 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/audit-event-v1.schema.json", + "title": "SqlGraph Audit Event v1", + "type": "object", + "additionalProperties": false, + "required": [ + "task_id", + "baseline_id", + "event_type", + "step", + "payload", + "evidence_version", + "policy_version", + "authorization_identity", + "idempotency_key", + "transition", + "sequence", + "timestamp", + "previous_hash", + "event_hash", + "event_id", + "schema_version" + ], + "properties": { + "task_id": {"type": "string", "minLength": 1}, + "baseline_id": {"type": "string", "minLength": 1}, + "event_type": {"type": "string", "minLength": 1}, + "step": {"type": "string", "minLength": 1}, + "payload": {"type": "object"}, + "evidence_version": {"type": "string"}, + "policy_version": {"type": "string"}, + "authorization_identity": {"type": "string"}, + "idempotency_key": {"type": "string"}, + "transition": {"type": "string"}, + "sequence": {"type": "integer", "minimum": 1}, + "timestamp": {"type": "string", "minLength": 1}, + "previous_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "event_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "event_id": {"type": "string", "pattern": "^event_[0-9a-f]{32}$"}, + "schema_version": {"const": "audit-event-v1"} + } +} diff --git a/schemas/autonomy-decision-v1.schema.json b/schemas/autonomy-decision-v1.schema.json new file mode 100644 index 0000000..087b98c --- /dev/null +++ b/schemas/autonomy-decision-v1.schema.json @@ -0,0 +1,33 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/autonomy-decision-v1.schema.json", + "title": "SqlGraph Autonomy Decision v1", + "type": "object", + "additionalProperties": false, + "required": [ + "level", + "requires_human_review", + "evidence_version", + "reasons", + "reversibility_veto", + "evidence_veto", + "authorization_veto", + "scoring_performed", + "within_l5_scope", + "policy_version", + "decision_id" + ], + "properties": { + "level": {"enum": ["L0", "L1", "L2", "L3", "L4", "L5"]}, + "requires_human_review": {"type": "boolean"}, + "evidence_version": {"type": "string"}, + "reasons": {"type": "array", "items": {"type": "string"}}, + "reversibility_veto": {"type": "boolean"}, + "evidence_veto": {"type": "boolean"}, + "authorization_veto": {"type": "boolean"}, + "scoring_performed": {"type": "boolean"}, + "within_l5_scope": {"type": "boolean"}, + "policy_version": {"type": "string"}, + "decision_id": {"type": "string", "pattern": "^decision_[0-9a-f]{32}$"} + } +} diff --git a/schemas/baseline-manifest-v1.schema.json b/schemas/baseline-manifest-v1.schema.json new file mode 100644 index 0000000..5d37611 --- /dev/null +++ b/schemas/baseline-manifest-v1.schema.json @@ -0,0 +1,63 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/baseline-manifest-v1.schema.json", + "title": "SqlGraph Baseline Manifest v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "baseline_id", + "source_hashes", + "dialect", + "parser_version", + "identity_rule_version", + "dependency_hashes", + "missing_dependencies", + "created_at" + ], + "properties": { + "schema_version": { + "const": "baseline-manifest-v1" + }, + "baseline_id": { + "type": "string", + "pattern": "^base_[0-9a-f]{64}$" + }, + "source_hashes": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "additionalProperties": false, + "required": ["name", "source_type", "source_path", "sha256"], + "properties": { + "name": {"type": "string", "minLength": 1}, + "source_type": {"type": "string", "minLength": 1}, + "source_path": {"type": "string"}, + "sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + } + } + }, + "dialect": {"type": "string"}, + "parser_version": {"type": "string", "minLength": 1}, + "identity_rule_version": {"type": "string", "minLength": 1}, + "dependency_hashes": { + "type": "object", + "additionalProperties": { + "type": "string", + "pattern": "^[0-9a-f]{64}$" + } + }, + "missing_dependencies": { + "type": "array", + "uniqueItems": true, + "items": { + "enum": ["schema", "udf_manifest", "parameters", "scheduler_manifest"] + } + }, + "created_at": { + "type": "string", + "minLength": 1 + } + } +} diff --git a/schemas/capabilities-v1.schema.json b/schemas/capabilities-v1.schema.json new file mode 100644 index 0000000..e632d2f --- /dev/null +++ b/schemas/capabilities-v1.schema.json @@ -0,0 +1,39 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/capabilities-v1.schema.json", + "title": "SqlGraph Capabilities Ledger v1", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "capabilities"], + "properties": { + "schema_version": {"const": "capabilities-v1"}, + "capabilities": { + "type": "array", + "minItems": 12, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "id", + "status", + "description", + "modules", + "tests", + "examples", + "limitations" + ], + "properties": { + "id": {"type": "string", "minLength": 1}, + "status": { + "enum": ["implemented", "experimental", "planned", "concept-only"] + }, + "description": {"type": "string", "minLength": 1}, + "modules": {"type": "array", "items": {"type": "string"}}, + "tests": {"type": "array", "items": {"type": "string"}}, + "examples": {"type": "array", "items": {"type": "string"}}, + "limitations": {"type": "array", "items": {"type": "string"}} + } + } + } + } +} diff --git a/schemas/evidence-bundle-v1.schema.json b/schemas/evidence-bundle-v1.schema.json new file mode 100644 index 0000000..2653f04 --- /dev/null +++ b/schemas/evidence-bundle-v1.schema.json @@ -0,0 +1,66 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/evidence-bundle-v1.schema.json", + "title": "SqlGraph Evidence Bundle v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "rule_version", + "task_id", + "baseline_id", + "intent", + "anchors", + "version", + "version_id", + "subgraph_hash", + "included_nodes", + "included_edges", + "excluded", + "supporting", + "counterevidence", + "coverage_contract", + "gaps", + "residual_unknowns", + "external_facts", + "parent_hash", + "history" + ], + "properties": { + "schema_version": {"const": "evidence-bundle-v1"}, + "rule_version": {"type": "string"}, + "task_id": {"type": "string", "minLength": 1}, + "baseline_id": {"type": "string", "minLength": 1}, + "intent": {"type": "string", "minLength": 1}, + "anchors": {"type": "array", "items": {"type": "string"}}, + "version": {"type": "integer", "minimum": 1}, + "version_id": {"type": "string", "minLength": 1}, + "subgraph_hash": {"type": "string", "pattern": "^evidence_[0-9a-f]{64}$"}, + "included_nodes": {"type": "array", "items": {"type": "string"}}, + "included_edges": {"type": "array", "items": {"type": "string"}}, + "excluded": {"type": "array", "items": {"type": "string"}}, + "supporting": {"$ref": "#/$defs/findings"}, + "counterevidence": {"$ref": "#/$defs/findings"}, + "coverage_contract": {"type": "object"}, + "gaps": {"type": "array", "items": {"type": "string"}}, + "residual_unknowns": {"type": "array", "items": {"type": "string"}}, + "external_facts": {"type": "array", "items": {"type": "string"}}, + "parent_hash": {"type": ["string", "null"]}, + "history": {"type": "array", "items": {"type": "object"}} + }, + "$defs": { + "findings": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": ["kind", "summary", "citations"], + "properties": { + "kind": {"type": "string"}, + "summary": {"type": "string"}, + "citations": {"type": "array", "items": {"type": "string"}} + } + } + } + } +} diff --git a/schemas/verification-report-v1.schema.json b/schemas/verification-report-v1.schema.json new file mode 100644 index 0000000..146d19b --- /dev/null +++ b/schemas/verification-report-v1.schema.json @@ -0,0 +1,36 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/Liyurun/SqlGraph/schemas/verification-report-v1.schema.json", + "title": "SqlGraph Verification Report v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "code", + "structure", + "runtime", + "outcome", + "closed_loop_status" + ], + "properties": { + "schema_version": {"const": "verification-report-v1"}, + "code": {"$ref": "#/$defs/layer"}, + "structure": {"$ref": "#/$defs/layer"}, + "runtime": {"$ref": "#/$defs/layer"}, + "outcome": {"enum": ["success", "failed", "incomplete"]}, + "closed_loop_status": {"enum": ["success", "failed", "incomplete"]} + }, + "$defs": { + "layer": { + "type": "object", + "additionalProperties": false, + "required": ["layer", "status", "evidence", "note"], + "properties": { + "layer": {"type": "string"}, + "status": {"enum": ["pass", "fail", "not_run"]}, + "evidence": {"type": "object"}, + "note": {"type": "string"} + } + } + } +} diff --git a/sqlgraph/actions/__init__.py b/sqlgraph/actions/__init__.py new file mode 100644 index 0000000..0042228 --- /dev/null +++ b/sqlgraph/actions/__init__.py @@ -0,0 +1,34 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public action planning, execution, and rollback interface.""" + +from sqlgraph.actions.adapters import ( + ActionAdapter, + AdapterExecutionError, + DuckDBTaskAdapter, + SqlFilePatchAdapter, +) +from sqlgraph.actions.engine import ActionEngine +from sqlgraph.actions.model import ( + ACTION_SCHEMA_VERSION, + ActionPlan, + ActionRequest, + DryRunResult, + ExecutionResult, + RollbackResult, +) + +__all__ = [ + "ACTION_SCHEMA_VERSION", + "ActionAdapter", + "ActionEngine", + "ActionPlan", + "ActionRequest", + "AdapterExecutionError", + "DryRunResult", + "DuckDBTaskAdapter", + "ExecutionResult", + "RollbackResult", + "SqlFilePatchAdapter", +] diff --git a/sqlgraph/actions/adapters.py b/sqlgraph/actions/adapters.py new file mode 100644 index 0000000..7efe81e --- /dev/null +++ b/sqlgraph/actions/adapters.py @@ -0,0 +1,175 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Built-in reversible action adapters.""" + +from __future__ import annotations + +import hashlib +import shutil +from pathlib import Path +from typing import Any, Protocol + + +class AdapterExecutionError(RuntimeError): + def __init__(self, message: str, rollback_state: dict[str, Any]): + super().__init__(message) + self.rollback_state = rollback_state + + +class ActionAdapter(Protocol): + name: str + + def dry_run(self, operations: tuple[dict[str, Any], ...]) -> tuple[str, ...]: + ... + + def execute( + self, + operations: tuple[dict[str, Any], ...], + ) -> tuple[tuple[str, ...], dict[str, Any]]: + ... + + def rollback(self, state: dict[str, Any]): + ... + + +def _sha256(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +class SqlFilePatchAdapter: + name = "sql_file_patch" + + def dry_run(self, operations: tuple[dict[str, Any], ...]) -> tuple[str, ...]: + checks = [] + for operation in operations: + path = Path(operation["path"]) + if not path.is_file(): + raise ValueError(f"SQL target does not exist: {path}") + content = path.read_text(encoding="utf-8") + if _sha256(content) != operation["expected_sha256"]: + raise ValueError(f"SQL precondition hash changed: {path}") + if content.count(operation["before"]) != 1: + raise ValueError(f"SQL patch must match exactly once: {path}") + checks.append(f"patch-ready:{path}") + return tuple(checks) + + def execute( + self, + operations: tuple[dict[str, Any], ...], + ) -> tuple[tuple[str, ...], dict[str, Any]]: + originals: dict[str, str] = {} + changed = [] + try: + for operation in operations: + path = Path(operation["path"]) + content = path.read_text(encoding="utf-8") + originals[str(path)] = content + path.write_text( + content.replace(operation["before"], operation["after"], 1), + encoding="utf-8", + ) + changed.append(str(path)) + if operation.get("inject_failure_after_apply"): + raise RuntimeError("injected failure after SQL patch") + except Exception as exc: + raise AdapterExecutionError(str(exc), {"originals": originals}) from exc + return tuple(changed), {"originals": originals} + + def rollback(self, state: dict[str, Any]): + from sqlgraph.actions.model import RollbackResult + + originals = state.get("originals", {}) + restored = [] + for raw_path, content in originals.items(): + path = Path(raw_path) + path.write_text(content, encoding="utf-8") + if path.read_text(encoding="utf-8") != content: + return RollbackResult( + status="failed", + verified=False, + restored=tuple(restored), + error=f"restore verification failed: {path}", + ) + restored.append(str(path)) + return RollbackResult( + status="success", + verified=True, + restored=tuple(restored), + ) + + +class DuckDBTaskAdapter: + name = "duckdb_tasks" + + def dry_run(self, operations: tuple[dict[str, Any], ...]) -> tuple[str, ...]: + checks = [] + for operation in operations: + database = Path(operation["database_path"]) + statements = tuple(operation.get("statements", ())) + if not database.is_file(): + raise ValueError(f"DuckDB database does not exist: {database}") + if not statements: + raise ValueError("DuckDB operation requires statements") + checks.append(f"duckdb-ready:{database}:{len(statements)}") + return tuple(checks) + + def execute( + self, + operations: tuple[dict[str, Any], ...], + ) -> tuple[tuple[str, ...], dict[str, Any]]: + import duckdb + + snapshots = {} + changed = [] + try: + for operation in operations: + database = Path(operation["database_path"]).resolve() + snapshot = Path(operation.get( + "snapshot_path", + f"{database}.sqlgraph-backup", + )).resolve() + shutil.copy2(database, snapshot) + snapshots[str(database)] = str(snapshot) + with duckdb.connect(str(database)) as connection: + connection.execute("BEGIN") + for statement in operation["statements"]: + connection.execute(statement) + if operation.get("inject_failure_after_apply"): + raise RuntimeError("injected failure after DuckDB execution") + connection.execute("COMMIT") + changed.append(str(database)) + except Exception as exc: + raise AdapterExecutionError(str(exc), {"snapshots": snapshots}) from exc + return tuple(changed), {"snapshots": snapshots} + + def rollback(self, state: dict[str, Any]): + from sqlgraph.actions.model import RollbackResult + + restored = [] + for raw_database, raw_snapshot in state.get("snapshots", {}).items(): + database = Path(raw_database) + snapshot = Path(raw_snapshot) + if not snapshot.is_file(): + return RollbackResult( + status="failed", + verified=False, + restored=tuple(restored), + error=f"snapshot missing: {snapshot}", + ) + shutil.copy2(snapshot, database) + if hashlib.sha256(database.read_bytes()).digest() != hashlib.sha256( + snapshot.read_bytes() + ).digest(): + return RollbackResult( + status="failed", + verified=False, + restored=tuple(restored), + error=f"snapshot verification failed: {database}", + ) + restored.append(str(database)) + return RollbackResult( + status="success", + verified=True, + restored=tuple(restored), + ) diff --git a/sqlgraph/actions/engine.py b/sqlgraph/actions/engine.py new file mode 100644 index 0000000..40fc3ec --- /dev/null +++ b/sqlgraph/actions/engine.py @@ -0,0 +1,138 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Policy-aware action planning and execution.""" + +from __future__ import annotations + +import json +from dataclasses import replace + +from sqlgraph.actions.adapters import ActionAdapter, AdapterExecutionError +from sqlgraph.actions.model import ( + ActionPlan, + ActionRequest, + DryRunResult, + ExecutionResult, +) +from sqlgraph.autonomy import AutonomyLevel +from sqlgraph.identity import stable_id + + +class ActionEngine: + def __init__(self, adapters: list[ActionAdapter]): + self._adapters = {adapter.name: adapter for adapter in adapters} + self._executions: dict[str, ExecutionResult] = {} + self._rollback_states: dict[str, dict] = {} + self.circuit_open = False + + def plan(self, request: ActionRequest) -> ActionPlan: + decision = request.decision + if decision.level != AutonomyLevel.L3_BOUNDED: + raise PermissionError( + f"decision {decision.level.value} does not authorize execution" + ) + if decision.requires_human_review: + raise PermissionError("human approval is required") + if decision.evidence_version != request.evidence_version: + raise PermissionError("decision and action evidence versions differ") + if request.adapter not in self._adapters: + raise ValueError(f"unknown action adapter: {request.adapter}") + payload = { + "task_id": request.task_id, + "baseline_id": request.baseline_id, + "evidence_version": request.evidence_version, + "decision_id": decision.decision_id, + "adapter": request.adapter, + "operations": request.operations, + "rollback_plan": request.rollback_plan, + } + canonical = json.dumps( + payload, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + ) + return ActionPlan( + **payload, + idempotency_key=stable_id("idem", canonical, 128), + ) + + def dry_run(self, plan: ActionPlan) -> DryRunResult: + if self.circuit_open: + return DryRunResult( + status="blocked", + checks=("circuit-open",), + ) + checks = self._adapters[plan.adapter].dry_run(plan.operations) + return DryRunResult(status="ready", checks=checks) + + def execute(self, plan: ActionPlan) -> ExecutionResult: + previous = self._executions.get(plan.idempotency_key) + if previous and previous.status == "success": + return replace(previous, status="noop") + if self.circuit_open: + return ExecutionResult( + execution_id=stable_id("execution", plan.idempotency_key, 128), + idempotency_key=plan.idempotency_key, + adapter=plan.adapter, + status="blocked", + circuit_open=True, + error="circuit is open", + ) + + adapter = self._adapters[plan.adapter] + try: + adapter.dry_run(plan.operations) + changes, rollback_state = adapter.execute(plan.operations) + except AdapterExecutionError as exc: + rollback = adapter.rollback(exc.rollback_state) + self.circuit_open = True + return ExecutionResult( + execution_id=stable_id("execution", plan.idempotency_key, 128), + idempotency_key=plan.idempotency_key, + adapter=plan.adapter, + status="rolled_back" if rollback.verified else "failed", + changes=(), + circuit_open=True, + rollback=rollback, + error=str(exc), + ) + except Exception as exc: + self.circuit_open = True + return ExecutionResult( + execution_id=stable_id("execution", plan.idempotency_key, 128), + idempotency_key=plan.idempotency_key, + adapter=plan.adapter, + status="failed", + circuit_open=True, + error=str(exc), + ) + + result = ExecutionResult( + execution_id=stable_id("execution", plan.idempotency_key, 128), + idempotency_key=plan.idempotency_key, + adapter=plan.adapter, + status="success", + changes=changes, + ) + self._executions[plan.idempotency_key] = result + self._rollback_states[result.execution_id] = rollback_state + return result + + def rollback(self, plan: ActionPlan, state: dict): + return self._adapters[plan.adapter].rollback(state) + + def rollback_execution( + self, + plan: ActionPlan, + execution: ExecutionResult, + ): + state = self._rollback_states.get(execution.execution_id) + if state is None: + raise ValueError( + f"rollback state is unavailable: {execution.execution_id}" + ) + result = self._adapters[plan.adapter].rollback(state) + self.circuit_open = True + return result diff --git a/sqlgraph/actions/model.py b/sqlgraph/actions/model.py new file mode 100644 index 0000000..81424bb --- /dev/null +++ b/sqlgraph/actions/model.py @@ -0,0 +1,72 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Action planning and execution result contracts.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from typing import Any + +from sqlgraph.autonomy import AutonomyDecision + + +ACTION_SCHEMA_VERSION = "action-plan-v1" + + +@dataclass(frozen=True) +class ActionRequest: + task_id: str + baseline_id: str + evidence_version: str + decision: AutonomyDecision + adapter: str + operations: tuple[dict[str, Any], ...] + rollback_plan: tuple[dict[str, Any], ...] = () + + +@dataclass(frozen=True) +class ActionPlan: + task_id: str + baseline_id: str + evidence_version: str + decision_id: str + adapter: str + operations: tuple[dict[str, Any], ...] + rollback_plan: tuple[dict[str, Any], ...] + idempotency_key: str + schema_version: str = ACTION_SCHEMA_VERSION + + def to_dict(self) -> dict: + return asdict(self) + + +@dataclass(frozen=True) +class DryRunResult: + status: str + checks: tuple[str, ...] = () + details: dict[str, Any] = field(default_factory=dict) + + +@dataclass(frozen=True) +class RollbackResult: + status: str + verified: bool + restored: tuple[str, ...] = () + error: str = "" + + +@dataclass(frozen=True) +class ExecutionResult: + execution_id: str + idempotency_key: str + adapter: str + status: str + changes: tuple[str, ...] = () + circuit_open: bool = False + rollback: RollbackResult | None = None + error: str = "" + + def to_dict(self) -> dict: + result = asdict(self) + return result diff --git a/sqlgraph/agent/__init__.py b/sqlgraph/agent/__init__.py new file mode 100644 index 0000000..bed7a66 --- /dev/null +++ b/sqlgraph/agent/__init__.py @@ -0,0 +1,161 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Compatibility exports for the governance reasoning module.""" + +from dataclasses import dataclass, field + +from sqlgraph.autonomy import AutonomyLevel, decide_autonomy +from sqlgraph.evidence import assess_sufficiency, build_evidence_subgraph +from sqlgraph.lineage import drilldown +from sqlgraph.reasoning import ( + STEPS, + GovernanceRequest, + GovernanceResult, + GovernanceRunner, +) +from sqlgraph.verify import verify + + +@dataclass +class StepRecord: + step: str + evidence_version: str + detail: dict = field(default_factory=dict) + transition: str | None = None + + +@dataclass +class AuditTrail: + task_id: str + action_type: str + records: list[StepRecord] = field(default_factory=list) + decision: dict | None = None + verify_report: dict | None = None + outcome: str = "pending" + + def to_dict(self) -> dict: + return { + "task_id": self.task_id, + "action_type": self.action_type, + "steps": [ + { + "step": record.step, + "evidence_version": record.evidence_version, + "detail": record.detail, + "transition": record.transition, + } + for record in self.records + ], + "decision": self.decision, + "verify_report": self.verify_report, + "outcome": self.outcome, + } + + def replay(self) -> list[str]: + return [ + f"{record.step}@{record.evidence_version}" + + (f" [{record.transition}]" if record.transition else "") + for record in self.records + ] + + +class GovernanceLoop: + """Compatibility facade for the original graph-only demonstration.""" + + def __init__(self, graph): + self.graph = graph + + def run( + self, + task_id, + source_table, + target_table, + action, + *, + governance_issue=None, + runtime_observed=None, + runtime_target=None, + ): + evidence = build_evidence_subgraph( + self.graph, + task_id, + [source_table, target_table], + ) + sufficiency = assess_sufficiency(evidence) + lineage = drilldown(self.graph, source_table, target_table) + decision = decide_autonomy(action) + executed = ( + decision.level == AutonomyLevel.L3_BOUNDED + and not decision.requires_human_review + and sufficiency.sufficient + and lineage.get("found") + ) + report = verify( + self.graph, + source_table, + target_table, + governance_issue=governance_issue, + runtime_observed=runtime_observed if executed else None, + runtime_target=runtime_target, + ) + raw_report = report.to_dict() + compatibility_report = { + "structure": raw_report["structure"], + "contract": raw_report["code"], + "runtime": raw_report["runtime"], + "closed_loop_status": raw_report["closed_loop_status"], + } + trail = AuditTrail(task_id, action.action_type) + version = evidence.version_id + details = ( + ("observe", {"sufficiency": sufficiency.action}, None), + ("explain", {"grounded": bool(lineage.get("found"))}, None), + ("propose", {"proposed_action": action.action_type}, None), + ( + "authorize", + decision.to_dict(), + "cognition->action", + ), + ( + "execute", + {"executed": executed}, + "authorized->external_execute" if executed else "halt->human_review", + ), + ( + "verify", + {"closed_loop": compatibility_report["closed_loop_status"]}, + "execute->reobserve" if executed else None, + ), + ( + "learn", + {"writeback": "signals_only"}, + None, + ), + ) + trail.records.extend( + StepRecord(step, version, detail, transition) + for step, detail, transition in details + ) + trail.decision = decision.to_dict() + trail.verify_report = compatibility_report + trail.outcome = ( + ( + "closed" + if compatibility_report["closed_loop_status"] == "success" + else compatibility_report["closed_loop_status"] + ) + if executed + else "held_for_human_review" + ) + return trail + +__all__ = [ + "STEPS", + "GovernanceRequest", + "GovernanceResult", + "GovernanceRunner", + "AuditTrail", + "GovernanceLoop", + "StepRecord", +] diff --git a/sqlgraph/analyze/anomaly.py b/sqlgraph/analyze/anomaly.py index 2434e50..563334b 100644 --- a/sqlgraph/analyze/anomaly.py +++ b/sqlgraph/analyze/anomaly.py @@ -372,6 +372,7 @@ def _model_metric( raw_scores, normalized_scores, labels, + strict=False, ): record["model_anomaly_score"] = normalized record["model_raw_score"] = round(raw_score, config.float_precision) diff --git a/sqlgraph/analyze/embeddings.py b/sqlgraph/analyze/embeddings.py index 9003076..1c5f36a 100644 --- a/sqlgraph/analyze/embeddings.py +++ b/sqlgraph/analyze/embeddings.py @@ -16,7 +16,7 @@ import hashlib from importlib import import_module import math -import random +import random # allow-random: deterministic analysis sampling uses an explicit seed from typing import Any, Mapping from sqlgraph.analyze.config import AnalysisConfig @@ -188,7 +188,10 @@ def cosine_similarity( """Return cosine similarity for two normalized-ish vectors.""" if left is None or right is None or len(left) != len(right): return None - dot = sum(float(a) * float(b) for a, b in zip(left, right)) + dot = sum( + float(a) * float(b) + for a, b in zip(left, right, strict=False) + ) left_norm = math.sqrt(sum(float(item) * float(item) for item in left)) right_norm = math.sqrt(sum(float(item) * float(item) for item in right)) if left_norm == 0.0 or right_norm == 0.0: diff --git a/sqlgraph/analyze/impact.py b/sqlgraph/analyze/impact.py index 3e0cd5c..ebd872b 100644 --- a/sqlgraph/analyze/impact.py +++ b/sqlgraph/analyze/impact.py @@ -8,7 +8,7 @@ from collections import Counter, deque from dataclasses import dataclass, field import math -import random +import random # allow-random: deterministic analysis sampling uses an explicit seed from typing import Any, Mapping from sqlgraph.analyze.config import AnalysisConfig diff --git a/sqlgraph/analyze/motifs.py b/sqlgraph/analyze/motifs.py index 83e5231..47d9f31 100644 --- a/sqlgraph/analyze/motifs.py +++ b/sqlgraph/analyze/motifs.py @@ -7,7 +7,7 @@ from collections import Counter, deque from dataclasses import dataclass, field -import random +import random # allow-random: deterministic analysis sampling uses an explicit seed from typing import Any, Mapping, Sequence from sqlgraph.analyze.config import AnalysisConfig diff --git a/sqlgraph/analyze/table_graph.py b/sqlgraph/analyze/table_graph.py index 978fd69..c3cf9af 100644 --- a/sqlgraph/analyze/table_graph.py +++ b/sqlgraph/analyze/table_graph.py @@ -38,6 +38,9 @@ class TableGraphEdge: target_id: str field_weight: int = 0 sql_weight: int = 0 + statement_refs: tuple[tuple[str, int], ...] = () + column_dependency_ids: tuple[str, ...] = () + transform_ids: tuple[str, ...] = () @property def total_weight(self) -> int: @@ -117,6 +120,9 @@ def to_dict(self) -> dict[str, Any]: "target_id": e.target_id, "field_weight": e.field_weight, "sql_weight": e.sql_weight, + "statement_refs": [list(ref) for ref in e.statement_refs], + "column_dependency_ids": list(e.column_dependency_ids), + "transform_ids": list(e.transform_ids), } for e in self.edges ], @@ -177,6 +183,8 @@ def build_table_graph(view: AnalysisView) -> TableGraph: # ── 3. Field weight ───────────────────────────────────────────────── field_weight: dict[tuple[str, str], int] = {} + field_edge_ids: dict[tuple[str, str], set[str]] = {} + transform_ids: dict[tuple[str, str], set[str]] = {} # Index: transform_id → output column_id transform_output: dict[str, str] = {} @@ -203,6 +211,7 @@ def build_table_graph(view: AnalysisView) -> TableGraph: if tgt_table is not None and tgt_table != src_table: key = (src_table, tgt_table) field_weight[key] = field_weight.get(key, 0) + 1 + field_edge_ids.setdefault(key, set()).add(str(edge["id"])) elif tgt_type == "transform": # Transform: source_column → transform → output_column @@ -212,31 +221,37 @@ def build_table_graph(view: AnalysisView) -> TableGraph: if tgt_table is not None and tgt_table != src_table: key = (src_table, tgt_table) field_weight[key] = field_weight.get(key, 0) + 1 + field_edge_ids.setdefault(key, set()).add(str(edge["id"])) + transform_ids.setdefault(key, set()).add(tgt_id) # ── 4. SQL weight ─────────────────────────────────────────────────── sql_weight: dict[tuple[str, str], int] = {} - sql_reads: dict[str, set[str]] = {} - sql_writes: dict[str, set[str]] = {} + sql_reads: dict[tuple[str, int], set[str]] = {} + sql_writes: dict[tuple[str, int], set[str]] = {} + statement_refs: dict[tuple[str, str], set[tuple[str, int]]] = {} for edge in view.iter_edges("reads_from"): sql_id = str(edge["source"]) + statement_key = (sql_id, int(edge.get("stmt_index", 0))) table_id = str(edge["target"]) if table_id in physical: - sql_reads.setdefault(sql_id, set()).add(table_id) + sql_reads.setdefault(statement_key, set()).add(table_id) for edge in view.iter_edges("writes_to"): sql_id = str(edge["source"]) + statement_key = (sql_id, int(edge.get("stmt_index", 0))) table_id = str(edge["target"]) if table_id in physical: - sql_writes.setdefault(sql_id, set()).add(table_id) + sql_writes.setdefault(statement_key, set()).add(table_id) - for sql_id, reads in sql_reads.items(): - writes = sql_writes.get(sql_id, set()) + for statement_key, reads in sql_reads.items(): + writes = sql_writes.get(statement_key, set()) for src_tbl in reads: for tgt_tbl in writes: if src_tbl != tgt_tbl: key = (src_tbl, tgt_tbl) sql_weight[key] = sql_weight.get(key, 0) + 1 + statement_refs.setdefault(key, set()).add(statement_key) # ── 5. Seed edge set from TABLE_LINEAGE and overlay weights ───────── edge_registry: dict[tuple[str, str], dict[str, int]] = {} @@ -248,6 +263,12 @@ def build_table_graph(view: AnalysisView) -> TableGraph: edge_registry.setdefault( (src_id, tgt_id), {"field_weight": 0, "sql_weight": 0} ) + for ref in edge.get("provenance", ()): + sql_id = ref.get("sql_id") + if sql_id: + statement_refs.setdefault((src_id, tgt_id), set()).add( + (str(sql_id), int(ref.get("stmt_index", 0))) + ) # Overlay field weights (may introduce edges not in TABLE_LINEAGE) for (src, tgt), fw in field_weight.items(): @@ -279,6 +300,9 @@ def build_table_graph(view: AnalysisView) -> TableGraph: target_id=tgt, field_weight=weights["field_weight"], sql_weight=weights["sql_weight"], + statement_refs=tuple(sorted(statement_refs.get((src, tgt), set()))), + column_dependency_ids=tuple(sorted(field_edge_ids.get((src, tgt), set()))), + transform_ids=tuple(sorted(transform_ids.get((src, tgt), set()))), ) edge_list.append(e) adjacency_builder.setdefault(src, []).append(e) diff --git a/sqlgraph/audit/__init__.py b/sqlgraph/audit/__init__.py new file mode 100644 index 0000000..da6daea --- /dev/null +++ b/sqlgraph/audit/__init__.py @@ -0,0 +1,20 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public append-only audit interface.""" + +from sqlgraph.audit.log import AuditLog +from sqlgraph.audit.model import ( + AUDIT_SCHEMA_VERSION, + AuditEvent, + IntegrityReport, + ReplayResult, +) + +__all__ = [ + "AUDIT_SCHEMA_VERSION", + "AuditEvent", + "AuditLog", + "IntegrityReport", + "ReplayResult", +] diff --git a/sqlgraph/audit/log.py b/sqlgraph/audit/log.py new file mode 100644 index 0000000..192a63c --- /dev/null +++ b/sqlgraph/audit/log.py @@ -0,0 +1,109 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""JSONL audit log with a verifiable hash chain.""" + +from __future__ import annotations + +import hashlib +import json +import os +from dataclasses import replace +from datetime import datetime, timezone +from pathlib import Path + +from sqlgraph.audit.model import ( + GENESIS_HASH, + AuditEvent, + IntegrityReport, + ReplayResult, +) +from sqlgraph.identity import stable_id + + +def _canonical(payload: dict) -> str: + return json.dumps( + payload, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + ) + + +def _event_hash(payload: dict) -> str: + content = {key: value for key, value in payload.items() if key != "event_hash"} + return hashlib.sha256(_canonical(content).encode("utf-8")).hexdigest() + + +class AuditLog: + def __init__(self, path: str | Path): + self.path = Path(path) + + def append(self, event: AuditEvent) -> AuditEvent: + current = self._read() + sequence = len(current) + 1 + previous_hash = current[-1].event_hash if current else GENESIS_HASH + timestamp = datetime.now(timezone.utc).isoformat() + event_id = stable_id( + "event", + f"{event.task_id}:{sequence}:{event.event_type}:{previous_hash}", + 128, + ) + prepared = replace( + event, + sequence=sequence, + timestamp=timestamp, + previous_hash=previous_hash, + event_id=event_id, + event_hash="", + ) + completed = replace( + prepared, + event_hash=_event_hash(prepared.to_dict()), + ) + self.path.parent.mkdir(parents=True, exist_ok=True) + with self.path.open("a", encoding="utf-8") as stream: + stream.write(_canonical(completed.to_dict()) + "\n") + stream.flush() + os.fsync(stream.fileno()) + return completed + + def verify_integrity(self) -> IntegrityReport: + errors = [] + previous_hash = GENESIS_HASH + try: + events = self._read() + except (json.JSONDecodeError, TypeError, ValueError) as exc: + return IntegrityReport(False, 0, (f"invalid audit JSON: {exc}",)) + + for expected_sequence, event in enumerate(events, start=1): + if event.sequence != expected_sequence: + errors.append( + f"sequence {event.sequence} expected {expected_sequence}" + ) + if event.previous_hash != previous_hash: + errors.append( + f"event {event.sequence} previous hash does not match" + ) + calculated = _event_hash(event.to_dict()) + if calculated != event.event_hash: + errors.append(f"event {event.sequence} hash does not match") + previous_hash = event.event_hash + return IntegrityReport(not errors, len(events), tuple(errors)) + + def replay(self, task_id: str) -> ReplayResult: + integrity = self.verify_integrity() + events = tuple( + event for event in self._read() if event.task_id == task_id + ) + return ReplayResult(task_id, events, integrity) + + def _read(self) -> list[AuditEvent]: + if not self.path.exists(): + return [] + events = [] + with self.path.open(encoding="utf-8") as stream: + for line in stream: + if line.strip(): + events.append(AuditEvent(**json.loads(line))) + return events diff --git a/sqlgraph/audit/model.py b/sqlgraph/audit/model.py new file mode 100644 index 0000000..112ce91 --- /dev/null +++ b/sqlgraph/audit/model.py @@ -0,0 +1,50 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Append-only governance audit contracts.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from typing import Any + + +AUDIT_SCHEMA_VERSION = "audit-event-v1" +GENESIS_HASH = "0" * 64 + + +@dataclass(frozen=True) +class AuditEvent: + task_id: str + baseline_id: str + event_type: str + step: str + payload: dict[str, Any] = field(default_factory=dict) + evidence_version: str = "" + policy_version: str = "" + authorization_identity: str = "" + idempotency_key: str = "" + transition: str = "" + sequence: int = 0 + timestamp: str = "" + previous_hash: str = GENESIS_HASH + event_hash: str = "" + event_id: str = "" + schema_version: str = AUDIT_SCHEMA_VERSION + + def to_dict(self) -> dict: + return asdict(self) + + +@dataclass(frozen=True) +class IntegrityReport: + valid: bool + event_count: int + errors: tuple[str, ...] = () + + +@dataclass(frozen=True) +class ReplayResult: + task_id: str + events: tuple[AuditEvent, ...] + integrity: IntegrityReport diff --git a/sqlgraph/autonomy/__init__.py b/sqlgraph/autonomy/__init__.py new file mode 100644 index 0000000..5206d2a --- /dev/null +++ b/sqlgraph/autonomy/__init__.py @@ -0,0 +1,24 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public autonomy policy interface.""" + +from sqlgraph.autonomy.decision import ( + POLICY_VERSION, + AuthorizationScope, + AutonomyDecision, + AutonomyLevel, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) + +__all__ = [ + "POLICY_VERSION", + "AuthorizationScope", + "AutonomyDecision", + "AutonomyLevel", + "GovernanceAction", + "ReversibilityEvidence", + "decide_autonomy", +] diff --git a/sqlgraph/autonomy/decision.py b/sqlgraph/autonomy/decision.py new file mode 100644 index 0000000..5c7cfab --- /dev/null +++ b/sqlgraph/autonomy/decision.py @@ -0,0 +1,179 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Evidence-bound autonomy decisions with pre-emptive safety gates.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from enum import Enum + +from sqlgraph.identity import stable_id + + +POLICY_VERSION = "autonomy-v1" +BLAST_RADIUS_REVIEW_THRESHOLD = 0.7 +OBJECT_RISK_REVIEW_THRESHOLD = 0.7 +HISTORICAL_RELIABILITY_MIN = 0.3 + + +class AutonomyLevel(str, Enum): + L0_OBSERVE = "L0" + L1_EXPLAIN = "L1" + L2_PROPOSE = "L2" + L3_BOUNDED = "L3" + L4_APPROVED = "L4" + L5_CONTINUOUS = "L5" + + +class AuthorizationScope(str, Enum): + NONE = "none" + SINGLE_L3 = "single_l3" + CONTINUOUS_L5 = "continuous_l5" + + +@dataclass(frozen=True) +class ReversibilityEvidence: + state_restorable: bool + external_effects_controlled: bool + rollback_verified: bool + references: tuple[str, ...] = () + + @property + def verified(self) -> bool: + return ( + self.state_restorable + and self.external_effects_controlled + and self.rollback_verified + ) + + +@dataclass(frozen=True) +class GovernanceAction: + action_type: str + evidence_version: str + evidence_grounded: bool + reversibility: ReversibilityEvidence + authorization_scope: AuthorizationScope = AuthorizationScope.NONE + authorization_identity: str = "" + blast_radius: float = 0.0 + object_risk: float = 0.0 + historical_reliability: float = 1.0 + roi: float = 0.0 + + def __post_init__(self) -> None: + for field_name in ( + "blast_radius", + "object_risk", + "historical_reliability", + ): + value = getattr(self, field_name) + if not 0.0 <= value <= 1.0: + raise ValueError(f"{field_name} must be between 0 and 1") + + +@dataclass(frozen=True) +class AutonomyDecision: + level: AutonomyLevel + requires_human_review: bool + evidence_version: str + reasons: tuple[str, ...] = () + reversibility_veto: bool = False + evidence_veto: bool = False + authorization_veto: bool = False + scoring_performed: bool = False + within_l5_scope: bool = False + policy_version: str = POLICY_VERSION + decision_id: str = field(init=False) + + def __post_init__(self) -> None: + key = "|".join(( + self.policy_version, + self.evidence_version, + self.level.value, + str(self.requires_human_review), + str(self.reversibility_veto), + str(self.evidence_veto), + str(self.authorization_veto), + *self.reasons, + )) + object.__setattr__(self, "decision_id", stable_id("decision", key, 128)) + + def to_dict(self) -> dict: + result = asdict(self) + result["level"] = self.level.value + return result + + +def decide_autonomy(action: GovernanceAction) -> AutonomyDecision: + reasons = [] + + if not action.reversibility.verified: + reasons.append( + "reversibility gate failed: state recovery, external effects, " + "and verified rollback are all required" + ) + return AutonomyDecision( + level=AutonomyLevel.L4_APPROVED, + requires_human_review=True, + evidence_version=action.evidence_version, + reasons=tuple(reasons), + reversibility_veto=True, + ) + + reasons.append("reversibility gate passed") + if not action.evidence_grounded or not action.evidence_version: + reasons.append( + "evidence gate failed: a grounded, versioned evidence bundle is required" + ) + return AutonomyDecision( + level=AutonomyLevel.L0_OBSERVE, + requires_human_review=True, + evidence_version=action.evidence_version, + reasons=tuple(reasons), + evidence_veto=True, + ) + + reasons.append("evidence gate passed") + if action.authorization_scope == AuthorizationScope.NONE: + reasons.append("authorization gate stopped execution at proposal") + return AutonomyDecision( + level=AutonomyLevel.L2_PROPOSE, + requires_human_review=True, + evidence_version=action.evidence_version, + reasons=tuple(reasons), + authorization_veto=True, + ) + + review_reasons = [] + if action.blast_radius >= BLAST_RADIUS_REVIEW_THRESHOLD: + review_reasons.append("blast radius requires per-action approval") + if action.object_risk >= OBJECT_RISK_REVIEW_THRESHOLD: + review_reasons.append("object risk requires per-action approval") + if action.historical_reliability < HISTORICAL_RELIABILITY_MIN: + review_reasons.append("historical reliability is below the safe threshold") + reasons.extend(review_reasons) + reasons.append(f"roi={action.roi} evaluated only after all hard gates") + + if review_reasons: + return AutonomyDecision( + level=AutonomyLevel.L4_APPROVED, + requires_human_review=True, + evidence_version=action.evidence_version, + reasons=tuple(reasons), + scoring_performed=True, + ) + + within_l5 = action.authorization_scope == AuthorizationScope.CONTINUOUS_L5 + reasons.append( + "action is allowed as bounded L3 execution" + + (" inside an L5 scope" if within_l5 else "") + ) + return AutonomyDecision( + level=AutonomyLevel.L3_BOUNDED, + requires_human_review=False, + evidence_version=action.evidence_version, + reasons=tuple(reasons), + scoring_performed=True, + within_l5_scope=within_l5, + ) diff --git a/sqlgraph/baseline/__init__.py b/sqlgraph/baseline/__init__.py new file mode 100644 index 0000000..c651fca --- /dev/null +++ b/sqlgraph/baseline/__init__.py @@ -0,0 +1,14 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public baseline registry interface.""" + +from sqlgraph.baseline.builder import SCHEMA_VERSION, build_baseline +from sqlgraph.baseline.model import BaselineManifest, SourceHash + +__all__ = [ + "BaselineManifest", + "SCHEMA_VERSION", + "SourceHash", + "build_baseline", +] diff --git a/sqlgraph/baseline/builder.py b/sqlgraph/baseline/builder.py new file mode 100644 index 0000000..e7ddac9 --- /dev/null +++ b/sqlgraph/baseline/builder.py @@ -0,0 +1,118 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Build deterministic manifests for all governance inputs.""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import asdict, is_dataclass +from datetime import datetime, timezone +from typing import Any, Mapping + +import sqlglot + +from sqlgraph.baseline.model import BaselineManifest, SourceHash +from sqlgraph.identity import IDENTITY_RULE_VERSION +from sqlgraph.input import SqlSource +from sqlgraph.input.csv_schema import SchemaRegistry + + +SCHEMA_VERSION = "baseline-manifest-v1" +_DEPENDENCIES = ("schema", "udf_manifest", "parameters", "scheduler_manifest") + + +def _sha256_bytes(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def _canonical_json(value: Any) -> str: + return json.dumps( + value, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + ) + + +def _plain(value: Any) -> Any: + if value is None: + return None + if isinstance(value, SchemaRegistry): + return value.to_dict() + if is_dataclass(value): + return asdict(value) + if isinstance(value, Mapping): + return dict(value) + return value + + +def _hash_dependency(value: Any) -> str: + return _sha256_bytes(_canonical_json(_plain(value)).encode("utf-8")) + + +def build_baseline( + source: SqlSource, + *, + dialect: str | None = None, + schema: SchemaRegistry | None = None, + parameters: Mapping[str, object] | None = None, + udf_manifest: Mapping[str, object] | None = None, + scheduler_manifest: Mapping[str, object] | None = None, +) -> BaselineManifest: + """Build a stable baseline ID while keeping creation time informational.""" + source_hashes = tuple(sorted( + ( + SourceHash( + name=item.name, + source_type=item.source_type, + source_path=item.source_path or "", + sha256=_sha256_bytes(item.content.encode("utf-8")), + ) + for item in source + ), + key=lambda item: ( + item.source_path, + item.name, + item.source_type, + item.sha256, + ), + )) + dependencies = { + "schema": schema, + "udf_manifest": udf_manifest, + "parameters": parameters, + "scheduler_manifest": scheduler_manifest, + } + dependency_hashes = { + name: _hash_dependency(value) + for name, value in dependencies.items() + if value is not None + } + missing = tuple( + name for name in _DEPENDENCIES if dependencies[name] is None + ) + identity_payload = { + "schema_version": SCHEMA_VERSION, + "sources": [asdict(item) for item in source_hashes], + "dialect": dialect or "", + "parser_version": f"sqlglot-{sqlglot.__version__}", + "identity_rule_version": IDENTITY_RULE_VERSION, + "dependency_hashes": dependency_hashes, + "missing_dependencies": missing, + } + baseline_id = "base_" + _sha256_bytes( + _canonical_json(identity_payload).encode("utf-8") + ) + return BaselineManifest( + schema_version=SCHEMA_VERSION, + baseline_id=baseline_id, + source_hashes=source_hashes, + dialect=dialect or "", + parser_version=f"sqlglot-{sqlglot.__version__}", + identity_rule_version=IDENTITY_RULE_VERSION, + dependency_hashes=dependency_hashes, + missing_dependencies=missing, + created_at=datetime.now(timezone.utc).isoformat(), + ) diff --git a/sqlgraph/baseline/model.py b/sqlgraph/baseline/model.py new file mode 100644 index 0000000..81a4d53 --- /dev/null +++ b/sqlgraph/baseline/model.py @@ -0,0 +1,43 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Versioned input baseline contracts.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass +from typing import Mapping + + +@dataclass(frozen=True, order=True) +class SourceHash: + name: str + source_type: str + source_path: str + sha256: str + + +@dataclass(frozen=True) +class BaselineManifest: + schema_version: str + baseline_id: str + source_hashes: tuple[SourceHash, ...] + dialect: str + parser_version: str + identity_rule_version: str + dependency_hashes: Mapping[str, str] + missing_dependencies: tuple[str, ...] + created_at: str + + def to_dict(self) -> dict: + return { + "schema_version": self.schema_version, + "baseline_id": self.baseline_id, + "source_hashes": [asdict(item) for item in self.source_hashes], + "dialect": self.dialect, + "parser_version": self.parser_version, + "identity_rule_version": self.identity_rule_version, + "dependency_hashes": dict(self.dependency_hashes), + "missing_dependencies": list(self.missing_dependencies), + "created_at": self.created_at, + } diff --git a/sqlgraph/builder/graph_builder.py b/sqlgraph/builder/graph_builder.py index 6f504aa..00a626f 100644 --- a/sqlgraph/builder/graph_builder.py +++ b/sqlgraph/builder/graph_builder.py @@ -3,10 +3,7 @@ # sqlgraph/builder/graph_builder.py from __future__ import annotations -import uuid -import hashlib import re -from typing import Optional from sqlgraph.model import ( PropertyGraph, SqlNode, TableNode, ColumnNode, TransformNode, Edge, EdgeType, ExpressionType, @@ -15,20 +12,18 @@ from sqlgraph.input.sql_source import SqlSource from sqlgraph.input.csv_schema import SchemaRegistry from sqlgraph.builder.table_registry import TableRegistry +from sqlgraph.identity import ( + CollisionRegistry, + edge_id, + environment_fingerprint, + node_id, +) from sqlgraph.utils.logging import log_info, log_warn -def _gen_id(prefix: str) -> str: - return f"{prefix}_{uuid.uuid4().hex[:10]}" - - def _det_id(prefix: str, key: str) -> str: - """确定性 id:由内容 key 生成,保证相同实体在多次运行/多条 SQL 间稳定 - - 取 24 个十六进制字符(96 bit)。在 160 万级表/字段规模下,40 bit 空间存在 - 实际可感知的碰撞风险,96 bit 可将碰撞概率降到可忽略。 - """ - return f"{prefix}_{hashlib.sha1(key.encode('utf-8')).hexdigest()[:24]}" + """Derive a stable graph identifier from a semantic key.""" + return node_id(prefix, key) def _expr_type_from_str(t: str) -> ExpressionType: @@ -72,21 +67,63 @@ def __init__(self, dialect: str | None = None, schema_registry: SchemaRegistry | self._expr_nodes: dict = {} self._edge_seen: set = set() self._current_ctes: list = [] - - def _add_edge_dedup(self, source_id: str, target_id: str, edge_type: EdgeType, **props) -> None: + self._collisions = CollisionRegistry() + self._coverage = { + "total": 0, + "ok": 0, + "partial": 0, + "failed": 0, + "reasons": [], + } + + def _add_edge_dedup( + self, + source_id: str, + target_id: str, + edge_type: EdgeType, + *, + context: str = "", + **props, + ) -> None: """按 (source,target,type) 去重后添加边,避免共享节点导致重复边""" - key = (source_id, target_id, edge_type) + key = (source_id, target_id, edge_type, context) if key in self._edge_seen: return self._edge_seen.add(key) self.graph.add_edge(Edge( - id=_gen_id("e"), + id=edge_id(source_id, target_id, edge_type, context), source_id=source_id, target_id=target_id, edge_type=edge_type, properties=props or {}, )) + def _add_read_write_edge( + self, + sql_id: str, + table_id: str, + edge_type: EdgeType, + stmt_index: int, + ) -> None: + self._add_edge_dedup( + sql_id, + table_id, + edge_type, + context=f"statement:{stmt_index}", + stmt_index=stmt_index, + ) + + def _finalize_metadata(self) -> None: + import sqlglot + + self.graph.metadata["environment"] = environment_fingerprint( + self.dialect, + f"sqlglot-{sqlglot.__version__}", + ) + self.graph.metadata["coverage"] = dict(self._coverage) + if self._collisions.has_collision(): + self.graph.metadata["id_collisions"] = list(self._collisions.collisions) + def build_from_source(self, source: SqlSource) -> PropertyGraph: """从 SqlSource 构建完整图""" log_info(f"Building graph from {len(source)} SQL source(s)") @@ -94,17 +131,23 @@ def build_from_source(self, source: SqlSource) -> PropertyGraph: df_failed = 0 df_failed_samples = [] for item in source: + self._coverage["total"] += 1 try: parse_result = self.parser.parse( item.content, name=item.name, file_path=item.source_path ) except Exception as e: + self._coverage["failed"] += 1 + self._coverage["reasons"].append( + {"name": item.name, "error": str(e)[:200]} + ) if item.source_type == "df_csv": df_failed += 1 if len(df_failed_samples) < 5: df_failed_samples.append(f"{item.name}: {e}") continue raise + self._coverage["ok"] += 1 self._apply_source_metadata(parse_result, item) results.append(parse_result) self._add_parse_result(parse_result, item) @@ -114,14 +157,18 @@ def build_from_source(self, source: SqlSource) -> PropertyGraph: f"Skipped {df_failed} df_csv SQL item(s) due to parse errors. " f"Samples: {' | '.join(df_failed_samples)}" ) + self._finalize_metadata() log_info(f"Graph built: {self.graph.stats()}") return self.graph def build_from_sql(self, sql: str, name: str = "query") -> PropertyGraph: """从单条 SQL 字符串构建图""" + self._coverage["total"] += 1 result = self.parser.parse(sql, name=name) + self._coverage["ok"] += 1 self._add_parse_result(result) self._link_cross_sql_lineage() + self._finalize_metadata() return self.graph def _apply_source_metadata(self, result: SqlParseResult, item) -> None: @@ -129,7 +176,11 @@ def _apply_source_metadata(self, result: SqlParseResult, item) -> None: meta = getattr(item, "metadata", {}) or {} content_hash = meta.get("content_hash") if content_hash: - result.sql_id = f"sql_{content_hash[:12]}" + source_uri = meta.get("source_uri") or result.sql_name + result.sql_id = node_id( + "sql", + f"{source_uri}::{self.dialect or ''}::{content_hash}", + ) raw_content = meta.get("raw_content") if raw_content: result.sql_content = raw_content @@ -151,23 +202,45 @@ def _add_parse_result(self, result: SqlParseResult, item=None) -> None: ) self.graph.add_node(sql_node) - for src in result.source_tables: - tname = _canonical_table_name(src["name"]) - if src.get("is_cte"): - continue - tid = self._ensure_table_node(tname, is_cte=False) - self._add_edge_dedup(result.sql_id, tid, EdgeType.READS_FROM) + statement_io = result.statement_io or [{ + "stmt_index": 0, + "sources": result.source_tables, + "targets": result.target_tables, + }] + for statement in statement_io: + stmt_index = statement["stmt_index"] + for src in statement["sources"]: + tname = _canonical_table_name(src["name"]) + if src.get("is_cte"): + continue + tid = self._ensure_table_node(tname, is_cte=False) + self._add_read_write_edge( + result.sql_id, + tid, + EdgeType.READS_FROM, + stmt_index, + ) - for tgt in result.target_tables: - tname = _canonical_table_name(tgt["name"]) - tid = self._ensure_table_node(tname, is_cte=False) - self._add_edge_dedup(result.sql_id, tid, EdgeType.WRITES_TO) - self.table_registry.register_producer(tname, result.sql_id, tid) + for tgt in statement["targets"]: + tname = _canonical_table_name(tgt["name"]) + tid = self._ensure_table_node(tname, is_cte=False) + self._add_read_write_edge( + result.sql_id, + tid, + EdgeType.WRITES_TO, + stmt_index, + ) + self.table_registry.register_producer(tname, result.sql_id, tid) if not result.target_tables and any(col.get("table") is None for col in result.columns): tname = _result_table_name(result.sql_name) tid = self._ensure_table_node(tname, is_cte=False) - self._add_edge_dedup(result.sql_id, tid, EdgeType.WRITES_TO) + self._add_read_write_edge( + result.sql_id, + tid, + EdgeType.WRITES_TO, + 0, + ) self.table_registry.register_producer(tname, result.sql_id, tid) for cte in result.cte_tables: @@ -217,6 +290,7 @@ def _ensure_table_node( ) if is_cte: node.name = table_name + self._collisions.register(node.id, f"table:{table_name}") self.graph.add_node(node) self._table_nodes[table_name] = node.id return node.id @@ -231,6 +305,7 @@ def _ensure_column_node(self, table_id: str, col_name: str, table_name_hint: str name=col_name, table_id=table_id, ) + self._collisions.register(node.id, f"column:{table_id}.{col_name}") self.graph.add_node(node) self._column_nodes[key] = node.id self._add_edge_dedup(table_id, node.id, EdgeType.HAS_COLUMN) @@ -269,6 +344,7 @@ def _ensure_expr_node(self, info: dict, output_name: str) -> str: op=info.get("op", ""), output_name=output_name, ) + self._collisions.register(node.id, f"transform:{merge_key}") self.graph.add_node(node) self._expr_nodes[merge_key] = node.id return node_id @@ -328,36 +404,94 @@ def _add_column_dependency(self, sql_id: str, col_info: dict) -> None: for out_col_id in out_col_ids: self._add_edge_dedup(expr_node_id, out_col_id, EdgeType.PRODUCES) + self._add_operand_edges(col_info, root_fp, expr_node_id, col_name) + + def _add_operand_edges( + self, + col_info: dict, + root_fingerprint: str, + root_node_id: str, + output_name: str, + ) -> None: + operand_nodes = col_info.get("operand_nodes") or {} + operand_edges = col_info.get("operand_edges") or [] + fingerprint_to_id = {root_fingerprint: root_node_id} + + def node_for(fingerprint: str) -> str: + if fingerprint in fingerprint_to_id: + return fingerprint_to_id[fingerprint] + info = operand_nodes[fingerprint] + node = self._ensure_expr_node( + info, + f"{output_name}::operand::{fingerprint}", + ) + fingerprint_to_id[fingerprint] = node + for physical_column in info.get("source_columns", []): + source = self._ensure_physical_column(physical_column) + if source: + self._add_edge_dedup( + source, + node, + EdgeType.COMPUTE_DEPENDENCY, + ) + return node + + for child_fingerprint, parent_fingerprint in operand_edges: + child = node_for(child_fingerprint) + parent = node_for(parent_fingerprint) + if child != parent: + self._add_edge_dedup(child, parent, EdgeType.EXPR_OPERAND) + def _link_cross_sql_lineage(self) -> None: """建立跨 SQL 的表级血缘边 (src_table -> dst_table)""" - existing = set() + existing = {} for edge in self.graph.edges: if edge.edge_type == EdgeType.TABLE_LINEAGE: - existing.add((edge.source_id, edge.target_id)) - sql_writes: dict = {} - sql_reads: dict = {} + existing[(edge.source_id, edge.target_id)] = edge + statement_writes: dict = {} + statement_reads: dict = {} for e in self.graph.edges: if e.edge_type == EdgeType.WRITES_TO: - sql_writes.setdefault(e.source_id, []).append(e.target_id) + key = (e.source_id, e.properties.get("stmt_index", 0)) + statement_writes.setdefault(key, []).append(e.target_id) elif e.edge_type == EdgeType.READS_FROM: - sql_reads.setdefault(e.source_id, []).append(e.target_id) + key = (e.source_id, e.properties.get("stmt_index", 0)) + statement_reads.setdefault(key, []).append(e.target_id) cte_table_ids = set() for node in self.graph.nodes: if isinstance(node, TableNode) and node.is_cte: cte_table_ids.add(node.id) - for sql_id, src_tables in sql_reads.items(): - dst_tables = sql_writes.get(sql_id, []) + for statement_key, src_tables in statement_reads.items(): + dst_tables = statement_writes.get(statement_key, []) + sql_id, stmt_index = statement_key for src_tid in src_tables: if src_tid in cte_table_ids: continue for dst_tid in dst_tables: if dst_tid in cte_table_ids: continue - if src_tid != dst_tid and (src_tid, dst_tid) not in existing: - self.graph.add_edge(Edge( - id=_gen_id("e"), - source_id=src_tid, - target_id=dst_tid, - edge_type=EdgeType.TABLE_LINEAGE, - )) - existing.add((src_tid, dst_tid)) + if src_tid == dst_tid: + continue + provenance = { + "sql_id": sql_id, + "stmt_index": stmt_index, + } + current = existing.get((src_tid, dst_tid)) + if current is not None: + values = current.properties.setdefault("provenance", []) + if provenance not in values: + values.append(provenance) + continue + lineage_edge = Edge( + id=edge_id( + src_tid, + dst_tid, + EdgeType.TABLE_LINEAGE, + ), + source_id=src_tid, + target_id=dst_tid, + edge_type=EdgeType.TABLE_LINEAGE, + properties={"provenance": [provenance]}, + ) + self.graph.add_edge(lineage_edge) + existing[(src_tid, dst_tid)] = lineage_edge diff --git a/sqlgraph/cli.py b/sqlgraph/cli.py index 8ead4a1..5204f62 100644 --- a/sqlgraph/cli.py +++ b/sqlgraph/cli.py @@ -49,6 +49,12 @@ help="SQL lineage graph construction tool - SQL 血缘图构建工具", add_completion=False, ) +governance_app = typer.Typer( + name="governance", + help="Run, verify, and replay evidence-bound governance scenarios.", + add_completion=False, +) +app.add_typer(governance_app, name="governance") # 创建 Rich 控制台实例,用于美化终端输出 console = Console() @@ -970,6 +976,70 @@ def _format_metric_names(names) -> str: return ", ".join(str(name) for name in names) if names else "-" +@governance_app.command("run") +def governance_run( + scenario: str = typer.Argument(..., help="Path to a governance scenario YAML."), + output: str = typer.Option( + "./governance_output", + "-o", + "--output", + help="Directory for the evidence and audit package.", + ), +): + """Run a declarative governance scenario.""" + from sqlgraph.reasoning.scenario import run_scenario + + result = run_scenario(scenario, output) + console.print( + f"[green]governance complete[/green] " + f"task={result.task_id} outcome={result.outcome}" + ) + console.print(f"audit={result.audit_path}") + + +@governance_app.command("verify") +def governance_verify( + output: str = typer.Argument(..., help="Governance output directory."), +): + """Verify package completeness and audit integrity.""" + from sqlgraph.reasoning.scenario import verify_scenario_output + + result = verify_scenario_output(output) + console.print( + f"valid={result['valid']} outcome={result['outcome']} " + f"events={result['integrity']['event_count']}" + ) + if not result["valid"]: + raise typer.Exit(code=1) + + +@governance_app.command("replay") +def governance_replay( + audit_path: str = typer.Argument(..., help="Path to audit.jsonl."), +): + """Replay the ordered events in a governance audit log.""" + from sqlgraph.audit import AuditLog + + path = os.path.abspath(audit_path) + if not os.path.isfile(path): + console.print(f"[red]audit file not found: {path}[/red]") + raise typer.Exit(code=2) + with open(path, encoding="utf-8") as stream: + first = next((json.loads(line) for line in stream if line.strip()), None) + if first is None: + console.print("[red]audit file is empty[/red]") + raise typer.Exit(code=2) + replay = AuditLog(path).replay(first["task_id"]) + if not replay.integrity.valid: + console.print("[red]audit integrity check failed[/red]") + raise typer.Exit(code=1) + for event in replay.events: + console.print( + f"{event.sequence:02d} {event.step} " + f"{event.transition or '-'} {event.event_hash[:12]}" + ) + + def main(): """CLI 主入口函数""" app() diff --git a/sqlgraph/contract/__init__.py b/sqlgraph/contract/__init__.py new file mode 100644 index 0000000..6af7e8c --- /dev/null +++ b/sqlgraph/contract/__init__.py @@ -0,0 +1,81 @@ +# sqlgraph/contract/__init__.py +"""业务契约模型(REQ-ADP-02)。 + +业务契约是对"预期业务语义"的版本化权威定义。与已实现语义(图中的实现事实) +冲突时,产出"治理问题"记录,任一方不得覆盖另一方(书稿第 2 章)。 +""" +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + + +@dataclass +class BusinessContract: + """一个指标/字段的业务契约(版本化权威定义)。""" + metric: str # 指标/字段名,如 "gmv" + version: str # 契约版本 + definition: str = "" # 含义(自然语言) + include_rules: list[str] = field(default_factory=list) # 纳入规则 + exclude_rules: list[str] = field(default_factory=list) # 排除规则 + data_type: str = "" # 类型 + unit: str = "" # 单位 + null_behavior: str = "" # 空值行为 + owner: str = "" # 责任人 + approval_history: list[dict] = field(default_factory=list) # 审批历史 + + def validate_completeness(self) -> list[str]: + """校验契约字段完备性,返回缺失的必填字段列表(空表示完备)。""" + missing = [] + if not self.metric: + missing.append("metric") + if not self.version: + missing.append("version") + if not self.definition: + missing.append("definition") + if not self.owner: + missing.append("owner") + return missing + + +@dataclass +class GovernanceIssue: + """治理问题:已实现语义与业务契约冲突的记录(不改写任何一方)。""" + metric: str + contract_version: str + implemented_semantics: str # 图中观察到的已实现语义(如指纹/表达式) + expected_semantics: str # 契约声明的预期语义 + kind: str = "caliber_conflict" + + def to_dict(self) -> dict: + return { + "kind": self.kind, + "metric": self.metric, + "contract_version": self.contract_version, + "implemented": self.implemented_semantics, + "expected": self.expected_semantics, + "note": "契约与已实现语义冲突,记录为治理问题;任一方不得覆盖另一方", + } + + +def check_conflict(contract: BusinessContract, implemented_semantics: str, + expected_marker: str) -> GovernanceIssue | None: + """对照契约与已实现语义,冲突时产出治理问题记录(REQ-ADP-02 AC2)。 + + Args: + contract: 业务契约 + implemented_semantics: 图中观察到的实现(如表达式/指纹的可读描述) + expected_marker: 用于判断实现是否符合契约的标记(简化:契约要求实现中 + 应包含的关键片段,如 "refund" 表示 gmv 口径应扣退款) + + Returns: + 冲突时返回 GovernanceIssue,否则 None。不改写 contract 或图。 + """ + if expected_marker and expected_marker not in implemented_semantics: + return GovernanceIssue( + metric=contract.metric, + contract_version=contract.version, + implemented_semantics=implemented_semantics, + expected_semantics=f"应满足契约标记: {expected_marker}", + ) + return None diff --git a/sqlgraph/evidence/__init__.py b/sqlgraph/evidence/__init__.py new file mode 100644 index 0000000..ebf4b07 --- /dev/null +++ b/sqlgraph/evidence/__init__.py @@ -0,0 +1,87 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public evidence collection and sufficiency interface.""" + +from __future__ import annotations + +from sqlgraph.evidence.engine import EvidenceEngine +from sqlgraph.evidence.model import ( + EVIDENCE_RULE_VERSION, + EVIDENCE_SCHEMA_VERSION, + EvidenceBundle, + EvidenceFinding, + EvidenceRequest, + SufficiencyDecision, +) + + +def build_evidence_subgraph( + graph, + task_id: str, + anchor_tables: list[str], + intent: str = "generic", +) -> EvidenceBundle: + """Compatibility helper for callers that do not yet pass a baseline.""" + request = EvidenceRequest( + task_id=task_id, + baseline_id=graph.metadata.get("baseline_id", "baseline-unbound"), + intent=intent, + anchors=tuple(anchor_tables), + direction="both", + max_depth=1, + ) + return EvidenceEngine(graph).collect(request) + + +def assess_sufficiency( + evidence: EvidenceBundle, + required_sources: list[str] | None = None, +) -> SufficiencyDecision: + """Compatibility sufficiency check using the bundle's coverage contract.""" + if required_sources is not None: + resolved_names = set( + evidence.coverage_contract.get("anchor_names", ()) + ) + missing_sources = sorted(set(required_sources) - resolved_names) + if missing_sources: + return SufficiencyDecision( + sufficient=False, + action="expand", + missing_obligations=("anchors_resolved",), + reasons=( + "required sources are outside the current evidence scope: " + + ", ".join(missing_sources), + ), + coverage_contract=evidence.coverage_contract, + disclosed_gaps=evidence.gaps, + residual_unknowns=evidence.residual_unknowns, + ) + if evidence.gaps or evidence.residual_unknowns: + return SufficiencyDecision( + sufficient=False, + action="escalate", + missing_obligations=("no_unresolved",), + reasons=evidence.gaps, + coverage_contract=evidence.coverage_contract, + disclosed_gaps=evidence.gaps, + residual_unknowns=evidence.residual_unknowns, + ) + return SufficiencyDecision( + sufficient=True, + action="stop", + coverage_contract=evidence.coverage_contract, + ) + + +__all__ = [ + "EVIDENCE_RULE_VERSION", + "EVIDENCE_SCHEMA_VERSION", + "EvidenceBundle", + "EvidenceEngine", + "EvidenceFinding", + "EvidenceRequest", + "SufficiencyDecision", + "assess_sufficiency", + "build_evidence_subgraph", +] diff --git a/sqlgraph/evidence/engine.py b/sqlgraph/evidence/engine.py new file mode 100644 index 0000000..d316646 --- /dev/null +++ b/sqlgraph/evidence/engine.py @@ -0,0 +1,315 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Evidence collection, expansion, and sufficiency decisions.""" + +from __future__ import annotations + +from collections import deque +from dataclasses import replace + +from sqlgraph.evidence.model import ( + EvidenceBundle, + EvidenceFinding, + EvidenceRequest, + SufficiencyDecision, +) +from sqlgraph.model import EdgeType, TableNode + + +class EvidenceEngine: + def __init__(self, graph): + self.graph = graph + self._edges = {edge.id: edge for edge in graph.edges} + + def collect(self, request: EvidenceRequest) -> EvidenceBundle: + if request.direction not in {"upstream", "downstream", "both"}: + raise ValueError("direction must be upstream, downstream, or both") + if request.max_depth < 0: + raise ValueError("max_depth must be non-negative") + + anchor_nodes = [ + self.graph.get_node_by_name(anchor) + for anchor in request.anchors + ] + resolved_anchors = tuple( + node.id for node in anchor_nodes if isinstance(node, TableNode) + ) + table_ids = self._grow_tables( + set(resolved_anchors), + request.direction, + request.max_depth, + ) + included_nodes, included_edges = self._collect_detail(table_ids) + excluded = tuple(sorted( + edge.id + for edge in self.graph.edges + if edge.id not in included_edges + and ( + edge.source_id in included_nodes + or edge.target_id in included_nodes + ) + )) + + supporting = tuple( + EvidenceFinding( + kind="lineage", + summary="included table lineage edge", + citations=(edge.id,), + ) + for edge in self.graph.edges + if edge.edge_type == EdgeType.TABLE_LINEAGE + and edge.id in included_edges + ) + counterevidence = self._counterevidence(request, resolved_anchors) + residual_unknowns = tuple(sorted( + node.id + for node in self.graph.nodes + if node.id in included_nodes + and ( + getattr(node, "name", "") == "UNKNOWN" + or getattr(node, "resolution", "") == "unresolved" + ) + )) + gaps = [] + if len(resolved_anchors) != len(request.anchors): + gaps.append("one or more anchors could not be resolved") + if residual_unknowns: + gaps.append("included evidence contains unresolved physical columns") + missing_external = sorted( + set(request.required_external_facts) - set(request.external_facts) + ) + if missing_external: + gaps.append("required external facts are missing") + + coverage_contract = { + "required": list(request.coverage_obligations), + "resolved_anchors": list(resolved_anchors), + "anchor_names": list(request.anchors), + "direction": request.direction, + "max_depth": request.max_depth, + "required_external_facts": list(request.required_external_facts), + "missing_external_facts": missing_external, + } + return EvidenceBundle( + task_id=request.task_id, + baseline_id=request.baseline_id, + intent=request.intent, + anchors=request.anchors, + version=1, + included_nodes=tuple(sorted(included_nodes)), + included_edges=tuple(sorted(included_edges)), + excluded=excluded, + supporting=supporting, + counterevidence=counterevidence, + coverage_contract=coverage_contract, + gaps=tuple(gaps), + residual_unknowns=residual_unknowns, + external_facts=request.external_facts, + ) + + def expand(self, bundle: EvidenceBundle, reason: str) -> EvidenceBundle: + contract = bundle.coverage_contract + request = EvidenceRequest( + task_id=bundle.task_id, + baseline_id=bundle.baseline_id, + intent=bundle.intent, + anchors=bundle.anchors, + direction=contract["direction"], + max_depth=int(contract["max_depth"]) + 1, + coverage_obligations=tuple(contract["required"]), + required_external_facts=tuple( + contract.get("required_external_facts", ()) + ), + external_facts=bundle.external_facts, + ) + expanded = self.collect(request) + return replace( + expanded, + version=bundle.version + 1, + parent_hash=bundle.subgraph_hash, + history=bundle.history + ({ + "from_version": bundle.version, + "to_version": bundle.version + 1, + "reason": reason, + },), + ) + + def assess(self, bundle: EvidenceBundle) -> SufficiencyDecision: + required = tuple(bundle.coverage_contract.get("required", ())) + missing = [] + if "anchors_resolved" in required and ( + len(bundle.coverage_contract.get("resolved_anchors", ())) + != len(bundle.anchors) + ): + missing.append("anchors_resolved") + if "lineage_path" in required and not self._has_anchor_path(bundle): + missing.append("lineage_path") + if "counterevidence_checked" in required and bundle.counterevidence is None: + missing.append("counterevidence_checked") + if "no_unresolved" in required and bundle.residual_unknowns: + missing.append("no_unresolved") + if ( + "external_facts" in required + and bundle.coverage_contract.get("missing_external_facts") + ): + missing.append("external_facts") + + if not missing and not bundle.gaps: + action = "stop" + elif "no_unresolved" in missing or "external_facts" in missing: + action = "escalate" + elif bundle.version == 1: + action = "expand" + else: + action = "degrade" + return SufficiencyDecision( + sufficient=not missing and not bundle.gaps, + action=action, + missing_obligations=tuple(missing), + reasons=tuple(bundle.gaps), + coverage_contract=bundle.coverage_contract, + disclosed_gaps=bundle.gaps, + residual_unknowns=bundle.residual_unknowns, + ) + + def _grow_tables( + self, + anchors: set[str], + direction: str, + max_depth: int, + ) -> set[str]: + upstream: dict[str, set[str]] = {} + downstream: dict[str, set[str]] = {} + for edge in self.graph.edges: + if edge.edge_type != EdgeType.TABLE_LINEAGE: + continue + downstream.setdefault(edge.source_id, set()).add(edge.target_id) + upstream.setdefault(edge.target_id, set()).add(edge.source_id) + + included = set(anchors) + queue = deque((anchor, 0) for anchor in sorted(anchors)) + while queue: + current, depth = queue.popleft() + if depth >= max_depth: + continue + neighbors = set() + if direction in {"downstream", "both"}: + neighbors |= downstream.get(current, set()) + if direction in {"upstream", "both"}: + neighbors |= upstream.get(current, set()) + for neighbor in sorted(neighbors): + if neighbor not in included: + included.add(neighbor) + queue.append((neighbor, depth + 1)) + return included + + def _collect_detail(self, table_ids: set[str]) -> tuple[set[str], set[str]]: + nodes = set(table_ids) + edges = set() + + for edge in self.graph.edges: + if ( + edge.edge_type == EdgeType.TABLE_LINEAGE + and edge.source_id in table_ids + and edge.target_id in table_ids + ): + edges.add(edge.id) + + changed = True + while changed: + changed = False + for edge in self.graph.edges: + include = False + if ( + edge.edge_type == EdgeType.HAS_COLUMN + and ( + edge.source_id in nodes + or edge.target_id in nodes + ) + ): + include = True + elif ( + edge.edge_type == EdgeType.PRODUCES + and edge.target_id in nodes + ): + include = True + elif ( + edge.edge_type in { + EdgeType.COMPUTE_DEPENDENCY, + EdgeType.EXPR_OPERAND, + } + and edge.target_id in nodes + ): + include = True + elif ( + edge.edge_type == EdgeType.CONTAINS + and edge.target_id in nodes + ): + include = True + elif ( + edge.edge_type in {EdgeType.READS_FROM, EdgeType.WRITES_TO} + and edge.target_id in table_ids + ): + include = True + if not include: + continue + old_size = len(nodes) + nodes.update((edge.source_id, edge.target_id)) + edges.add(edge.id) + changed = changed or len(nodes) != old_size + return nodes, edges + + def _counterevidence( + self, + request: EvidenceRequest, + resolved_anchors: tuple[str, ...], + ) -> tuple[EvidenceFinding, ...]: + if len(resolved_anchors) < 2: + return () + expected_source = resolved_anchors[0] + target = resolved_anchors[-1] + findings = [] + for edge in self.graph.edges: + if ( + edge.edge_type == EdgeType.TABLE_LINEAGE + and edge.target_id == target + and edge.source_id != expected_source + ): + source = self.graph.get_node(edge.source_id) + findings.append(EvidenceFinding( + kind="alternative_upstream", + summary=( + f"{getattr(source, 'full_name', edge.source_id)} " + "also contributes to the target" + ), + citations=(edge.id,), + )) + return tuple(sorted(findings, key=lambda item: item.citations)) + + def _has_anchor_path(self, bundle: EvidenceBundle) -> bool: + if len(bundle.anchors) < 2: + return bool(bundle.included_nodes) + source = self.graph.get_node_by_name(bundle.anchors[0]) + target = self.graph.get_node_by_name(bundle.anchors[-1]) + if source is None or target is None: + return False + allowed_edges = set(bundle.included_edges) + adjacency: dict[str, set[str]] = {} + for edge in self.graph.edges: + if ( + edge.id in allowed_edges + and edge.edge_type == EdgeType.TABLE_LINEAGE + ): + adjacency.setdefault(edge.source_id, set()).add(edge.target_id) + seen = set() + stack = [source.id] + while stack: + current = stack.pop() + if current == target.id: + return True + if current in seen: + continue + seen.add(current) + stack.extend(adjacency.get(current, ())) + return False diff --git a/sqlgraph/evidence/model.py b/sqlgraph/evidence/model.py new file mode 100644 index 0000000..190e040 --- /dev/null +++ b/sqlgraph/evidence/model.py @@ -0,0 +1,146 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Versioned evidence contracts.""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import asdict, dataclass, field + + +EVIDENCE_SCHEMA_VERSION = "evidence-bundle-v1" +EVIDENCE_RULE_VERSION = "evidence-v1" + + +@dataclass(frozen=True) +class EvidenceRequest: + task_id: str + baseline_id: str + intent: str + anchors: tuple[str, ...] + direction: str = "both" + max_depth: int = 1 + coverage_obligations: tuple[str, ...] = ( + "anchors_resolved", + "lineage_path", + "counterevidence_checked", + "no_unresolved", + ) + required_external_facts: tuple[str, ...] = () + external_facts: tuple[str, ...] = () + + +@dataclass(frozen=True) +class EvidenceFinding: + kind: str + summary: str + citations: tuple[str, ...] = () + + +@dataclass(frozen=True) +class EvidenceBundle: + task_id: str + baseline_id: str + intent: str + anchors: tuple[str, ...] + version: int + included_nodes: tuple[str, ...] + included_edges: tuple[str, ...] + excluded: tuple[str, ...] + supporting: tuple[EvidenceFinding, ...] + counterevidence: tuple[EvidenceFinding, ...] + coverage_contract: dict + gaps: tuple[str, ...] + residual_unknowns: tuple[str, ...] + external_facts: tuple[str, ...] = () + parent_hash: str | None = None + history: tuple[dict, ...] = () + schema_version: str = EVIDENCE_SCHEMA_VERSION + rule_version: str = EVIDENCE_RULE_VERSION + + @property + def subgraph_hash(self) -> str: + payload = { + "schema_version": self.schema_version, + "rule_version": self.rule_version, + "task_id": self.task_id, + "baseline_id": self.baseline_id, + "intent": self.intent, + "anchors": self.anchors, + "version": self.version, + "included_nodes": self.included_nodes, + "included_edges": self.included_edges, + "excluded": self.excluded, + "supporting": [asdict(item) for item in self.supporting], + "counterevidence": [asdict(item) for item in self.counterevidence], + "coverage_contract": self.coverage_contract, + "gaps": self.gaps, + "residual_unknowns": self.residual_unknowns, + "external_facts": self.external_facts, + "parent_hash": self.parent_hash, + "history": self.history, + } + canonical = json.dumps( + payload, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + ) + return "evidence_" + hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + @property + def version_id(self) -> str: + return f"{self.task_id}@v{self.version}#{self.subgraph_hash[-12:]}" + + @property + def node_ids(self) -> set[str]: + return set(self.included_nodes) + + @property + def edge_ids(self) -> set[str]: + return set(self.included_edges) + + @property + def coverage(self) -> dict: + return self.coverage_contract + + def to_dict(self) -> dict: + return { + "schema_version": self.schema_version, + "rule_version": self.rule_version, + "task_id": self.task_id, + "baseline_id": self.baseline_id, + "intent": self.intent, + "anchors": list(self.anchors), + "version": self.version, + "version_id": self.version_id, + "subgraph_hash": self.subgraph_hash, + "included_nodes": list(self.included_nodes), + "included_edges": list(self.included_edges), + "excluded": list(self.excluded), + "supporting": [asdict(item) for item in self.supporting], + "counterevidence": [asdict(item) for item in self.counterevidence], + "coverage_contract": self.coverage_contract, + "gaps": list(self.gaps), + "residual_unknowns": list(self.residual_unknowns), + "external_facts": list(self.external_facts), + "parent_hash": self.parent_hash, + "history": list(self.history), + } + + +@dataclass(frozen=True) +class SufficiencyDecision: + sufficient: bool + action: str + missing_obligations: tuple[str, ...] = () + reasons: tuple[str, ...] = () + coverage_contract: dict = field(default_factory=dict) + disclosed_gaps: tuple[str, ...] = () + residual_unknowns: tuple[str, ...] = () + + @property + def decision(self) -> str: + return self.action diff --git a/sqlgraph/graphrag/__init__.py b/sqlgraph/graphrag/__init__.py new file mode 100644 index 0000000..54a7adc --- /dev/null +++ b/sqlgraph/graphrag/__init__.py @@ -0,0 +1,16 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Grounded GraphRAG contracts.""" + +from sqlgraph.graphrag.grounding import ( + GroundedAssertion, + GroundingReport, + validate_assertions, +) + +__all__ = [ + "GroundedAssertion", + "GroundingReport", + "validate_assertions", +] diff --git a/sqlgraph/graphrag/grounding.py b/sqlgraph/graphrag/grounding.py new file mode 100644 index 0000000..daae806 --- /dev/null +++ b/sqlgraph/graphrag/grounding.py @@ -0,0 +1,84 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Validate natural-language assertions against an evidence bundle.""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Sequence + +from sqlgraph.evidence import EvidenceBundle + + +@dataclass(frozen=True) +class GroundedAssertion: + statement: str + citations: tuple[str, ...] + baseline_id: str + evidence_hash: str + + +@dataclass(frozen=True) +class GroundingReport: + status: str + assertion_count: int + valid_assertions: int + invalid_citations: tuple[str, ...] = () + out_of_scope_citations: tuple[str, ...] = () + version_mismatches: tuple[str, ...] = () + + def to_dict(self) -> dict: + return { + "status": self.status, + "assertion_count": self.assertion_count, + "valid_assertions": self.valid_assertions, + "invalid_citations": list(self.invalid_citations), + "out_of_scope_citations": list(self.out_of_scope_citations), + "version_mismatches": list(self.version_mismatches), + } + + +def validate_assertions( + assertions: Sequence[GroundedAssertion], + evidence: EvidenceBundle, +) -> GroundingReport: + """Reject assertions whose citations cannot be verified in the bundle.""" + allowed = set(evidence.included_nodes) | set(evidence.included_edges) + excluded = set(evidence.excluded) + invalid = set() + out_of_scope = set() + mismatches = [] + valid = 0 + + for index, assertion in enumerate(assertions): + mismatch = ( + assertion.baseline_id != evidence.baseline_id + or assertion.evidence_hash != evidence.subgraph_hash + ) + if mismatch: + mismatches.append(f"assertion:{index}") + continue + if not assertion.citations: + invalid.add(f"assertion:{index}:missing-citation") + continue + assertion_invalid = False + for citation in assertion.citations: + if citation in excluded: + out_of_scope.add(citation) + assertion_invalid = True + elif citation not in allowed: + invalid.add(citation) + assertion_invalid = True + if not assertion_invalid: + valid += 1 + + rejected = bool(invalid or out_of_scope or mismatches or valid != len(assertions)) + return GroundingReport( + status="rejected" if rejected else "grounded", + assertion_count=len(assertions), + valid_assertions=valid, + invalid_citations=tuple(sorted(invalid)), + out_of_scope_citations=tuple(sorted(out_of_scope)), + version_mismatches=tuple(mismatches), + ) diff --git a/sqlgraph/identity/__init__.py b/sqlgraph/identity/__init__.py new file mode 100644 index 0000000..a2eee4a --- /dev/null +++ b/sqlgraph/identity/__init__.py @@ -0,0 +1,75 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Deterministic identities for graph and governance artifacts.""" + +from __future__ import annotations + +import hashlib + + +IDENTITY_RULE_VERSION = "id-v2" + + +def _digest(text: str, bits: int) -> str: + if bits <= 0 or bits % 4: + raise ValueError("bits must be a positive multiple of four") + return hashlib.sha256(text.encode("utf-8")).hexdigest()[: bits // 4] + + +def stable_id(prefix: str, semantic_key: str, bits: int = 96) -> str: + """Return a stable identifier derived from a semantic key.""" + return f"{prefix}_{_digest(semantic_key, bits)}" + + +def node_id(prefix: str, semantic_key: str) -> str: + return stable_id(prefix, semantic_key, bits=96) + + +def edge_id( + source_id: str, + target_id: str, + edge_type, + context: str = "", +) -> str: + edge_name = getattr(edge_type, "value", str(edge_type)) + semantic_key = f"{source_id}->{target_id}:{edge_name}:{context}" + return stable_id("e", semantic_key, bits=96) + + +def content_fingerprint(text: str, bits: int = 128) -> str: + return _digest(text, bits) + + +class CollisionRegistry: + """Detect the same derived ID being assigned to different semantic keys.""" + + def __init__(self): + self._seen: dict[str, str] = {} + self.collisions: list[tuple[str, str, str]] = [] + + def register(self, identifier: str, semantic_key: str) -> bool: + existing = self._seen.get(identifier) + if existing is None: + self._seen[identifier] = semantic_key + return True + if existing == semantic_key: + return True + self.collisions.append((identifier, existing, semantic_key)) + return False + + def has_collision(self) -> bool: + return bool(self.collisions) + + +def environment_fingerprint( + dialect: str | None, + parser_version: str, + config_hash: str = "", +) -> dict[str, str]: + return { + "identity_rule_version": IDENTITY_RULE_VERSION, + "dialect": dialect or "", + "parser_version": parser_version, + "config_hash": config_hash, + } diff --git a/sqlgraph/input/csv_schema.py b/sqlgraph/input/csv_schema.py index 88c9817..b822c64 100644 --- a/sqlgraph/input/csv_schema.py +++ b/sqlgraph/input/csv_schema.py @@ -125,6 +125,20 @@ def has_table(self, table_name: str) -> bool: return True return False + def to_dict(self) -> dict: + """Return a stable, serializable schema snapshot.""" + return { + name: [ + { + "name": column.name, + "data_type": column.data_type, + "description": column.description, + } + for column in table.columns + ] + for name, table in sorted(self._tables.items()) + } + @classmethod def from_csv(cls, csv_path: str) -> "SchemaRegistry": """从 CSV 文件加载 Schema diff --git a/sqlgraph/lineage/__init__.py b/sqlgraph/lineage/__init__.py new file mode 100644 index 0000000..131e682 --- /dev/null +++ b/sqlgraph/lineage/__init__.py @@ -0,0 +1,112 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Trace a projected table edge back to statement and column evidence.""" + +from __future__ import annotations + +from sqlgraph.model import ColumnNode, EdgeType, TransformNode + + +def _column_ids(graph, table_id: str) -> set[str]: + return { + edge.target_id + for edge in graph.edges + if edge.edge_type == EdgeType.HAS_COLUMN + and edge.source_id == table_id + } + + +def _column_name(graph, column_id: str) -> str: + column = graph.get_node(column_id) + if not isinstance(column, ColumnNode): + return column_id + table = graph.get_node(column.table_id) if column.table_id else None + table_name = getattr(table, "full_name", None) + return f"{table_name}.{column.name}" if table_name else column.name + + +def drilldown(graph, source_table: str, target_table: str) -> dict: + """Return the evidence that produced one direct table-lineage edge.""" + source = graph.get_node_by_name(source_table) + target = graph.get_node_by_name(target_table) + if source is None or target is None: + return {"found": False, "reason": "table not found"} + + lineage_edge = next( + ( + edge for edge in graph.edges + if edge.edge_type == EdgeType.TABLE_LINEAGE + and edge.source_id == source.id + and edge.target_id == target.id + ), + None, + ) + if lineage_edge is None: + return {"found": False, "reason": "no table_lineage edge"} + + provenance = tuple(lineage_edge.properties.get("provenance", ())) + target_columns = _column_ids(graph, target.id) + column_paths = [] + transform_ids: set[str] = set() + + for target_column_id in sorted(target_columns): + producers = [ + edge.source_id + for edge in graph.edges + if edge.edge_type == EdgeType.PRODUCES + and edge.target_id == target_column_id + ] + for transform_id in producers: + transform = graph.get_node(transform_id) + if not isinstance(transform, TransformNode): + continue + dependencies = [ + edge.source_id + for edge in graph.edges + if edge.edge_type == EdgeType.COMPUTE_DEPENDENCY + and edge.target_id == transform_id + ] + matching = [ + dependency + for dependency in dependencies + if getattr(graph.get_node(dependency), "table_id", None) == source.id + ] + if not matching: + continue + transform_ids.add(transform_id) + column_paths.append({ + "target_column": _column_name(graph, target_column_id), + "transform": transform_id, + "source_columns": tuple( + sorted(_column_name(graph, item) for item in matching) + ), + }) + + passthrough = [ + edge.source_id + for edge in graph.edges + if edge.edge_type == EdgeType.COMPUTE_DEPENDENCY + and edge.target_id == target_column_id + and getattr(graph.get_node(edge.source_id), "table_id", None) == source.id + ] + if passthrough: + column_paths.append({ + "target_column": _column_name(graph, target_column_id), + "transform": None, + "source_columns": tuple( + sorted(_column_name(graph, item) for item in passthrough) + ), + }) + + return { + "found": True, + "edge": (source.id, target.id), + "edge_id": lineage_edge.id, + "provenance": list(provenance), + "statements": sorted({ + ref["sql_id"] for ref in provenance if ref.get("sql_id") + }), + "transforms": sorted(transform_ids), + "column_paths": column_paths, + } diff --git a/sqlgraph/metrics/__init__.py b/sqlgraph/metrics/__init__.py new file mode 100644 index 0000000..d6a623a --- /dev/null +++ b/sqlgraph/metrics/__init__.py @@ -0,0 +1,143 @@ +# sqlgraph/metrics/__init__.py +"""结构风险指标与残余风险(REQ-MET-01 / REQ-MET-02)。 + +基于 TableGraph 计算可达性(潜在结构影响面)、桥接分等结构指标。核心纪律: +**指标结果必须可还原为图上可指认的证据**(可达路径 / 桥接边),不得由模型 +语感或单一分数直接裁决治理结论。 +""" +from __future__ import annotations + +from sqlgraph.model import EdgeType, TableNode + + +def _table_adjacency(graph) -> dict[str, list[str]]: + """由 table_lineage 边构造表级有向邻接表 (src -> [dst])。""" + adj: dict[str, list[str]] = {} + for e in graph.edges: + if e.edge_type == EdgeType.TABLE_LINEAGE: + adj.setdefault(e.source_id, []).append(e.target_id) + return adj + + +def reachable_set(graph, table_name: str) -> list[str]: + """沿已收录血缘可达的下游表集合(潜在结构影响面上界,非实际变化集)。""" + node = graph.get_node_by_name(table_name) + if not node: + return [] + adj = _table_adjacency(graph) + seen: set[str] = set() + stack = list(adj.get(node.id, [])) + while stack: + cur = stack.pop() + if cur in seen: + continue + seen.add(cur) + stack.extend(adj.get(cur, [])) + return sorted( + graph.get_node(tid).full_name for tid in seen if graph.get_node(tid) + ) + + +def blast_score(graph, table_name: str) -> dict: + """影响面分数(REQ-MET-01):可达下游表数量 + 可解释的可达集证据。 + + Returns: {"score", "reachable", "evidence": {"paths": ...}} + """ + reachable = reachable_set(graph, table_name) + return { + "table": table_name, + "score": len(reachable), + "reachable": reachable, + "evidence": { + "kind": "reachable_set", + "note": "沿已收录 table_lineage 边可达的下游集合,是潜在结构影响面上界,非实际变化集", + "members": reachable, + }, + } + + +def bridge_score(graph, table_name: str) -> dict: + """桥接分(REQ-MET-01,简化参考实现):删点后有多少下游可达性被切断。 + + 以"删除该表后,其直接下游从该表上游不再可达的比例"近似桥接性。结果附带 + 被切断的下游作为可指认证据。真正的割点判定需完整连通性测试,此处是参考 + 实现,可替换。 + """ + node = graph.get_node_by_name(table_name) + if not node: + return {"table": table_name, "score": 0.0, "evidence": {}} + adj = _table_adjacency(graph) + # 该表的上游(谁指向它)与下游 + upstream = [s for s, dsts in adj.items() if node.id in dsts] + downstream = adj.get(node.id, []) + if not downstream: + return {"table": table_name, "score": 0.0, + "evidence": {"kind": "cut_test", "cut_off": []}} + # 删除 node 后,从上游还能否到达每个下游 + def _reach_without(start: str, blocked: str) -> set[str]: + seen, stack = set(), list(adj.get(start, [])) + while stack: + c = stack.pop() + if c == blocked or c in seen: + continue + seen.add(c) + stack.extend(x for x in adj.get(c, []) if x != blocked) + return seen + reachable_after: set[str] = set() + for up in upstream: + reachable_after |= _reach_without(up, node.id) + cut_off = [d for d in downstream if d not in reachable_after] + score = len(cut_off) / len(downstream) if downstream else 0.0 + return { + "table": table_name, + "score": round(score, 4), + "evidence": { + "kind": "cut_test", + "cut_off": [graph.get_node(d).full_name for d in cut_off if graph.get_node(d)], + "note": "删除该表后,从其上游不再可达的直接下游(近似桥接性证据)", + }, + } + + +# --- REQ-MET-02: roi 与残余风险 --------------------------------------------- + +# 残余风险的已知类别(书稿第五部):证据边界、解析遗漏、共同模式错误、 +# 监控延迟、回滚失败。 +RESIDUAL_RISK_CATEGORIES = [ + "evidence_boundary", # 证据边界(如可达集只是上界) + "parse_omission", # 解析遗漏(动态 SQL / UNKNOWN 列) + "common_mode_error", # 共同模式错误 + "monitoring_delay", # 监控延迟 + "rollback_failure", # 回滚失败 +] + + +def roi(before: float, after: float, target: float | None = None, + tolerance: float = 0.05) -> dict: + """口径/成本修复的 roi 指标(REQ-MET-02)。 + + 可复算:给定修复前后值,返回变化与(可选)相对预演目标的误差带判定。 + 书稿示例:0.9 -> 1.78(预演目标约 1.8,落在可接受误差内)。 + """ + result = { + "before": before, + "after": after, + "delta": round(after - before, 6), + } + if target is not None: + rel_err = abs(after - target) / abs(target) if target else float("inf") + result["target"] = target + result["relative_error"] = round(rel_err, 6) + result["within_tolerance"] = rel_err <= tolerance + return result + + +def with_residual_risk(conclusion: dict, categories: list[str] | None = None) -> dict: + """给一个治理结论附加残余风险披露字段(REQ-MET-02 AC2)。 + + residual_risk 非空,显式列举已知残余风险类别,不允许"零风险"承诺。 + """ + cats = categories if categories is not None else list(RESIDUAL_RISK_CATEGORIES) + out = dict(conclusion) + out["residual_risk"] = cats + return out diff --git a/sqlgraph/model/graph.py b/sqlgraph/model/graph.py index b06e5c1..2dcb418 100644 --- a/sqlgraph/model/graph.py +++ b/sqlgraph/model/graph.py @@ -14,6 +14,7 @@ def __init__(self): self._nodes: dict = {} self._edges: list = [] self._node_by_name: dict = {} + self.metadata: dict[str, Any] = {} @property def nodes(self) -> list: @@ -114,4 +115,5 @@ def to_dict(self) -> dict: return { "nodes": [n.to_dict() for n in self._nodes.values()], "edges": [e.to_dict() for e in self._edges], + "metadata": self.metadata, } diff --git a/sqlgraph/parser/base.py b/sqlgraph/parser/base.py index 2bd9466..f27b710 100644 --- a/sqlgraph/parser/base.py +++ b/sqlgraph/parser/base.py @@ -3,7 +3,6 @@ # sqlgraph/parser/base.py from __future__ import annotations -import uuid import hashlib import re from typing import Optional @@ -11,6 +10,7 @@ from sqlglot import exp from sqlgraph.utils.logging import log_info, log_warn from sqlgraph.input.csv_schema import SchemaRegistry +from sqlgraph.identity import node_id from sqlgraph.utils.errors import SqlParseError from sqlgraph.parser import expr_dag @@ -71,18 +71,6 @@ def resolve(self, col) -> str: return f"{UNKNOWN_TABLE}.{col_name}" -def _gen_id(prefix: str = "n") -> str: - """生成唯一节点 ID - - Args: - prefix: ID 前缀,默认为 "n" - - Returns: - 唯一 ID 字符串,格式为 "{prefix}_{8位十六进制随机数}" - """ - return f"{prefix}_{uuid.uuid4().hex[:8]}" - - class SqlParseResult: """单条 SQL 解析结果 @@ -113,6 +101,7 @@ def __init__(self): self.cte_tables: list[dict] = [] self.columns: list[dict] = [] self.errors: list[str] = [] + self.statement_io: list[dict] = [] class SqlParser: @@ -150,7 +139,10 @@ def parse(self, sql: str, name: str = "sql", file_path: str | None = None) -> Sq SqlParseError: SQL 解析失败时抛出 """ result = SqlParseResult() - result.sql_id = _gen_id("sql") + result.sql_id = node_id( + "sql", + f"{file_path or name}::{self.dialect or ''}::{sql}", + ) result.sql_name = name result.sql_content = sql result.file_path = file_path @@ -166,10 +158,30 @@ def parse(self, sql: str, name: str = "sql", file_path: str | None = None) -> Sq statements = sqlglot.parse(sql, read=self.dialect) except Exception as e: raise SqlParseError(f"Failed to parse SQL: {e}", sql=sql, file_path=file_path) + stmt_index = 0 for stmt in statements: if stmt is None: continue + self._cte_aliases = {} + self._relation_aliases = {} + self._cte_relation_names = set() + self._current_source_tables = [] + self._current_all_sources = [] + self._parsed_derived_queries = set() + source_start = len(result.source_tables) + target_start = len(result.target_tables) + self._statement_source_start = source_start self._parse_statement(stmt, result) + result.statement_io.append({ + "stmt_index": stmt_index, + "sources": [ + dict(entry) for entry in result.source_tables[source_start:] + ], + "targets": [ + dict(entry) for entry in result.target_tables[target_start:] + ], + }) + stmt_index += 1 return result def _parse_statement(self, stmt, result: SqlParseResult) -> None: @@ -350,7 +362,12 @@ def _parse_select( elif alias and alias != tname: self._relation_aliases[alias] = tname is_cte = tname in self._cte_relation_names - already_added = any(t["name"] == tname and t.get("alias") == alias for t in result.source_tables) + already_added = any( + t["name"] == tname and t.get("alias") == alias + for t in result.source_tables[ + getattr(self, "_statement_source_start", 0): + ] + ) already_in_target = any(t["name"] == tname for t in result.target_tables) if not already_added and not already_in_target: result.source_tables.append({"name": tname, "alias": alias, "is_cte": is_cte}) @@ -421,12 +438,15 @@ def _parse_columns( "physical_column": None, "expr_root": None, "expr_nodes": {}, + "operand_nodes": {}, + "operand_edges": [], } if lateral_source: root_fp, nodes = expr_dag.decompose(dag_expr, resolver.resolve, dialect=self.dialect or None) col_entry["expr_root"] = root_fp col_entry["expr_nodes"] = nodes + self._fill_operands(col_entry, dag_expr, resolver) elif not lateral_match and expr_dag.is_passthrough(inner): # 纯透传列:不建表达式节点,直接记录物理列 col_entry["passthrough"] = True @@ -435,9 +455,29 @@ def _parse_columns( root_fp, nodes = expr_dag.decompose(dag_expr, resolver.resolve, dialect=self.dialect or None) col_entry["expr_root"] = root_fp col_entry["expr_nodes"] = nodes + self._fill_operands(col_entry, dag_expr, resolver) result.columns.append(col_entry) + def _fill_operands( + self, + col_entry: dict, + expression, + resolver: ColumnResolver, + ) -> None: + """Record nested expression relationships without changing root identity.""" + try: + _, nodes, edges = expr_dag.decompose_operands( + expression, + resolver.resolve, + dialect=self.dialect or None, + ) + except Exception as exc: + log_warn(f"operand decomposition skipped: {exc}") + return + col_entry["operand_nodes"] = nodes + col_entry["operand_edges"] = edges + def _analyze_expression(self, expr) -> dict: """分析表达式类型 diff --git a/sqlgraph/parser/expr_dag.py b/sqlgraph/parser/expr_dag.py index 106f25d..29b67c9 100644 --- a/sqlgraph/parser/expr_dag.py +++ b/sqlgraph/parser/expr_dag.py @@ -131,3 +131,81 @@ def decompose(expr, resolve_column, dialect=None): } } return fp, nodes + + +_TRANSPARENT = (exp.Paren, exp.Ordered, exp.Alias) +_DECOMPOSABLE = ( + exp.Add, + exp.Sub, + exp.Mul, + exp.Div, + exp.Mod, + exp.Case, + exp.Coalesce, + exp.Cast, + exp.Round, + exp.And, + exp.Or, + exp.AggFunc, + exp.Anonymous, + exp.Func, +) + + +def decompose_operands(expr, resolve_column, dialect=None): + """Describe nested expression nodes as child-to-parent operand edges. + + The root fingerprint remains the conservative fingerprint produced by + :func:`decompose`. No algebraic equivalence rules are introduced. + """ + root_fp, _ = decompose(expr, resolve_column, dialect) + nodes: dict[str, dict] = {} + edges: list[tuple[str, str]] = [] + seen_edges: set[tuple[str, str]] = set() + + def register(node) -> str: + canonical = _canonical(node, resolve_column, dialect) + fingerprint = _fp(canonical) + if fingerprint not in nodes: + columns = ( + [node] + if isinstance(node, exp.Column) + else list(node.find_all(exp.Column)) + ) + source_columns = [] + seen_columns = set() + for column in columns: + physical = resolve_column(column) + if physical not in seen_columns: + seen_columns.add(physical) + source_columns.append(physical) + nodes[fingerprint] = { + "fingerprint": fingerprint, + "op": getattr(node, "key", "expr"), + "expr_type": classify_expr_type(node), + "expression": _display(node, resolve_column, dialect), + "canonical": canonical, + "source_columns": source_columns, + } + return fingerprint + + def walk(node, parent_fingerprint: str) -> None: + while isinstance(node, _TRANSPARENT): + inner = node.args.get("this") + if inner is None: + return + node = inner + if not isinstance(node, _DECOMPOSABLE): + return + fingerprint = register(node) + edge = (fingerprint, parent_fingerprint) + if fingerprint != parent_fingerprint and edge not in seen_edges: + seen_edges.add(edge) + edges.append(edge) + for child in node.iter_expressions(): + walk(child, fingerprint) + + register(expr) + for child in expr.iter_expressions(): + walk(child, root_fp) + return root_fp, nodes, edges diff --git a/sqlgraph/reasoning/__init__.py b/sqlgraph/reasoning/__init__.py new file mode 100644 index 0000000..649a673 --- /dev/null +++ b/sqlgraph/reasoning/__init__.py @@ -0,0 +1,18 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public seven-step governance runner interface.""" + +from sqlgraph.reasoning.runner import ( + STEPS, + GovernanceRequest, + GovernanceResult, + GovernanceRunner, +) + +__all__ = [ + "STEPS", + "GovernanceRequest", + "GovernanceResult", + "GovernanceRunner", +] diff --git a/sqlgraph/reasoning/runner.py b/sqlgraph/reasoning/runner.py new file mode 100644 index 0000000..688dcdb --- /dev/null +++ b/sqlgraph/reasoning/runner.py @@ -0,0 +1,287 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Seven-step governance orchestration over explicit module interfaces.""" + +from __future__ import annotations + +from dataclasses import dataclass, replace +from pathlib import Path +from typing import Any + +from sqlgraph.actions import ( + ActionEngine, + ActionPlan, + ActionRequest, + ExecutionResult, +) +from sqlgraph.audit import AuditEvent, AuditLog +from sqlgraph.autonomy import ( + AutonomyDecision, + AutonomyLevel, + GovernanceAction, + decide_autonomy, +) +from sqlgraph.baseline import BaselineManifest +from sqlgraph.evidence import EvidenceBundle, EvidenceEngine, EvidenceRequest +from sqlgraph.graphrag import GroundedAssertion, validate_assertions +from sqlgraph.verification import ( + LayerResult, + VerificationEngine, + VerificationReport, +) + + +STEPS = ( + "observe", + "explain", + "propose", + "authorize", + "execute", + "verify", + "learn", +) + + +@dataclass(frozen=True) +class GovernanceRequest: + task_id: str + baseline: BaselineManifest + graph: Any + evidence_request: EvidenceRequest + action: GovernanceAction + adapter: str + operations: tuple[dict[str, Any], ...] + source_path: str + source_table: str + target_table: str + dialect: str | None = None + rollback_plan: tuple[dict[str, Any], ...] = () + runtime_observed: dict | None = None + + +@dataclass(frozen=True) +class GovernanceResult: + task_id: str + outcome: str + evidence: EvidenceBundle + decision: AutonomyDecision + action_plan: ActionPlan | None + execution: ExecutionResult + verification: VerificationReport + events: tuple[AuditEvent, ...] + audit_path: str + + def to_dict(self) -> dict: + return { + "task_id": self.task_id, + "outcome": self.outcome, + "evidence": self.evidence.to_dict(), + "decision": self.decision.to_dict(), + "action_plan": ( + self.action_plan.to_dict() if self.action_plan else None + ), + "execution": self.execution.to_dict(), + "verification": self.verification.to_dict(), + "events": [event.to_dict() for event in self.events], + "audit_path": self.audit_path, + } + + +class GovernanceRunner: + def __init__( + self, + evidence_engine: EvidenceEngine, + action_engine: ActionEngine, + verification_engine: VerificationEngine, + audit_log: AuditLog, + ): + self.evidence_engine = evidence_engine + self.action_engine = action_engine + self.verification_engine = verification_engine + self.audit_log = audit_log + + def run(self, request: GovernanceRequest) -> GovernanceResult: + events = [] + + def record( + step: str, + payload: dict, + *, + evidence_version: str = "", + policy_version: str = "", + authorization_identity: str = "", + idempotency_key: str = "", + transition: str = "", + ) -> None: + events.append(self.audit_log.append(AuditEvent( + task_id=request.task_id, + baseline_id=request.baseline.baseline_id, + event_type=step, + step=step, + payload=payload, + evidence_version=evidence_version, + policy_version=policy_version, + authorization_identity=authorization_identity, + idempotency_key=idempotency_key, + transition=transition, + ))) + + evidence = self.evidence_engine.collect(request.evidence_request) + sufficiency = self.evidence_engine.assess(evidence) + record( + "observe", + { + "evidence_hash": evidence.subgraph_hash, + "sufficiency": sufficiency.action, + "missing_obligations": list(sufficiency.missing_obligations), + }, + evidence_version=evidence.version_id, + ) + + citations = tuple( + citation + for finding in evidence.supporting + for citation in finding.citations + ) + grounding = validate_assertions( + [GroundedAssertion( + statement=( + f"{request.target_table} depends on " + f"{request.source_table}" + ), + citations=citations, + baseline_id=request.baseline.baseline_id, + evidence_hash=evidence.subgraph_hash, + )], + evidence, + ) + grounded = grounding.status == "grounded" and sufficiency.sufficient + record( + "explain", + grounding.to_dict(), + evidence_version=evidence.version_id, + ) + record( + "propose", + { + "action_type": request.action.action_type, + "adapter": request.adapter, + "operation_count": len(request.operations), + }, + evidence_version=evidence.version_id, + ) + + action = replace( + request.action, + evidence_version=evidence.version_id, + evidence_grounded=grounded, + ) + decision = decide_autonomy(action) + record( + "authorize", + decision.to_dict(), + evidence_version=evidence.version_id, + policy_version=decision.policy_version, + authorization_identity=action.authorization_identity, + transition="cognition->action", + ) + + plan = None + if ( + decision.level == AutonomyLevel.L3_BOUNDED + and not decision.requires_human_review + ): + plan = self.action_engine.plan(ActionRequest( + task_id=request.task_id, + baseline_id=request.baseline.baseline_id, + evidence_version=evidence.version_id, + decision=decision, + adapter=request.adapter, + operations=request.operations, + rollback_plan=request.rollback_plan, + )) + dry_run = self.action_engine.dry_run(plan) + execution = self.action_engine.execute(plan) + execute_payload = { + "dry_run": dry_run.status, + **execution.to_dict(), + } + transition = "authorized->external_execute" + else: + execution = ExecutionResult( + execution_id="", + idempotency_key="", + adapter=request.adapter, + status="blocked", + error="autonomy decision did not authorize L3 execution", + ) + execute_payload = execution.to_dict() + transition = "halt->human_review" + record( + "execute", + execute_payload, + evidence_version=evidence.version_id, + policy_version=decision.policy_version, + authorization_identity=action.authorization_identity, + idempotency_key=execution.idempotency_key, + transition=transition, + ) + + if plan is not None and execution.status in {"success", "noop"}: + verification = self.verification_engine.verify( + plan, + execution, + source_path=request.source_path, + source_table=request.source_table, + target_table=request.target_table, + dialect=request.dialect, + runtime_observed=request.runtime_observed, + ) + verification_payload = verification.to_dict() + if verification.outcome == "failed" and execution.status == "success": + rollback = self.action_engine.rollback_execution(plan, execution) + verification_payload["rollback"] = { + "status": rollback.status, + "verified": rollback.verified, + "restored": list(rollback.restored), + "error": rollback.error, + } + else: + verification = VerificationReport( + code=LayerResult("code", "not_run"), + structure=LayerResult("structure", "not_run"), + runtime=LayerResult("runtime", "not_run"), + ) + verification_payload = verification.to_dict() + record( + "verify", + verification_payload, + evidence_version=evidence.version_id, + transition="execute->reobserve" if plan else "", + ) + + if execution.status == "blocked": + outcome = "held_for_human_review" + else: + outcome = verification.outcome + record( + "learn", + { + "outcome": outcome, + "writeback": "signals_only", + "authorization_expanded": False, + }, + evidence_version=evidence.version_id, + ) + return GovernanceResult( + task_id=request.task_id, + outcome=outcome, + evidence=evidence, + decision=decision, + action_plan=plan, + execution=execution, + verification=verification, + events=tuple(events), + audit_path=str(Path(self.audit_log.path).resolve()), + ) diff --git a/sqlgraph/reasoning/scenario.py b/sqlgraph/reasoning/scenario.py new file mode 100644 index 0000000..cf58022 --- /dev/null +++ b/sqlgraph/reasoning/scenario.py @@ -0,0 +1,200 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Load declarative governance scenarios and export their evidence packages.""" + +from __future__ import annotations + +import hashlib +import json +import shutil +from pathlib import Path + +import yaml + +from sqlgraph.actions import ( + ActionEngine, + DuckDBTaskAdapter, + SqlFilePatchAdapter, +) +from sqlgraph.audit import AuditLog +from sqlgraph.autonomy import ( + AuthorizationScope, + GovernanceAction, + ReversibilityEvidence, +) +from sqlgraph.baseline import build_baseline +from sqlgraph.evidence import EvidenceEngine, EvidenceRequest +from sqlgraph.input import SqlSource +from sqlgraph.reasoning.runner import ( + GovernanceRequest, + GovernanceResult, + GovernanceRunner, +) +from sqlgraph.serialize.json_output import to_json +from sqlgraph.verification import VerificationEngine + + +def _write_json(path: Path, payload: dict) -> None: + path.write_text( + json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + + +def load_scenario(path: str | Path) -> dict: + scenario_path = Path(path).resolve() + payload = yaml.safe_load(scenario_path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError("scenario must be a YAML object") + required = { + "task_id", + "intent", + "dialect", + "sql_file", + "source_table", + "target_table", + "action", + } + missing = sorted(required - payload.keys()) + if missing: + raise ValueError(f"scenario is missing fields: {', '.join(missing)}") + payload["_scenario_path"] = str(scenario_path) + return payload + + +def run_scenario( + scenario_path: str | Path, + output_dir: str | Path, +) -> GovernanceResult: + scenario = load_scenario(scenario_path) + source_scenario = Path(scenario["_scenario_path"]) + output = Path(output_dir).resolve() + workspace = output / "workspace" + workspace.mkdir(parents=True, exist_ok=True) + source_sql = (source_scenario.parent / scenario["sql_file"]).resolve() + working_sql = workspace / source_sql.name + shutil.copy2(source_sql, working_sql) + + source = SqlSource.from_file(str(working_sql)) + baseline = build_baseline( + source, + dialect=scenario["dialect"], + parameters=scenario.get("parameters", {}), + udf_manifest=scenario.get("udf_manifest", {}), + scheduler_manifest=scenario.get("scheduler_manifest", {}), + ) + from sqlgraph.api import build_graph + + graph = build_graph(source, dialect=scenario["dialect"]) + graph.metadata["baseline_id"] = baseline.baseline_id + action_data = scenario["action"] + reversibility = action_data["reversibility"] + action = GovernanceAction( + action_type=action_data["type"], + evidence_version="pending", + evidence_grounded=False, + reversibility=ReversibilityEvidence( + state_restorable=bool(reversibility["state_restorable"]), + external_effects_controlled=bool( + reversibility["external_effects_controlled"] + ), + rollback_verified=bool(reversibility["rollback_verified"]), + references=tuple(reversibility.get("references", ())), + ), + authorization_scope=AuthorizationScope( + action_data.get("authorization_scope", "none") + ), + authorization_identity=action_data.get("authorization_identity", ""), + blast_radius=float(action_data.get("blast_radius", 0.0)), + object_risk=float(action_data.get("object_risk", 0.0)), + historical_reliability=float( + action_data.get("historical_reliability", 1.0) + ), + roi=float(action_data.get("roi", 0.0)), + ) + before = working_sql.read_text(encoding="utf-8") + after = action_data.get("after", before) + operations = ({ + "path": str(working_sql), + "before": before, + "after": after, + "expected_sha256": hashlib.sha256(before.encode("utf-8")).hexdigest(), + },) + audit_path = output / "audit.jsonl" + if audit_path.exists(): + audit_path.unlink() + runner = GovernanceRunner( + EvidenceEngine(graph), + ActionEngine([SqlFilePatchAdapter(), DuckDBTaskAdapter()]), + VerificationEngine(), + AuditLog(audit_path), + ) + request = GovernanceRequest( + task_id=scenario["task_id"], + baseline=baseline, + graph=graph, + evidence_request=EvidenceRequest( + task_id=scenario["task_id"], + baseline_id=baseline.baseline_id, + intent=scenario["intent"], + anchors=( + scenario["source_table"], + scenario["target_table"], + ), + direction="both", + max_depth=int(scenario.get("max_depth", 2)), + coverage_obligations=( + "anchors_resolved", + "lineage_path", + "counterevidence_checked", + "no_unresolved", + ), + ), + action=action, + adapter=action_data.get("adapter", "sql_file_patch"), + operations=operations, + source_path=str(working_sql), + source_table=scenario["source_table"], + target_table=scenario["target_table"], + dialect=scenario["dialect"], + runtime_observed=scenario.get("runtime"), + ) + result = runner.run(request) + _write_json(output / "baseline.json", baseline.to_dict()) + _write_json(output / "graph.json", to_json(graph)) + _write_json(output / "evidence.json", result.evidence.to_dict()) + _write_json(output / "decision.json", result.decision.to_dict()) + _write_json(output / "verification.json", result.verification.to_dict()) + _write_json(output / "result.json", result.to_dict()) + return result + + +def verify_scenario_output(output_dir: str | Path) -> dict: + output = Path(output_dir).resolve() + required = ( + "baseline.json", + "graph.json", + "evidence.json", + "decision.json", + "verification.json", + "result.json", + "audit.jsonl", + ) + missing = [name for name in required if not (output / name).is_file()] + integrity = AuditLog(output / "audit.jsonl").verify_integrity() + result = ( + json.loads((output / "result.json").read_text(encoding="utf-8")) + if not missing + else {} + ) + return { + "valid": not missing and integrity.valid, + "missing": missing, + "integrity": { + "valid": integrity.valid, + "event_count": integrity.event_count, + "errors": list(integrity.errors), + }, + "outcome": result.get("outcome"), + } diff --git a/sqlgraph/serialize/graphrag.py b/sqlgraph/serialize/graphrag.py index 2b52986..904a4c6 100644 --- a/sqlgraph/serialize/graphrag.py +++ b/sqlgraph/serialize/graphrag.py @@ -26,6 +26,21 @@ from sqlgraph.utils.logging import log_info +def evidence_to_context(evidence) -> Dict[str, Any]: + """Serialize only the graph facts allowed by an evidence bundle.""" + return { + "baseline_id": evidence.baseline_id, + "evidence_hash": evidence.subgraph_hash, + "intent": evidence.intent, + "anchors": list(evidence.anchors), + "allowed_node_ids": list(evidence.included_nodes), + "allowed_edge_ids": list(evidence.included_edges), + "coverage_contract": evidence.coverage_contract, + "gaps": list(evidence.gaps), + "residual_unknowns": list(evidence.residual_unknowns), + } + + def to_graphrag(graph: PropertyGraph, output_path: str) -> Dict[str, Any]: """将图输出为 GraphRAG entity/relation payload 格式(schema v2) diff --git a/sqlgraph/serve/server.py b/sqlgraph/serve/server.py index 56275f7..f687f8a 100644 --- a/sqlgraph/serve/server.py +++ b/sqlgraph/serve/server.py @@ -23,7 +23,7 @@ from sqlgraph.serve.stats import AnalysisSnapshot, build_index_stats, table_stats from sqlgraph.playground import graph_to_playground_payload, find_free_port -_WEB_DIR = os.path.join(os.path.dirname(__file__), "web") +_WEB_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "web") _STATIC_DIR = os.path.join(_WEB_DIR, "static") _ENV = Environment(loader=FileSystemLoader(_WEB_DIR), autoescape=select_autoescape(["html", "j2"])) diff --git a/sqlgraph/serve/theme.py b/sqlgraph/serve/theme.py new file mode 100644 index 0000000..0c2c442 --- /dev/null +++ b/sqlgraph/serve/theme.py @@ -0,0 +1,16 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Shared visual assets for Explorer-derived offline reports.""" + +from __future__ import annotations + +from pathlib import Path + + +STATIC_DIR = Path(__file__).resolve().parent / "web" / "static" + + +def load_explorer_css() -> str: + """Load the canonical GitHub Explorer stylesheet.""" + return (STATIC_DIR / "app.css").read_text(encoding="utf-8") diff --git a/sqlgraph/verification/__init__.py b/sqlgraph/verification/__init__.py new file mode 100644 index 0000000..e531525 --- /dev/null +++ b/sqlgraph/verification/__init__.py @@ -0,0 +1,18 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Public independent verification interface.""" + +from sqlgraph.verification.engine import VerificationEngine +from sqlgraph.verification.model import ( + VERIFICATION_SCHEMA_VERSION, + LayerResult, + VerificationReport, +) + +__all__ = [ + "VERIFICATION_SCHEMA_VERSION", + "LayerResult", + "VerificationEngine", + "VerificationReport", +] diff --git a/sqlgraph/verification/engine.py b/sqlgraph/verification/engine.py new file mode 100644 index 0000000..949ae64 --- /dev/null +++ b/sqlgraph/verification/engine.py @@ -0,0 +1,120 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Verify actual code, independently rebuilt structure, and runtime effects.""" + +from __future__ import annotations + +from pathlib import Path + +from sqlgraph.actions import ActionPlan, ExecutionResult +from sqlgraph.api import build_graph +from sqlgraph.lineage import drilldown +from sqlgraph.verification.model import LayerResult, VerificationReport + + +class VerificationEngine: + def verify( + self, + plan: ActionPlan, + execution: ExecutionResult, + *, + source_path: str | Path, + source_table: str, + target_table: str, + dialect: str | None = None, + runtime_observed: dict | None = None, + ) -> VerificationReport: + return VerificationReport( + code=self.verify_code(plan, execution), + structure=self.verify_structure( + source_path, + source_table, + target_table, + dialect=dialect, + ), + runtime=self.verify_runtime(runtime_observed), + ) + + def verify_code( + self, + plan: ActionPlan, + execution: ExecutionResult, + ) -> LayerResult: + if execution.status not in {"success", "noop"}: + return LayerResult( + "code", + "fail", + evidence={"execution_status": execution.status}, + note="the planned change did not complete", + ) + mismatches = [] + checked = [] + for operation in plan.operations: + raw_path = operation.get("path") + expected = operation.get("after") + if not raw_path or expected is None: + continue + path = Path(raw_path) + content = path.read_text(encoding="utf-8") if path.is_file() else "" + checked.append(str(path)) + if content.count(str(expected)) != 1: + mismatches.append(str(path)) + return LayerResult( + "code", + "fail" if mismatches else "pass", + evidence={ + "checked_paths": checked, + "mismatches": mismatches, + "execution_id": execution.execution_id, + }, + note="actual files were compared with the approved action plan", + ) + + def verify_structure( + self, + source_path: str | Path, + source_table: str, + target_table: str, + *, + dialect: str | None = None, + ) -> LayerResult: + path = Path(source_path) + if not path.exists(): + return LayerResult( + "structure", + "fail", + evidence={"source_path": str(path)}, + note="changed source is missing", + ) + graph = build_graph(str(path), dialect=dialect) + evidence = drilldown(graph, source_table, target_table) + return LayerResult( + "structure", + "pass" if evidence.get("found") else "fail", + evidence={ + "rebuilt_from_source": True, + "source_path": str(path.resolve()), + "graph_environment": graph.metadata.get("environment", {}), + "lineage": evidence, + }, + note="structure was rebuilt from the changed source", + ) + + def verify_runtime(self, observed: dict | None) -> LayerResult: + if observed is None: + return LayerResult( + "runtime", + "not_run", + note="runtime verification was not executed", + ) + checks_passed = bool(observed.get("checks_passed", False)) + reports_recomputed = bool(observed.get("reports_recomputed", False)) + new_alerts = int(observed.get("new_alerts", 0)) + passed = checks_passed and reports_recomputed and new_alerts == 0 + return LayerResult( + "runtime", + "pass" if passed else "fail", + evidence=dict(observed), + note="runtime effects were checked independently", + ) diff --git a/sqlgraph/verification/model.py b/sqlgraph/verification/model.py new file mode 100644 index 0000000..9aeb0fd --- /dev/null +++ b/sqlgraph/verification/model.py @@ -0,0 +1,55 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Independent verification result contracts.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from typing import Any + + +VERIFICATION_SCHEMA_VERSION = "verification-report-v1" + + +@dataclass(frozen=True) +class LayerResult: + layer: str + status: str + evidence: dict[str, Any] = field(default_factory=dict) + note: str = "" + + def __post_init__(self) -> None: + if self.status not in {"pass", "fail", "not_run"}: + raise ValueError("status must be pass, fail, or not_run") + + +@dataclass(frozen=True) +class VerificationReport: + code: LayerResult + structure: LayerResult + runtime: LayerResult + schema_version: str = VERIFICATION_SCHEMA_VERSION + + @property + def outcome(self) -> str: + layers = (self.code, self.structure, self.runtime) + if any(layer.status == "fail" for layer in layers): + return "failed" + if any(layer.status == "not_run" for layer in layers): + return "incomplete" + return "success" + + @property + def closed_loop_status(self) -> str: + return self.outcome + + def to_dict(self) -> dict: + return { + "schema_version": self.schema_version, + "code": asdict(self.code), + "structure": asdict(self.structure), + "runtime": asdict(self.runtime), + "outcome": self.outcome, + "closed_loop_status": self.outcome, + } diff --git a/sqlgraph/verify/__init__.py b/sqlgraph/verify/__init__.py new file mode 100644 index 0000000..f88fa99 --- /dev/null +++ b/sqlgraph/verify/__init__.py @@ -0,0 +1,65 @@ +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Compatibility verification helpers for graph-only callers.""" + +from __future__ import annotations + +from sqlgraph.lineage import drilldown +from sqlgraph.verification import LayerResult, VerificationReport + + +def verify( + graph, + source_table: str, + target_table: str, + governance_issue=None, + runtime_observed: dict | None = None, + runtime_target: float | None = None, +) -> VerificationReport: + lineage = drilldown(graph, source_table, target_table) + structure = LayerResult( + "structure", + "pass" if lineage.get("found") else "fail", + evidence=lineage, + ) + contract = LayerResult( + "code", + "fail" if governance_issue is not None else "pass", + evidence=( + governance_issue.to_dict() + if hasattr(governance_issue, "to_dict") + else governance_issue or {} + ), + note="legacy contract check exposed through the code layer", + ) + if runtime_observed is None: + runtime = LayerResult("runtime", "not_run") + else: + value = runtime_observed.get("value") + passed = ( + runtime_observed.get("reports_recomputed", True) + and runtime_observed.get("new_alerts", 0) == 0 + ) + evidence = dict(runtime_observed) + if runtime_target is not None and value is not None: + relative_error = ( + abs(value - runtime_target) / abs(runtime_target) + if runtime_target + else float("inf") + ) + evidence["relative_error"] = relative_error + passed = passed and relative_error <= 0.05 + runtime = LayerResult( + "runtime", + "pass" if passed else "fail", + evidence=evidence, + ) + return VerificationReport( + code=contract, + structure=structure, + runtime=runtime, + ) + + +__all__ = ["LayerResult", "VerificationReport", "verify"] diff --git a/tests/contract/__init__.py b/tests/contract/__init__.py new file mode 100644 index 0000000..d65803f --- /dev/null +++ b/tests/contract/__init__.py @@ -0,0 +1 @@ +"""Persisted artifact contract tests.""" diff --git a/tests/contract/test_baseline_manifest.py b/tests/contract/test_baseline_manifest.py new file mode 100644 index 0000000..99af158 --- /dev/null +++ b/tests/contract/test_baseline_manifest.py @@ -0,0 +1,58 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import jsonschema + +from sqlgraph.baseline import build_baseline +from sqlgraph.input import SqlSource +from sqlgraph.input.csv_schema import SchemaRegistry + + +def _source() -> SqlSource: + return SqlSource.from_string( + "INSERT INTO dst SELECT id FROM src", + name="build_dst", + ) + + +def test_baseline_id_is_stable_and_timestamp_is_not_identity(): + first = build_baseline(_source(), dialect="spark") + second = build_baseline(_source(), dialect="spark") + + assert first.baseline_id == second.baseline_id + assert first.created_at + assert first.to_dict()["baseline_id"] == first.baseline_id + + +def test_missing_schema_and_external_inputs_are_disclosed(): + baseline = build_baseline(_source(), dialect="spark") + + assert set(baseline.missing_dependencies) == { + "schema", + "udf_manifest", + "parameters", + "scheduler_manifest", + } + + +def test_schema_changes_baseline_identity(): + first = SchemaRegistry.from_dict({"src": ["id"]}) + second = SchemaRegistry.from_dict({"src": ["id", "amount"]}) + + assert build_baseline(_source(), dialect="spark", schema=first).baseline_id != ( + build_baseline(_source(), dialect="spark", schema=second).baseline_id + ) + + +def test_baseline_matches_json_schema(): + baseline = build_baseline(_source(), dialect="spark") + schema_path = ( + Path(__file__).parents[2] + / "schemas" + / "baseline-manifest-v1.schema.json" + ) + schema = json.loads(schema_path.read_text(encoding="utf-8")) + + jsonschema.validate(baseline.to_dict(), schema) diff --git a/tests/contract/test_capabilities.py b/tests/contract/test_capabilities.py new file mode 100644 index 0000000..193fad0 --- /dev/null +++ b/tests/contract/test_capabilities.py @@ -0,0 +1,11 @@ +from __future__ import annotations + +from pathlib import Path + +from tools.check_capabilities import check_capabilities + + +def test_implemented_capabilities_reference_existing_assets(): + root = Path(__file__).parents[2] + + assert check_capabilities(root) == [] diff --git a/tests/golden/__init__.py b/tests/golden/__init__.py new file mode 100644 index 0000000..21c7c66 --- /dev/null +++ b/tests/golden/__init__.py @@ -0,0 +1,2 @@ +# tests/golden/__init__.py +"""Golden 回归数据集与快照(对应需求规格书 §14.2 DS-01~DS-06)。""" diff --git a/tests/golden/_harness.py b/tests/golden/_harness.py new file mode 100644 index 0000000..1380661 --- /dev/null +++ b/tests/golden/_harness.py @@ -0,0 +1,124 @@ +# tests/golden/_harness.py +"""Golden 回归公共工具(REQ-ARCH-02 / REQ-CLI-02 TC2)。 + +把 ``tests/golden/fixtures/*.sql`` 构建成图,导出一份**规范化结构签名**并与 +``tests/golden/snapshots/*.json`` 基线比对。产物漂移即测试失败并输出可读 diff。 + +规范化签名的设计取舍: +- 纳入:节点/边总数、按类型计数、全量节点 ID(含类型与名字)、全量边 + (source->target:type)、table_lineage 的表名对、覆盖报告。 + 这些字段完整刻画「图的确定性结构」,任何真实结构漂移都会改变签名。 +- 剔除:环境指纹里的 ``parser_version``(内嵌 sqlglot 版本号),它随运行环境 + 合法变化,且已由 REQ-ARCH-02 AC2 单独覆盖;纳入签名会让 golden 在纯粹的 + 依赖升级下误报。身份规则版本 ``identity_rule_version`` 仍纳入,因为它一旦 + 变化就意味着 ID 派生规则改变,正是应当触发 rebless 的信号。 + +维护方式:图确有预期变化时,运行 ``python -m tests.golden.regen`` 重新固化 +基线(rebless),并在 PR 中说明变更的 REQ 编号。 +""" +from __future__ import annotations + +import json +from pathlib import Path + +from sqlgraph.api import build_graph + +_HERE = Path(__file__).resolve().parent +FIXTURES_DIR = _HERE / "fixtures" +SNAPSHOTS_DIR = _HERE / "snapshots" +DIALECT = "spark" + + +def list_fixtures() -> list[Path]: + """返回全部 golden fixture(按文件名排序,保证确定顺序)。""" + return sorted(FIXTURES_DIR.glob("*.sql")) + + +def build_fixture_graph(path: Path): + """构建单个 fixture 的图。""" + sql = path.read_text(encoding="utf-8") + return build_graph(sql, dialect=DIALECT) + + +def _name_map(graph) -> dict: + return {n.id: n.name for n in graph.nodes} + + +def canonical_signature(graph) -> dict: + """把图规范化为可 JSON 序列化、可稳定 diff 的结构签名。""" + nm = _name_map(graph) + + nodes = sorted( + ( + { + "id": n.id, + "type": getattr(n.node_type, "value", str(n.node_type)), + "name": n.name, + } + for n in graph.nodes + ), + key=lambda d: d["id"], + ) + edges = sorted( + ( + { + "id": e.id, + "source": e.source_id, + "target": e.target_id, + "type": getattr(e.edge_type, "value", str(e.edge_type)), + } + for e in graph.edges + ), + key=lambda d: (d["id"], d["source"], d["target"], d["type"]), + ) + + by_node_type: dict[str, int] = {} + for n in nodes: + by_node_type[n["type"]] = by_node_type.get(n["type"], 0) + 1 + by_edge_type: dict[str, int] = {} + for e in edges: + by_edge_type[e["type"]] = by_edge_type.get(e["type"], 0) + 1 + + table_lineage = sorted( + [nm.get(e["source"]), nm.get(e["target"])] + for e in edges + if e["type"] == "table_lineage" + ) + + coverage = graph.metadata.get("coverage", {}) + identity_rule_version = graph.metadata.get("environment", {}).get( + "identity_rule_version" + ) + + return { + "counts": { + "nodes": len(nodes), + "edges": len(edges), + "by_node_type": dict(sorted(by_node_type.items())), + "by_edge_type": dict(sorted(by_edge_type.items())), + }, + "identity_rule_version": identity_rule_version, + "nodes": nodes, + "edges": edges, + "table_lineage": table_lineage, + "coverage": coverage, + } + + +def snapshot_path(fixture: Path) -> Path: + """fixture 对应的基线快照文件路径。""" + return SNAPSHOTS_DIR / f"{fixture.stem}.json" + + +def dumps(signature: dict) -> str: + """规范化 JSON 文本(排序键、UTF-8、末尾换行)。""" + return json.dumps(signature, ensure_ascii=False, indent=2, sort_keys=True) + "\n" + + +def load_snapshot(fixture: Path) -> dict: + return json.loads(snapshot_path(fixture).read_text(encoding="utf-8")) + + +def write_snapshot(fixture: Path, signature: dict) -> None: + SNAPSHOTS_DIR.mkdir(parents=True, exist_ok=True) + snapshot_path(fixture).write_text(dumps(signature), encoding="utf-8") diff --git a/tests/golden/fixtures/ds01_basic_lineage.sql b/tests/golden/fixtures/ds01_basic_lineage.sql new file mode 100644 index 0000000..0b34e6d --- /dev/null +++ b/tests/golden/fixtures/ds01_basic_lineage.sql @@ -0,0 +1,10 @@ +-- DS-01 基础血缘:单语句 SELECT/INSERT/JOIN。 +-- 覆盖 reads_from / writes_to / has_column / table_lineage 四类基础关系。 +INSERT INTO dws_order_enriched_di +SELECT + o.order_id, + o.user_id, + u.user_name, + o.amount +FROM dwd_order_di o +JOIN dim_user u ON o.user_id = u.user_id diff --git a/tests/golden/fixtures/ds02_expr_dag.sql b/tests/golden/fixtures/ds02_expr_dag.sql new file mode 100644 index 0000000..21de712 --- /dev/null +++ b/tests/golden/fixtures/ds02_expr_dag.sql @@ -0,0 +1,9 @@ +-- DS-02 表达式 DAG:嵌套算术 / CASE / 函数。 +-- 覆盖 contains / compute_dependency / produces / expr_operand 四类计算关系。 +INSERT INTO dws_user_value_di +SELECT + user_id, + (base_score + bonus_score) * weight AS total_score, + CASE WHEN active_days >= 7 THEN 'active' ELSE 'inactive' END AS status, + COALESCE(pay_amount, 0) + COALESCE(refund_amount, 0) AS net_amount +FROM dwd_user_metric_di diff --git a/tests/golden/fixtures/ds03_multi_stmt.sql b/tests/golden/fixtures/ds03_multi_stmt.sql new file mode 100644 index 0000000..bd9cb88 --- /dev/null +++ b/tests/golden/fixtures/ds03_multi_stmt.sql @@ -0,0 +1,5 @@ +-- DS-03 多语句文件:两条以上互不相关语句。 +-- 用于验证「无笛卡尔伪边」(REQ-LIN-01):只应产出 s1->d1、s2->d2, +-- 不得因同处一个文件而产生跨语句伪边。 +INSERT INTO dwd_click_di SELECT log_id, ad_id, ts FROM ods_click_log_di; +INSERT INTO dwd_impression_di SELECT log_id, ad_id, ts FROM ods_impression_log_di; diff --git a/tests/golden/fixtures/ds04_cte_subquery.sql b/tests/golden/fixtures/ds04_cte_subquery.sql new file mode 100644 index 0000000..565ca90 --- /dev/null +++ b/tests/golden/fixtures/ds04_cte_subquery.sql @@ -0,0 +1,20 @@ +-- DS-04 CTE 与子查询:CTE 作为 table 节点、子查询字段血缘。 +INSERT INTO dws_ad_daily_di +WITH click_agg AS ( + SELECT ad_id, dt, COUNT(*) AS click_cnt + FROM dwd_click_di + GROUP BY ad_id, dt +), +imp_agg AS ( + SELECT ad_id, dt, COUNT(*) AS imp_cnt + FROM dwd_impression_di + GROUP BY ad_id, dt +) +SELECT + c.ad_id, + c.dt, + c.click_cnt, + i.imp_cnt, + c.click_cnt / i.imp_cnt AS ctr +FROM click_agg c +JOIN imp_agg i ON c.ad_id = i.ad_id AND c.dt = i.dt diff --git a/tests/golden/fixtures/ds06_whitespace_variant.sql b/tests/golden/fixtures/ds06_whitespace_variant.sql new file mode 100644 index 0000000..8d95963 --- /dev/null +++ b/tests/golden/fixtures/ds06_whitespace_variant.sql @@ -0,0 +1,6 @@ +-- DS-06 确定性样本:与 ds01 语义相同,仅空白 / 大小写不同。 +-- 用于验证身份稳定(REQ-ID-01):关键节点 ID 应与规范写法一致。 +insert into dws_order_enriched_di +SELECT o.order_id, o.user_id, u.user_name,o.amount +FROM dwd_order_di o +JOIN dim_user u ON o.user_id=u.user_id diff --git a/tests/golden/regen.py b/tests/golden/regen.py new file mode 100644 index 0000000..bf6fd7c --- /dev/null +++ b/tests/golden/regen.py @@ -0,0 +1,32 @@ +# tests/golden/regen.py +"""重新固化 golden 基线(rebless)。 + +仅当图确有**预期**变化时运行,并在 PR 中标注涉及的 REQ 编号: + + python -m tests.golden.regen + +它会为 ``fixtures/`` 下每个 SQL 重新生成规范化结构签名并写入 ``snapshots/``。 +日常 CI 绝不自动运行本脚本——那会让 golden 回归失去意义。 +""" +from __future__ import annotations + +from tests.golden import _harness as h + + +def main() -> int: + fixtures = h.list_fixtures() + if not fixtures: + print("未发现任何 golden fixture。") + return 1 + for fx in fixtures: + graph = h.build_fixture_graph(fx) + sig = h.canonical_signature(graph) + h.write_snapshot(fx, sig) + print(f"已固化基线: {fx.name} -> {h.snapshot_path(fx).name} " + f"(nodes={sig['counts']['nodes']}, edges={sig['counts']['edges']})") + print(f"共固化 {len(fixtures)} 个基线。") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/golden/snapshots/ds01_basic_lineage.json b/tests/golden/snapshots/ds01_basic_lineage.json new file mode 100644 index 0000000..51213cb --- /dev/null +++ b/tests/golden/snapshots/ds01_basic_lineage.json @@ -0,0 +1,202 @@ +{ + "counts": { + "by_edge_type": { + "compute_dependency": 4, + "has_column": 8, + "reads_from": 2, + "table_lineage": 2, + "writes_to": 1 + }, + "by_node_type": { + "column": 8, + "sql": 1, + "table": 3 + }, + "edges": 17, + "nodes": 12 + }, + "coverage": { + "failed": 0, + "ok": 1, + "partial": 0, + "reasons": [], + "total": 1 + }, + "edges": [ + { + "id": "e_15d2320d7e01914a5c34ec32", + "source": "sql_04fadf4c73b33358f6edf51e", + "target": "tbl_c4d0cac04b1ef5b337d8f7b8", + "type": "reads_from" + }, + { + "id": "e_1d63b8dd4a250242f079aaae", + "source": "col_cd3c51d4f0369696ed1edcd1", + "target": "col_78eaf5857233bb51de95a97b", + "type": "compute_dependency" + }, + { + "id": "e_2cd22f1d9f3684e0292e1f06", + "source": "col_9a4a7bfc1637f9344131553c", + "target": "col_83efbc91d74d432a112f5f02", + "type": "compute_dependency" + }, + { + "id": "e_3123b8adbe6bddfb498d740e", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "col_cd3c51d4f0369696ed1edcd1", + "type": "has_column" + }, + { + "id": "e_39c08f9f482b58b60841a3b9", + "source": "tbl_1b6a4c53bdb6aa04297ca4c8", + "target": "tbl_26f98d454abeb56dd8a6f492", + "type": "table_lineage" + }, + { + "id": "e_41f952bb940e58d30fd3304c", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "col_0f782de896cd1690b6f6929f", + "type": "has_column" + }, + { + "id": "e_489e1c4c7fc7c424686687e9", + "source": "sql_04fadf4c73b33358f6edf51e", + "target": "tbl_26f98d454abeb56dd8a6f492", + "type": "writes_to" + }, + { + "id": "e_552cf08442978e1eace263a9", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_78eaf5857233bb51de95a97b", + "type": "has_column" + }, + { + "id": "e_5c63ed749bfc8f74d9c2aa5e", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_0d14b47f2feed2e7c1c488da", + "type": "has_column" + }, + { + "id": "e_9db622eb7f6a3f9fc8aa9854", + "source": "col_7d27d8740225ab319f84a43c", + "target": "col_fd1a1a150af28166e93a9fb7", + "type": "compute_dependency" + }, + { + "id": "e_b22b05c4e88a23514a360267", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_83efbc91d74d432a112f5f02", + "type": "has_column" + }, + { + "id": "e_bcd6ddbd0d8ff09e11465b70", + "source": "col_0f782de896cd1690b6f6929f", + "target": "col_0d14b47f2feed2e7c1c488da", + "type": "compute_dependency" + }, + { + "id": "e_bf3a8067565322a3875e6012", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "col_7d27d8740225ab319f84a43c", + "type": "has_column" + }, + { + "id": "e_c12fbe9067e0f91af05e74df", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "tbl_26f98d454abeb56dd8a6f492", + "type": "table_lineage" + }, + { + "id": "e_c6737114061e0da0e85697dd", + "source": "tbl_1b6a4c53bdb6aa04297ca4c8", + "target": "col_9a4a7bfc1637f9344131553c", + "type": "has_column" + }, + { + "id": "e_d93bbd7043ce74ee174cdc00", + "source": "sql_04fadf4c73b33358f6edf51e", + "target": "tbl_1b6a4c53bdb6aa04297ca4c8", + "type": "reads_from" + }, + { + "id": "e_ef1a03d8b204f5514d9ff131", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_fd1a1a150af28166e93a9fb7", + "type": "has_column" + } + ], + "identity_rule_version": "id-v2", + "nodes": [ + { + "id": "col_0d14b47f2feed2e7c1c488da", + "name": "order_id", + "type": "column" + }, + { + "id": "col_0f782de896cd1690b6f6929f", + "name": "order_id", + "type": "column" + }, + { + "id": "col_78eaf5857233bb51de95a97b", + "name": "amount", + "type": "column" + }, + { + "id": "col_7d27d8740225ab319f84a43c", + "name": "user_id", + "type": "column" + }, + { + "id": "col_83efbc91d74d432a112f5f02", + "name": "user_name", + "type": "column" + }, + { + "id": "col_9a4a7bfc1637f9344131553c", + "name": "user_name", + "type": "column" + }, + { + "id": "col_cd3c51d4f0369696ed1edcd1", + "name": "amount", + "type": "column" + }, + { + "id": "col_fd1a1a150af28166e93a9fb7", + "name": "user_id", + "type": "column" + }, + { + "id": "sql_04fadf4c73b33358f6edf51e", + "name": "inline_sql", + "type": "sql" + }, + { + "id": "tbl_1b6a4c53bdb6aa04297ca4c8", + "name": "dim_user", + "type": "table" + }, + { + "id": "tbl_26f98d454abeb56dd8a6f492", + "name": "dws_order_enriched_di", + "type": "table" + }, + { + "id": "tbl_c4d0cac04b1ef5b337d8f7b8", + "name": "dwd_order_di", + "type": "table" + } + ], + "table_lineage": [ + [ + "dim_user", + "dws_order_enriched_di" + ], + [ + "dwd_order_di", + "dws_order_enriched_di" + ] + ] +} diff --git a/tests/golden/snapshots/ds02_expr_dag.json b/tests/golden/snapshots/ds02_expr_dag.json new file mode 100644 index 0000000..a4aa2c8 --- /dev/null +++ b/tests/golden/snapshots/ds02_expr_dag.json @@ -0,0 +1,361 @@ +{ + "counts": { + "by_edge_type": { + "compute_dependency": 12, + "contains": 3, + "expr_operand": 4, + "has_column": 11, + "produces": 3, + "reads_from": 1, + "table_lineage": 1, + "writes_to": 1 + }, + "by_node_type": { + "column": 11, + "sql": 1, + "table": 2, + "transform": 7 + }, + "edges": 36, + "nodes": 21 + }, + "coverage": { + "failed": 0, + "ok": 1, + "partial": 0, + "reasons": [], + "total": 1 + }, + "edges": [ + { + "id": "e_021306f738dcf9883fa20c90", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_a23914f6b90d25b34b044275", + "type": "has_column" + }, + { + "id": "e_0ef9a9ff24ce0cf8848f6383", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_06441fde90754757b5779ff8", + "type": "has_column" + }, + { + "id": "e_12e82a860777ffd438e21808", + "source": "tbl_85ac6b96dd07668776b1d8d0", + "target": "col_ff170de8698fcb6322f09c03", + "type": "has_column" + }, + { + "id": "e_1dba84392366e8a82a9de95c", + "source": "expr_68cb1c8d5cca5b5bbaef2abf", + "target": "expr_b939fff64b4305a6be94666d", + "type": "expr_operand" + }, + { + "id": "e_21ea0f639c994f553d9e81b5", + "source": "tbl_85ac6b96dd07668776b1d8d0", + "target": "col_8dccb2b9f470ee0ecba471d1", + "type": "has_column" + }, + { + "id": "e_27ee14f0872f9e6fa17962a7", + "source": "sql_9066d2306662ed7af1d2b961", + "target": "expr_b939fff64b4305a6be94666d", + "type": "contains" + }, + { + "id": "e_2dd8e02b2ab8b1103bd53b2a", + "source": "col_5e0f941f7f808d1ca3365fd9", + "target": "expr_488bc78483e98125d5304322", + "type": "compute_dependency" + }, + { + "id": "e_2e7e7ab2bc3986f2c050d01f", + "source": "tbl_85ac6b96dd07668776b1d8d0", + "target": "col_85e98bfdc4fe7d6e5f3708eb", + "type": "has_column" + }, + { + "id": "e_2fde1e1265c72ca7038cf24c", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_149da087f5665bf7e61e4c1c", + "type": "has_column" + }, + { + "id": "e_3bda680c89ae550062f43d8b", + "source": "expr_812be1f3f3d52bd14a76039a", + "target": "expr_488bc78483e98125d5304322", + "type": "expr_operand" + }, + { + "id": "e_3c2fa00f5348447844b686b2", + "source": "col_a23914f6b90d25b34b044275", + "target": "expr_812be1f3f3d52bd14a76039a", + "type": "compute_dependency" + }, + { + "id": "e_509aab5983b1cb923e0070b4", + "source": "sql_9066d2306662ed7af1d2b961", + "target": "expr_1ea1f4c4f29fc1dbdcb93e5d", + "type": "contains" + }, + { + "id": "e_52672dba211992297dea4bc9", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "tbl_85ac6b96dd07668776b1d8d0", + "type": "table_lineage" + }, + { + "id": "e_55546c2d70b1f4bdef6d58c5", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_e77b072519bb386d30d7cfaa", + "type": "has_column" + }, + { + "id": "e_5d710a8307dab802db300bfb", + "source": "col_06441fde90754757b5779ff8", + "target": "expr_b939fff64b4305a6be94666d", + "type": "compute_dependency" + }, + { + "id": "e_6120a29442d45ee25afa3158", + "source": "col_06441fde90754757b5779ff8", + "target": "expr_68cb1c8d5cca5b5bbaef2abf", + "type": "compute_dependency" + }, + { + "id": "e_68aa31b63cad7b5fce7d499e", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_5e0f941f7f808d1ca3365fd9", + "type": "has_column" + }, + { + "id": "e_6b53dcfee604c0bf7ac80e34", + "source": "col_e77b072519bb386d30d7cfaa", + "target": "expr_1ea1f4c4f29fc1dbdcb93e5d", + "type": "compute_dependency" + }, + { + "id": "e_6fdab75d13882dc97667159c", + "source": "sql_9066d2306662ed7af1d2b961", + "target": "tbl_85ac6b96dd07668776b1d8d0", + "type": "writes_to" + }, + { + "id": "e_72b69ebb962fa9dc0caadddc", + "source": "col_16fd262a31187c551aa344aa", + "target": "col_ff170de8698fcb6322f09c03", + "type": "compute_dependency" + }, + { + "id": "e_77e5b66f499949715c3b7d6e", + "source": "col_e77b072519bb386d30d7cfaa", + "target": "expr_e20560a9236b4cfc39bae28a", + "type": "compute_dependency" + }, + { + "id": "e_7a9547dd3d785f7632e84562", + "source": "sql_9066d2306662ed7af1d2b961", + "target": "expr_488bc78483e98125d5304322", + "type": "contains" + }, + { + "id": "e_7b91a7768f4b366c6d49abae", + "source": "col_149da087f5665bf7e61e4c1c", + "target": "expr_3f37b8f2db622b1703c5c247", + "type": "compute_dependency" + }, + { + "id": "e_82508f99ad58af050c512684", + "source": "expr_3f37b8f2db622b1703c5c247", + "target": "expr_b939fff64b4305a6be94666d", + "type": "expr_operand" + }, + { + "id": "e_9ccd2c2cfd22b2e2d515ed63", + "source": "col_f69747bb922872b593a5d822", + "target": "expr_488bc78483e98125d5304322", + "type": "compute_dependency" + }, + { + "id": "e_c7abcddf920e6c37669a0ffc", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_f69747bb922872b593a5d822", + "type": "has_column" + }, + { + "id": "e_cb775580d8026326a4e3ff4e", + "source": "col_a23914f6b90d25b34b044275", + "target": "expr_488bc78483e98125d5304322", + "type": "compute_dependency" + }, + { + "id": "e_e22b74163c48a5a4c1d88fff", + "source": "sql_9066d2306662ed7af1d2b961", + "target": "tbl_fe3101537f9d6bca4e1bb3ac", + "type": "reads_from" + }, + { + "id": "e_e424776d679aad8217505fc1", + "source": "col_f69747bb922872b593a5d822", + "target": "expr_812be1f3f3d52bd14a76039a", + "type": "compute_dependency" + }, + { + "id": "e_e727e8c8a1f7a9f2174f7bcf", + "source": "tbl_fe3101537f9d6bca4e1bb3ac", + "target": "col_16fd262a31187c551aa344aa", + "type": "has_column" + }, + { + "id": "e_e8fba91214acf9611925c435", + "source": "col_149da087f5665bf7e61e4c1c", + "target": "expr_b939fff64b4305a6be94666d", + "type": "compute_dependency" + }, + { + "id": "e_e967ce9349688759acb6459e", + "source": "expr_488bc78483e98125d5304322", + "target": "col_8dccb2b9f470ee0ecba471d1", + "type": "produces" + }, + { + "id": "e_ed0541a41cb86f5c76b511ef", + "source": "expr_1ea1f4c4f29fc1dbdcb93e5d", + "target": "col_582a1f3a8283948df3934583", + "type": "produces" + }, + { + "id": "e_ef588380240d5490fd893cfc", + "source": "expr_b939fff64b4305a6be94666d", + "target": "col_85e98bfdc4fe7d6e5f3708eb", + "type": "produces" + }, + { + "id": "e_f6ee19c00fde0cb87b6940f2", + "source": "tbl_85ac6b96dd07668776b1d8d0", + "target": "col_582a1f3a8283948df3934583", + "type": "has_column" + }, + { + "id": "e_f704063542c940b5422b0f05", + "source": "expr_e20560a9236b4cfc39bae28a", + "target": "expr_1ea1f4c4f29fc1dbdcb93e5d", + "type": "expr_operand" + } + ], + "identity_rule_version": "id-v2", + "nodes": [ + { + "id": "col_06441fde90754757b5779ff8", + "name": "pay_amount", + "type": "column" + }, + { + "id": "col_149da087f5665bf7e61e4c1c", + "name": "refund_amount", + "type": "column" + }, + { + "id": "col_16fd262a31187c551aa344aa", + "name": "user_id", + "type": "column" + }, + { + "id": "col_582a1f3a8283948df3934583", + "name": "status", + "type": "column" + }, + { + "id": "col_5e0f941f7f808d1ca3365fd9", + "name": "weight", + "type": "column" + }, + { + "id": "col_85e98bfdc4fe7d6e5f3708eb", + "name": "net_amount", + "type": "column" + }, + { + "id": "col_8dccb2b9f470ee0ecba471d1", + "name": "total_score", + "type": "column" + }, + { + "id": "col_a23914f6b90d25b34b044275", + "name": "bonus_score", + "type": "column" + }, + { + "id": "col_e77b072519bb386d30d7cfaa", + "name": "active_days", + "type": "column" + }, + { + "id": "col_f69747bb922872b593a5d822", + "name": "base_score", + "type": "column" + }, + { + "id": "col_ff170de8698fcb6322f09c03", + "name": "user_id", + "type": "column" + }, + { + "id": "expr_1ea1f4c4f29fc1dbdcb93e5d", + "name": "CASE WHEN dwd_user_metric_di.active_days >= 7 THEN 'active' ELSE 'inactive' END", + "type": "transform" + }, + { + "id": "expr_3f37b8f2db622b1703c5c247", + "name": "COALESCE(dwd_user_metric_di.refund_amount, 0)", + "type": "transform" + }, + { + "id": "expr_488bc78483e98125d5304322", + "name": "(dwd_user_metric_di.base_score + dwd_user_metric_di.bonus_score) * dwd_user_metric_di.weight", + "type": "transform" + }, + { + "id": "expr_68cb1c8d5cca5b5bbaef2abf", + "name": "COALESCE(dwd_user_metric_di.pay_amount, 0)", + "type": "transform" + }, + { + "id": "expr_812be1f3f3d52bd14a76039a", + "name": "dwd_user_metric_di.base_score + dwd_user_metric_di.bonus_score", + "type": "transform" + }, + { + "id": "expr_b939fff64b4305a6be94666d", + "name": "COALESCE(dwd_user_metric_di.pay_amount, 0) + COALESCE(dwd_user_metric_di.refund_amount, 0)", + "type": "transform" + }, + { + "id": "expr_e20560a9236b4cfc39bae28a", + "name": "IF(dwd_user_metric_di.active_days >= 7, 'active')", + "type": "transform" + }, + { + "id": "sql_9066d2306662ed7af1d2b961", + "name": "inline_sql", + "type": "sql" + }, + { + "id": "tbl_85ac6b96dd07668776b1d8d0", + "name": "dws_user_value_di", + "type": "table" + }, + { + "id": "tbl_fe3101537f9d6bca4e1bb3ac", + "name": "dwd_user_metric_di", + "type": "table" + } + ], + "table_lineage": [ + [ + "dwd_user_metric_di", + "dws_user_value_di" + ] + ] +} diff --git a/tests/golden/snapshots/ds03_multi_stmt.json b/tests/golden/snapshots/ds03_multi_stmt.json new file mode 100644 index 0000000..7a1be1c --- /dev/null +++ b/tests/golden/snapshots/ds03_multi_stmt.json @@ -0,0 +1,269 @@ +{ + "counts": { + "by_edge_type": { + "compute_dependency": 6, + "has_column": 12, + "reads_from": 2, + "table_lineage": 2, + "writes_to": 2 + }, + "by_node_type": { + "column": 12, + "sql": 1, + "table": 4 + }, + "edges": 24, + "nodes": 17 + }, + "coverage": { + "failed": 0, + "ok": 1, + "partial": 0, + "reasons": [], + "total": 1 + }, + "edges": [ + { + "id": "e_07040b9bc5501c1122cad022", + "source": "tbl_3fda5a1108524666b6ab93d0", + "target": "col_1373c45efd85c44994f1ec12", + "type": "has_column" + }, + { + "id": "e_0aa3be6501e2b83fe060b785", + "source": "tbl_3fda5a1108524666b6ab93d0", + "target": "col_3159a26009fa74f9c6cdda08", + "type": "has_column" + }, + { + "id": "e_0feb09459c7889a7f75cd4ad", + "source": "col_73650d50bb36f9959d87cb18", + "target": "col_84e5a871f0f6d67f2bc54a93", + "type": "compute_dependency" + }, + { + "id": "e_19916d7d56df93468cb1c15e", + "source": "sql_7330cfdb3d1737123adb642c", + "target": "tbl_6bfb9d03c633526e6e31d9e0", + "type": "reads_from" + }, + { + "id": "e_2864983dfc7ed3dd5f42dddc", + "source": "sql_7330cfdb3d1737123adb642c", + "target": "tbl_3fda5a1108524666b6ab93d0", + "type": "reads_from" + }, + { + "id": "e_2b5ccc37a6187c62d73abf30", + "source": "tbl_6bfb9d03c633526e6e31d9e0", + "target": "col_73650d50bb36f9959d87cb18", + "type": "has_column" + }, + { + "id": "e_3259c5ee2e7026ad51db9fae", + "source": "tbl_03f6dcdd8473326bcb523136", + "target": "col_6a31d78e5b0105044b42157e", + "type": "has_column" + }, + { + "id": "e_3fa9d3c9a27e0633938c71e6", + "source": "tbl_6bfb9d03c633526e6e31d9e0", + "target": "col_32310ca3b5230d25d00d8e27", + "type": "has_column" + }, + { + "id": "e_65cf112f7a0aad7fb62004fc", + "source": "col_119f77efe99ee69f4f54dd6c", + "target": "col_10d357a73df3bde7c861c593", + "type": "compute_dependency" + }, + { + "id": "e_749acc92998b0e6595c16a99", + "source": "tbl_03f6dcdd8473326bcb523136", + "target": "col_5aedc0df57fcb8dd48d59daa", + "type": "has_column" + }, + { + "id": "e_80f7725dcda7a922cf03536f", + "source": "tbl_6bfb9d03c633526e6e31d9e0", + "target": "tbl_c6f3eeeef4d1609179d994f6", + "type": "table_lineage" + }, + { + "id": "e_8967928f00bf4a5353659088", + "source": "tbl_3fda5a1108524666b6ab93d0", + "target": "tbl_03f6dcdd8473326bcb523136", + "type": "table_lineage" + }, + { + "id": "e_98f58ee281d794307028ded3", + "source": "tbl_c6f3eeeef4d1609179d994f6", + "target": "col_67e78d3174a2792a8b388169", + "type": "has_column" + }, + { + "id": "e_acc8c390bf6e87d94227f8cc", + "source": "col_8eda04d287a8fd35932c92cb", + "target": "col_67e78d3174a2792a8b388169", + "type": "compute_dependency" + }, + { + "id": "e_bc964251edce65b022ecb0a9", + "source": "tbl_c6f3eeeef4d1609179d994f6", + "target": "col_84e5a871f0f6d67f2bc54a93", + "type": "has_column" + }, + { + "id": "e_c2a067f70aa41c1f959c55a2", + "source": "tbl_03f6dcdd8473326bcb523136", + "target": "col_10d357a73df3bde7c861c593", + "type": "has_column" + }, + { + "id": "e_cd7158e0418d49c1810f5538", + "source": "tbl_3fda5a1108524666b6ab93d0", + "target": "col_119f77efe99ee69f4f54dd6c", + "type": "has_column" + }, + { + "id": "e_e1ff37d346934758876b5a65", + "source": "tbl_c6f3eeeef4d1609179d994f6", + "target": "col_f8af616e99742b4e95771742", + "type": "has_column" + }, + { + "id": "e_e6a2d8c14eba99497952d234", + "source": "sql_7330cfdb3d1737123adb642c", + "target": "tbl_c6f3eeeef4d1609179d994f6", + "type": "writes_to" + }, + { + "id": "e_ee0d26f77ba269553bcd178b", + "source": "col_1373c45efd85c44994f1ec12", + "target": "col_5aedc0df57fcb8dd48d59daa", + "type": "compute_dependency" + }, + { + "id": "e_f2ec3a30c4fb01d5e40bae2c", + "source": "tbl_6bfb9d03c633526e6e31d9e0", + "target": "col_8eda04d287a8fd35932c92cb", + "type": "has_column" + }, + { + "id": "e_f786f00d366d7ab678bdb53a", + "source": "sql_7330cfdb3d1737123adb642c", + "target": "tbl_03f6dcdd8473326bcb523136", + "type": "writes_to" + }, + { + "id": "e_f8f3fff1154b68996c376a6a", + "source": "col_32310ca3b5230d25d00d8e27", + "target": "col_f8af616e99742b4e95771742", + "type": "compute_dependency" + }, + { + "id": "e_f99116d73c4797b35822a1e4", + "source": "col_3159a26009fa74f9c6cdda08", + "target": "col_6a31d78e5b0105044b42157e", + "type": "compute_dependency" + } + ], + "identity_rule_version": "id-v2", + "nodes": [ + { + "id": "col_10d357a73df3bde7c861c593", + "name": "log_id", + "type": "column" + }, + { + "id": "col_119f77efe99ee69f4f54dd6c", + "name": "log_id", + "type": "column" + }, + { + "id": "col_1373c45efd85c44994f1ec12", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_3159a26009fa74f9c6cdda08", + "name": "ts", + "type": "column" + }, + { + "id": "col_32310ca3b5230d25d00d8e27", + "name": "ts", + "type": "column" + }, + { + "id": "col_5aedc0df57fcb8dd48d59daa", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_67e78d3174a2792a8b388169", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_6a31d78e5b0105044b42157e", + "name": "ts", + "type": "column" + }, + { + "id": "col_73650d50bb36f9959d87cb18", + "name": "log_id", + "type": "column" + }, + { + "id": "col_84e5a871f0f6d67f2bc54a93", + "name": "log_id", + "type": "column" + }, + { + "id": "col_8eda04d287a8fd35932c92cb", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_f8af616e99742b4e95771742", + "name": "ts", + "type": "column" + }, + { + "id": "sql_7330cfdb3d1737123adb642c", + "name": "inline_sql", + "type": "sql" + }, + { + "id": "tbl_03f6dcdd8473326bcb523136", + "name": "dwd_click_di", + "type": "table" + }, + { + "id": "tbl_3fda5a1108524666b6ab93d0", + "name": "ods_click_log_di", + "type": "table" + }, + { + "id": "tbl_6bfb9d03c633526e6e31d9e0", + "name": "ods_impression_log_di", + "type": "table" + }, + { + "id": "tbl_c6f3eeeef4d1609179d994f6", + "name": "dwd_impression_di", + "type": "table" + } + ], + "table_lineage": [ + [ + "ods_click_log_di", + "dwd_click_di" + ], + [ + "ods_impression_log_di", + "dwd_impression_di" + ] + ] +} diff --git a/tests/golden/snapshots/ds04_cte_subquery.json b/tests/golden/snapshots/ds04_cte_subquery.json new file mode 100644 index 0000000..7d162d1 --- /dev/null +++ b/tests/golden/snapshots/ds04_cte_subquery.json @@ -0,0 +1,379 @@ +{ + "counts": { + "by_edge_type": { + "compute_dependency": 10, + "contains": 3, + "has_column": 15, + "produces": 3, + "reads_from": 2, + "table_lineage": 2, + "writes_to": 1 + }, + "by_node_type": { + "column": 15, + "sql": 1, + "table": 5, + "transform": 3 + }, + "edges": 36, + "nodes": 24 + }, + "coverage": { + "failed": 0, + "ok": 1, + "partial": 0, + "reasons": [], + "total": 1 + }, + "edges": [ + { + "id": "e_02f146bb8bf05247c24d9776", + "source": "sql_f8ef7924dcbf0860dd46248f", + "target": "tbl_c6f3eeeef4d1609179d994f6", + "type": "reads_from" + }, + { + "id": "e_0325db9dacb73139ab616f34", + "source": "tbl_ee5dca11b432ecb5755b7b30", + "target": "col_9d1fa65d7ea95c6da27b15b7", + "type": "has_column" + }, + { + "id": "e_03861bcd04772c1b8cfda8d6", + "source": "col_5aedc0df57fcb8dd48d59daa", + "target": "col_83e1cbf52417906222a1852d", + "type": "compute_dependency" + }, + { + "id": "e_1129dcc2d48f4c82f1ca02ea", + "source": "tbl_d35b8b77323a4ef05d4c2039", + "target": "col_5513d11f6b126834ccabaf65", + "type": "has_column" + }, + { + "id": "e_201dec6f0ae9b0a75cae8ca5", + "source": "sql_f8ef7924dcbf0860dd46248f", + "target": "expr_b823cf2c863ecfc59bac7d00", + "type": "contains" + }, + { + "id": "e_2154aa34e0ea22085c7bbebc", + "source": "tbl_d35b8b77323a4ef05d4c2039", + "target": "col_41b786e9c7de4f956d281f36", + "type": "has_column" + }, + { + "id": "e_2234fcf53f946cca2a91e80f", + "source": "tbl_c6f3eeeef4d1609179d994f6", + "target": "tbl_ee5dca11b432ecb5755b7b30", + "type": "table_lineage" + }, + { + "id": "e_3862e5a306f94b539792cf79", + "source": "sql_f8ef7924dcbf0860dd46248f", + "target": "tbl_ee5dca11b432ecb5755b7b30", + "type": "writes_to" + }, + { + "id": "e_58b379a916cd43ece979be85", + "source": "col_4a74a22e5f6794d3dd6bb1e9", + "target": "col_41b786e9c7de4f956d281f36", + "type": "compute_dependency" + }, + { + "id": "e_5b04c0c2f0c89f970f8a622b", + "source": "expr_b823cf2c863ecfc59bac7d00", + "target": "col_5513d11f6b126834ccabaf65", + "type": "produces" + }, + { + "id": "e_63555a154789540a30227431", + "source": "tbl_ee5dca11b432ecb5755b7b30", + "target": "col_6c140965a8059c8112111ec1", + "type": "has_column" + }, + { + "id": "e_6d0a529a440ea6bd4c2cce1c", + "source": "col_79ba38503a37a242cbd9a798", + "target": "col_479a5500c13505c2815e20c4", + "type": "compute_dependency" + }, + { + "id": "e_6eb35d6e257399f66df3974a", + "source": "tbl_52f4b46f12b4cbf97311cefb", + "target": "col_1d1bf3eb404f357006e73fd9", + "type": "has_column" + }, + { + "id": "e_72ff3b81d19d2d1ce4ee49bf", + "source": "col_37a9a80c164e141259ab9316", + "target": "expr_cd18956f701a0610e32d3c4d", + "type": "compute_dependency" + }, + { + "id": "e_749acc92998b0e6595c16a99", + "source": "tbl_03f6dcdd8473326bcb523136", + "target": "col_5aedc0df57fcb8dd48d59daa", + "type": "has_column" + }, + { + "id": "e_76f43420e18682440498b971", + "source": "col_41b786e9c7de4f956d281f36", + "target": "col_bb0d4a36c8f13e92abf6aded", + "type": "compute_dependency" + }, + { + "id": "e_77252bde31c862e44ec0b665", + "source": "expr_0f3d67b6b2c998b98340bae5", + "target": "col_37a9a80c164e141259ab9316", + "type": "produces" + }, + { + "id": "e_7ea3aaa2e5576756ebc0fbc5", + "source": "col_37a9a80c164e141259ab9316", + "target": "col_9d1fa65d7ea95c6da27b15b7", + "type": "compute_dependency" + }, + { + "id": "e_86297a4f9b35338aab0876b8", + "source": "tbl_ee5dca11b432ecb5755b7b30", + "target": "col_69e11c87cd5c40ac08576430", + "type": "has_column" + }, + { + "id": "e_96ca1ccc3d707a415e070f28", + "source": "tbl_52f4b46f12b4cbf97311cefb", + "target": "col_479a5500c13505c2815e20c4", + "type": "has_column" + }, + { + "id": "e_98f58ee281d794307028ded3", + "source": "tbl_c6f3eeeef4d1609179d994f6", + "target": "col_67e78d3174a2792a8b388169", + "type": "has_column" + }, + { + "id": "e_9caa2c53a05a9e41f7df82df", + "source": "tbl_03f6dcdd8473326bcb523136", + "target": "col_4a74a22e5f6794d3dd6bb1e9", + "type": "has_column" + }, + { + "id": "e_9fc08ec8c2673cc2890275c6", + "source": "col_5513d11f6b126834ccabaf65", + "target": "expr_cd18956f701a0610e32d3c4d", + "type": "compute_dependency" + }, + { + "id": "e_a116cbd22fe9d23908ab96e4", + "source": "sql_f8ef7924dcbf0860dd46248f", + "target": "expr_0f3d67b6b2c998b98340bae5", + "type": "contains" + }, + { + "id": "e_ac3cdad46b1a496d891926bb", + "source": "sql_f8ef7924dcbf0860dd46248f", + "target": "tbl_03f6dcdd8473326bcb523136", + "type": "reads_from" + }, + { + "id": "e_bb61a771a2b3f0b69f60e1c0", + "source": "expr_cd18956f701a0610e32d3c4d", + "target": "col_6c140965a8059c8112111ec1", + "type": "produces" + }, + { + "id": "e_bba4c3922268a866c47f740f", + "source": "col_83e1cbf52417906222a1852d", + "target": "col_69e11c87cd5c40ac08576430", + "type": "compute_dependency" + }, + { + "id": "e_c853002e96483077c0a662cd", + "source": "tbl_03f6dcdd8473326bcb523136", + "target": "tbl_ee5dca11b432ecb5755b7b30", + "type": "table_lineage" + }, + { + "id": "e_ca1cdf78423067f4ca7f168b", + "source": "tbl_c6f3eeeef4d1609179d994f6", + "target": "col_79ba38503a37a242cbd9a798", + "type": "has_column" + }, + { + "id": "e_cd654582d274453063b506e4", + "source": "tbl_d35b8b77323a4ef05d4c2039", + "target": "col_83e1cbf52417906222a1852d", + "type": "has_column" + }, + { + "id": "e_d75ba74741ede0b49335aae9", + "source": "tbl_ee5dca11b432ecb5755b7b30", + "target": "col_26884c3d3a6e38a2c8ca1f29", + "type": "has_column" + }, + { + "id": "e_d8c9f88638d2f98925173c2c", + "source": "tbl_ee5dca11b432ecb5755b7b30", + "target": "col_bb0d4a36c8f13e92abf6aded", + "type": "has_column" + }, + { + "id": "e_dc9c4f74aa335643001fa3c4", + "source": "sql_f8ef7924dcbf0860dd46248f", + "target": "expr_cd18956f701a0610e32d3c4d", + "type": "contains" + }, + { + "id": "e_e9163c8b2454cb5ea73ca4cd", + "source": "tbl_52f4b46f12b4cbf97311cefb", + "target": "col_37a9a80c164e141259ab9316", + "type": "has_column" + }, + { + "id": "e_ef00857d6f1eefeb867c97e1", + "source": "col_5513d11f6b126834ccabaf65", + "target": "col_26884c3d3a6e38a2c8ca1f29", + "type": "compute_dependency" + }, + { + "id": "e_fcf0b58aab0de7915712b8ef", + "source": "col_67e78d3174a2792a8b388169", + "target": "col_1d1bf3eb404f357006e73fd9", + "type": "compute_dependency" + } + ], + "identity_rule_version": "id-v2", + "nodes": [ + { + "id": "col_1d1bf3eb404f357006e73fd9", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_26884c3d3a6e38a2c8ca1f29", + "name": "click_cnt", + "type": "column" + }, + { + "id": "col_37a9a80c164e141259ab9316", + "name": "imp_cnt", + "type": "column" + }, + { + "id": "col_41b786e9c7de4f956d281f36", + "name": "dt", + "type": "column" + }, + { + "id": "col_479a5500c13505c2815e20c4", + "name": "dt", + "type": "column" + }, + { + "id": "col_4a74a22e5f6794d3dd6bb1e9", + "name": "dt", + "type": "column" + }, + { + "id": "col_5513d11f6b126834ccabaf65", + "name": "click_cnt", + "type": "column" + }, + { + "id": "col_5aedc0df57fcb8dd48d59daa", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_67e78d3174a2792a8b388169", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_69e11c87cd5c40ac08576430", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_6c140965a8059c8112111ec1", + "name": "ctr", + "type": "column" + }, + { + "id": "col_79ba38503a37a242cbd9a798", + "name": "dt", + "type": "column" + }, + { + "id": "col_83e1cbf52417906222a1852d", + "name": "ad_id", + "type": "column" + }, + { + "id": "col_9d1fa65d7ea95c6da27b15b7", + "name": "imp_cnt", + "type": "column" + }, + { + "id": "col_bb0d4a36c8f13e92abf6aded", + "name": "dt", + "type": "column" + }, + { + "id": "expr_0f3d67b6b2c998b98340bae5", + "name": "COUNT(*)", + "type": "transform" + }, + { + "id": "expr_b823cf2c863ecfc59bac7d00", + "name": "COUNT(*)", + "type": "transform" + }, + { + "id": "expr_cd18956f701a0610e32d3c4d", + "name": "subq_82b712a5b53d9dec.click_cnt / subq_9b67a3031aaea184.imp_cnt", + "type": "transform" + }, + { + "id": "sql_f8ef7924dcbf0860dd46248f", + "name": "inline_sql", + "type": "sql" + }, + { + "id": "tbl_03f6dcdd8473326bcb523136", + "name": "dwd_click_di", + "type": "table" + }, + { + "id": "tbl_52f4b46f12b4cbf97311cefb", + "name": "subq_9b67a3031aaea184", + "type": "table" + }, + { + "id": "tbl_c6f3eeeef4d1609179d994f6", + "name": "dwd_impression_di", + "type": "table" + }, + { + "id": "tbl_d35b8b77323a4ef05d4c2039", + "name": "subq_82b712a5b53d9dec", + "type": "table" + }, + { + "id": "tbl_ee5dca11b432ecb5755b7b30", + "name": "dws_ad_daily_di", + "type": "table" + } + ], + "table_lineage": [ + [ + "dwd_click_di", + "dws_ad_daily_di" + ], + [ + "dwd_impression_di", + "dws_ad_daily_di" + ] + ] +} diff --git a/tests/golden/snapshots/ds06_whitespace_variant.json b/tests/golden/snapshots/ds06_whitespace_variant.json new file mode 100644 index 0000000..1023f89 --- /dev/null +++ b/tests/golden/snapshots/ds06_whitespace_variant.json @@ -0,0 +1,202 @@ +{ + "counts": { + "by_edge_type": { + "compute_dependency": 4, + "has_column": 8, + "reads_from": 2, + "table_lineage": 2, + "writes_to": 1 + }, + "by_node_type": { + "column": 8, + "sql": 1, + "table": 3 + }, + "edges": 17, + "nodes": 12 + }, + "coverage": { + "failed": 0, + "ok": 1, + "partial": 0, + "reasons": [], + "total": 1 + }, + "edges": [ + { + "id": "e_1d63b8dd4a250242f079aaae", + "source": "col_cd3c51d4f0369696ed1edcd1", + "target": "col_78eaf5857233bb51de95a97b", + "type": "compute_dependency" + }, + { + "id": "e_2cd22f1d9f3684e0292e1f06", + "source": "col_9a4a7bfc1637f9344131553c", + "target": "col_83efbc91d74d432a112f5f02", + "type": "compute_dependency" + }, + { + "id": "e_3123b8adbe6bddfb498d740e", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "col_cd3c51d4f0369696ed1edcd1", + "type": "has_column" + }, + { + "id": "e_39c08f9f482b58b60841a3b9", + "source": "tbl_1b6a4c53bdb6aa04297ca4c8", + "target": "tbl_26f98d454abeb56dd8a6f492", + "type": "table_lineage" + }, + { + "id": "e_41f952bb940e58d30fd3304c", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "col_0f782de896cd1690b6f6929f", + "type": "has_column" + }, + { + "id": "e_552cf08442978e1eace263a9", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_78eaf5857233bb51de95a97b", + "type": "has_column" + }, + { + "id": "e_5c63ed749bfc8f74d9c2aa5e", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_0d14b47f2feed2e7c1c488da", + "type": "has_column" + }, + { + "id": "e_9db622eb7f6a3f9fc8aa9854", + "source": "col_7d27d8740225ab319f84a43c", + "target": "col_fd1a1a150af28166e93a9fb7", + "type": "compute_dependency" + }, + { + "id": "e_b22b05c4e88a23514a360267", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_83efbc91d74d432a112f5f02", + "type": "has_column" + }, + { + "id": "e_b38f99eeb3620829e012b2fe", + "source": "sql_5079a3b1e8064933b111bfdc", + "target": "tbl_c4d0cac04b1ef5b337d8f7b8", + "type": "reads_from" + }, + { + "id": "e_bcd6ddbd0d8ff09e11465b70", + "source": "col_0f782de896cd1690b6f6929f", + "target": "col_0d14b47f2feed2e7c1c488da", + "type": "compute_dependency" + }, + { + "id": "e_bf3a8067565322a3875e6012", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "col_7d27d8740225ab319f84a43c", + "type": "has_column" + }, + { + "id": "e_c12fbe9067e0f91af05e74df", + "source": "tbl_c4d0cac04b1ef5b337d8f7b8", + "target": "tbl_26f98d454abeb56dd8a6f492", + "type": "table_lineage" + }, + { + "id": "e_c4e30c7ed043287612298a3b", + "source": "sql_5079a3b1e8064933b111bfdc", + "target": "tbl_1b6a4c53bdb6aa04297ca4c8", + "type": "reads_from" + }, + { + "id": "e_c6737114061e0da0e85697dd", + "source": "tbl_1b6a4c53bdb6aa04297ca4c8", + "target": "col_9a4a7bfc1637f9344131553c", + "type": "has_column" + }, + { + "id": "e_ef1a03d8b204f5514d9ff131", + "source": "tbl_26f98d454abeb56dd8a6f492", + "target": "col_fd1a1a150af28166e93a9fb7", + "type": "has_column" + }, + { + "id": "e_f265bf2101ec3c77dad8d879", + "source": "sql_5079a3b1e8064933b111bfdc", + "target": "tbl_26f98d454abeb56dd8a6f492", + "type": "writes_to" + } + ], + "identity_rule_version": "id-v2", + "nodes": [ + { + "id": "col_0d14b47f2feed2e7c1c488da", + "name": "order_id", + "type": "column" + }, + { + "id": "col_0f782de896cd1690b6f6929f", + "name": "order_id", + "type": "column" + }, + { + "id": "col_78eaf5857233bb51de95a97b", + "name": "amount", + "type": "column" + }, + { + "id": "col_7d27d8740225ab319f84a43c", + "name": "user_id", + "type": "column" + }, + { + "id": "col_83efbc91d74d432a112f5f02", + "name": "user_name", + "type": "column" + }, + { + "id": "col_9a4a7bfc1637f9344131553c", + "name": "user_name", + "type": "column" + }, + { + "id": "col_cd3c51d4f0369696ed1edcd1", + "name": "amount", + "type": "column" + }, + { + "id": "col_fd1a1a150af28166e93a9fb7", + "name": "user_id", + "type": "column" + }, + { + "id": "sql_5079a3b1e8064933b111bfdc", + "name": "inline_sql", + "type": "sql" + }, + { + "id": "tbl_1b6a4c53bdb6aa04297ca4c8", + "name": "dim_user", + "type": "table" + }, + { + "id": "tbl_26f98d454abeb56dd8a6f492", + "name": "dws_order_enriched_di", + "type": "table" + }, + { + "id": "tbl_c4d0cac04b1ef5b337d8f7b8", + "name": "dwd_order_di", + "type": "table" + } + ], + "table_lineage": [ + [ + "dim_user", + "dws_order_enriched_di" + ], + [ + "dwd_order_di", + "dws_order_enriched_di" + ] + ] +} diff --git a/tests/golden/test_corpus.py b/tests/golden/test_corpus.py new file mode 100644 index 0000000..c6fda6c --- /dev/null +++ b/tests/golden/test_corpus.py @@ -0,0 +1,49 @@ +"""Additional deterministic golden cases covering syntax and dialect boundaries.""" + +from __future__ import annotations + +import json + +import pytest + +from sqlgraph.api import build_graph + + +CASES = ( + ("join_alias", "spark", "INSERT INTO d SELECT a.id FROM a JOIN b ON a.id=b.id"), + ("window", "spark", "INSERT INTO d SELECT id, ROW_NUMBER() OVER (PARTITION BY id ORDER BY ts) AS rn FROM s"), + ("case", "hive", "INSERT INTO d SELECT CASE WHEN x>0 THEN 1 ELSE 0 END AS flag FROM s"), + ("cast", "presto", "INSERT INTO d SELECT CAST(amount AS DOUBLE) AS amount FROM s"), + ("aggregate", "spark", "INSERT INTO d SELECT k, SUM(v) AS total FROM s GROUP BY k"), + ("union", "spark", "INSERT INTO d SELECT id FROM a UNION ALL SELECT id FROM b"), + ("nested", "spark", "INSERT INTO d SELECT id FROM (SELECT id FROM s) q"), + ("udf", "spark", "INSERT INTO d SELECT custom_udf(value) AS value FROM s"), + ("coalesce", "hive", "INSERT INTO d SELECT COALESCE(a, b, 0) AS value FROM s"), + ("bigquery", "bigquery", "SELECT id, SAFE_CAST(value AS INT64) AS value FROM src"), + ("mysql", "mysql", "SELECT id, IFNULL(value, 0) AS value FROM src"), + ("postgres", "postgres", "SELECT id, value::DOUBLE PRECISION AS value FROM src"), + ("duckdb", "duckdb", "CREATE TABLE d AS SELECT id, value / 2 AS value FROM s"), + ("qualified", "spark", "INSERT INTO db.d SELECT db.s.id FROM db.s"), + ("multi_join", "spark", "INSERT INTO d SELECT a.id FROM a JOIN b ON a.id=b.id JOIN c ON b.id=c.id"), +) + + +def _signature(graph) -> str: + payload = graph.to_dict() + payload["nodes"] = sorted(payload["nodes"], key=lambda item: item["id"]) + payload["edges"] = sorted(payload["edges"], key=lambda item: item["id"]) + return json.dumps(payload, sort_keys=True, ensure_ascii=True) + + +def test_golden_corpus_reaches_twenty_cases(): + # Five snapshot fixtures plus these fifteen cases. + assert len(CASES) + 5 >= 20 + + +@pytest.mark.parametrize(("name", "dialect", "sql"), CASES, ids=[case[0] for case in CASES]) +def test_additional_golden_case_is_reproducible(name, dialect, sql): + first = build_graph(sql, dialect=dialect) + second = build_graph(sql, dialect=dialect) + + assert _signature(first) == _signature(second), name + assert first.metadata["coverage"]["failed"] == 0 diff --git a/tests/golden/test_golden.py b/tests/golden/test_golden.py new file mode 100644 index 0000000..c3c9ac4 --- /dev/null +++ b/tests/golden/test_golden.py @@ -0,0 +1,74 @@ +# tests/golden/test_golden.py +"""Golden 回归测试(REQ-ARCH-02 / REQ-CLI-02 TC2)。 + +对 §14.2 的 golden fixture 集: +- 断言每个 fixture 的规范化结构签名与基线快照逐字节一致;漂移即失败并输出 + 可读 diff(REQ-CLI-02 TC2:golden 图漂移时回归测试失败并输出 diff)。 +- 断言同一输入连续构建两次签名相等(REQ-ARCH-02 AC1:确定性可复算)。 +- 断言 8 类边在 golden 图全集上均有非零实例(呼应 REQ-GRAPH-02 AC2)。 +""" +from __future__ import annotations + +import difflib + +import pytest + +from tests.golden import _harness as h + +_FIXTURES = h.list_fixtures() +_FIXTURE_IDS = [f.stem for f in _FIXTURES] + +# 8 类边的权威集合(REQ-GRAPH-02)。 +_ALL_EDGE_TYPES = { + "reads_from", "writes_to", "has_column", "contains", + "compute_dependency", "produces", "expr_operand", "table_lineage", +} + + +def test_golden_fixtures_present(): + """至少覆盖 DS-01~DS-04、DS-06 五个数据集。""" + assert len(_FIXTURES) >= 5, "golden fixture 数量不足,检查 tests/golden/fixtures/" + + +@pytest.mark.parametrize("fixture", _FIXTURES, ids=_FIXTURE_IDS) +def test_golden_signature_matches_baseline(fixture): + """规范化结构签名与基线快照逐字节一致;漂移输出 diff。""" + snap = h.snapshot_path(fixture) + assert snap.exists(), ( + f"缺少基线快照 {snap.name};如为新增 fixture,请运行 " + f"`python -m tests.golden.regen` 固化基线。" + ) + actual = h.dumps(h.canonical_signature(h.build_fixture_graph(fixture))) + expected = snap.read_text(encoding="utf-8") + if actual != expected: + diff = "".join( + difflib.unified_diff( + expected.splitlines(keepends=True), + actual.splitlines(keepends=True), + fromfile=f"baseline/{snap.name}", + tofile=f"current/{fixture.stem}", + ) + ) + pytest.fail( + f"golden 图漂移:{fixture.name} 的结构签名与基线不一致。\n" + f"若为预期变更,请运行 `python -m tests.golden.regen` 并在 PR 标注 REQ 编号。\n" + f"--- diff ---\n{diff}" + ) + + +@pytest.mark.parametrize("fixture", _FIXTURES, ids=_FIXTURE_IDS) +def test_golden_build_is_reproducible(fixture): + """REQ-ARCH-02 AC1:同一输入连续构建两次,规范化签名相等。""" + s1 = h.dumps(h.canonical_signature(h.build_fixture_graph(fixture))) + s2 = h.dumps(h.canonical_signature(h.build_fixture_graph(fixture))) + assert s1 == s2 + + +def test_all_eight_edge_types_present_across_golden(): + """REQ-GRAPH-02 AC2:8 类边在 golden 图全集上均有非零实例。""" + seen: set[str] = set() + for fixture in _FIXTURES: + sig = h.canonical_signature(h.build_fixture_graph(fixture)) + seen |= set(sig["counts"]["by_edge_type"].keys()) + missing = _ALL_EDGE_TYPES - seen + assert not missing, f"以下边类型在 golden 集上没有实例,覆盖不足: {sorted(missing)}" diff --git a/tests/safety/__init__.py b/tests/safety/__init__.py new file mode 100644 index 0000000..1b6c7fe --- /dev/null +++ b/tests/safety/__init__.py @@ -0,0 +1 @@ +"""Governance safety tests.""" diff --git a/tests/safety/test_audit_integrity.py b/tests/safety/test_audit_integrity.py new file mode 100644 index 0000000..32302b3 --- /dev/null +++ b/tests/safety/test_audit_integrity.py @@ -0,0 +1,17 @@ +from __future__ import annotations + +from sqlgraph.audit import AuditEvent, AuditLog + + +def test_reordered_events_break_hash_chain(tmp_path): + path = tmp_path / "audit.jsonl" + log = AuditLog(path) + log.append(AuditEvent("task-1", "base-1", "observe", "observe")) + log.append(AuditEvent("task-1", "base-1", "verify", "verify")) + lines = path.read_text(encoding="utf-8").splitlines() + path.write_text("\n".join(reversed(lines)) + "\n", encoding="utf-8") + + report = log.verify_integrity() + + assert not report.valid + assert report.errors diff --git a/tests/safety/test_autonomy_gates.py b/tests/safety/test_autonomy_gates.py new file mode 100644 index 0000000..fc470d7 --- /dev/null +++ b/tests/safety/test_autonomy_gates.py @@ -0,0 +1,47 @@ +from __future__ import annotations + +from sqlgraph.autonomy import ( + AuthorizationScope, + AutonomyLevel, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) + + +def test_high_roi_cannot_override_unverified_rollback(): + action = GovernanceAction( + action_type="drop_table", + evidence_version="task-1@v1#abc", + evidence_grounded=True, + reversibility=ReversibilityEvidence( + state_restorable=True, + external_effects_controlled=True, + rollback_verified=False, + ), + authorization_scope=AuthorizationScope.CONTINUOUS_L5, + historical_reliability=1.0, + roi=1_000_000, + ) + + decision = decide_autonomy(action) + + assert decision.level == AutonomyLevel.L4_APPROVED + assert decision.requires_human_review + assert decision.reversibility_veto + assert not decision.scoring_performed + + +def test_missing_evidence_version_rejects_explanation(): + action = GovernanceAction( + action_type="add_tag", + evidence_version="", + evidence_grounded=True, + reversibility=ReversibilityEvidence(True, True, True), + authorization_scope=AuthorizationScope.SINGLE_L3, + ) + + decision = decide_autonomy(action) + + assert decision.level == AutonomyLevel.L0_OBSERVE + assert decision.evidence_veto diff --git a/tests/safety/test_evidence_gates.py b/tests/safety/test_evidence_gates.py new file mode 100644 index 0000000..0befe38 --- /dev/null +++ b/tests/safety/test_evidence_gates.py @@ -0,0 +1,34 @@ +from __future__ import annotations + +from sqlgraph.api import build_graph +from sqlgraph.evidence import EvidenceEngine, EvidenceRequest + + +def test_unresolved_column_forces_escalation(): + graph = build_graph( + "INSERT INTO dst SELECT id FROM left_table l " + "JOIN right_table r ON l.key = r.key", + dialect="spark", + ) + request = EvidenceRequest( + task_id="ambiguous", + baseline_id="base_ambiguous", + intent="caliber_repair", + anchors=("left_table", "dst"), + direction="both", + max_depth=2, + coverage_obligations=( + "anchors_resolved", + "lineage_path", + "counterevidence_checked", + "no_unresolved", + ), + ) + + decision = EvidenceEngine(graph).assess( + EvidenceEngine(graph).collect(request) + ) + + assert not decision.sufficient + assert decision.action == "escalate" + assert "no_unresolved" in decision.missing_obligations diff --git a/tests/safety/test_execution_recovery.py b/tests/safety/test_execution_recovery.py new file mode 100644 index 0000000..e50d70a --- /dev/null +++ b/tests/safety/test_execution_recovery.py @@ -0,0 +1,47 @@ +from __future__ import annotations + +import hashlib + +from sqlgraph.actions import ActionEngine, ActionRequest, SqlFilePatchAdapter +from sqlgraph.autonomy import ( + AuthorizationScope, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) + + +def test_injected_failure_opens_circuit_and_restores_snapshot(tmp_path): + path = tmp_path / "query.sql" + before = "SELECT value * 100 FROM source" + path.write_text(before, encoding="utf-8") + decision = decide_autonomy(GovernanceAction( + action_type="sql_patch", + evidence_version="task-1@v1#abc", + evidence_grounded=True, + reversibility=ReversibilityEvidence(True, True, True), + authorization_scope=AuthorizationScope.SINGLE_L3, + )) + request = ActionRequest( + task_id="task-1", + baseline_id="base-1", + evidence_version="task-1@v1#abc", + decision=decision, + adapter="sql_file_patch", + operations=({ + "path": str(path), + "before": before, + "after": "SELECT value FROM source", + "expected_sha256": hashlib.sha256(before.encode()).hexdigest(), + "inject_failure_after_apply": True, + },), + ) + engine = ActionEngine([SqlFilePatchAdapter()]) + + result = engine.execute(engine.plan(request)) + + assert result.status == "rolled_back" + assert result.circuit_open + assert result.rollback is not None + assert result.rollback.verified + assert path.read_text(encoding="utf-8") == before diff --git a/tests/safety/test_verification_failure.py b/tests/safety/test_verification_failure.py new file mode 100644 index 0000000..beaadc3 --- /dev/null +++ b/tests/safety/test_verification_failure.py @@ -0,0 +1,27 @@ +from __future__ import annotations + +from sqlgraph.verification import LayerResult, VerificationReport + + +def test_runtime_not_run_is_incomplete(): + report = VerificationReport( + code=LayerResult("code", "pass"), + structure=LayerResult("structure", "pass"), + runtime=LayerResult("runtime", "not_run"), + ) + + assert report.outcome == "incomplete" + + +def test_runtime_failure_blocks_success_even_if_code_and_structure_pass(): + report = VerificationReport( + code=LayerResult("code", "pass"), + structure=LayerResult("structure", "pass"), + runtime=LayerResult( + "runtime", + "fail", + evidence={"new_alerts": 1}, + ), + ) + + assert report.outcome == "failed" diff --git a/tests/scale/__init__.py b/tests/scale/__init__.py new file mode 100644 index 0000000..cc8798e --- /dev/null +++ b/tests/scale/__init__.py @@ -0,0 +1 @@ +"""Scale benchmark tests.""" diff --git a/tests/scale/test_generated_warehouse.py b/tests/scale/test_generated_warehouse.py new file mode 100644 index 0000000..1943986 --- /dev/null +++ b/tests/scale/test_generated_warehouse.py @@ -0,0 +1,13 @@ +from __future__ import annotations + +from tools.benchmark_scale import run_scale_benchmark + + +def test_generated_sql_scale_smoke(): + result = run_scale_benchmark(100) + + assert result.parsed == 100 + assert result.deterministic + assert result.peak_memory_mb > 0 + assert result.node_count > 100 + assert result.edge_count > 100 diff --git a/tests/test_actions/__init__.py b/tests/test_actions/__init__.py new file mode 100644 index 0000000..eda2730 --- /dev/null +++ b/tests/test_actions/__init__.py @@ -0,0 +1 @@ +"""Action engine tests.""" diff --git a/tests/test_actions/test_engine.py b/tests/test_actions/test_engine.py new file mode 100644 index 0000000..d1177d5 --- /dev/null +++ b/tests/test_actions/test_engine.py @@ -0,0 +1,70 @@ +from __future__ import annotations + +import hashlib + +from sqlgraph.actions import ActionEngine, ActionRequest, SqlFilePatchAdapter +from sqlgraph.autonomy import ( + AuthorizationScope, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) + + +def _decision(): + return decide_autonomy(GovernanceAction( + action_type="sql_patch", + evidence_version="task-1@v1#abc", + evidence_grounded=True, + reversibility=ReversibilityEvidence(True, True, True), + authorization_scope=AuthorizationScope.SINGLE_L3, + )) + + +def _request(path, before, after, *, inject_failure=False): + return ActionRequest( + task_id="task-1", + baseline_id="base-1", + evidence_version="task-1@v1#abc", + decision=_decision(), + adapter="sql_file_patch", + operations=({ + "path": str(path), + "before": before, + "after": after, + "expected_sha256": hashlib.sha256(before.encode()).hexdigest(), + "inject_failure_after_apply": inject_failure, + },), + ) + + +def test_repeat_execution_is_noop(tmp_path): + path = tmp_path / "query.sql" + path.write_text("SELECT value * 100 FROM source", encoding="utf-8") + engine = ActionEngine([SqlFilePatchAdapter()]) + plan = engine.plan(_request( + path, + "SELECT value * 100 FROM source", + "SELECT value FROM source", + )) + + first = engine.execute(plan) + second = engine.execute(plan) + + assert first.status == "success" + assert second.status == "noop" + assert second.execution_id == first.execution_id + assert path.read_text(encoding="utf-8") == "SELECT value FROM source" + + +def test_dry_run_does_not_modify_target(tmp_path): + path = tmp_path / "query.sql" + before = "SELECT value * 100 FROM source" + path.write_text(before, encoding="utf-8") + engine = ActionEngine([SqlFilePatchAdapter()]) + plan = engine.plan(_request(path, before, "SELECT value FROM source")) + + result = engine.dry_run(plan) + + assert result.status == "ready" + assert path.read_text(encoding="utf-8") == before diff --git a/tests/test_audit/__init__.py b/tests/test_audit/__init__.py new file mode 100644 index 0000000..51c4485 --- /dev/null +++ b/tests/test_audit/__init__.py @@ -0,0 +1 @@ +"""Audit log tests.""" diff --git a/tests/test_audit/test_log.py b/tests/test_audit/test_log.py new file mode 100644 index 0000000..2e1ccee --- /dev/null +++ b/tests/test_audit/test_log.py @@ -0,0 +1,42 @@ +from __future__ import annotations + +import json + +from sqlgraph.audit import AuditEvent, AuditLog + + +def _event(step: str) -> AuditEvent: + return AuditEvent( + task_id="task-1", + baseline_id="base-1", + event_type=step, + step=step, + payload={"status": "ok"}, + ) + + +def test_append_only_log_verifies_and_replays(tmp_path): + log = AuditLog(tmp_path / "audit.jsonl") + for step in ("observe", "authorize", "verify"): + log.append(_event(step)) + + integrity = log.verify_integrity() + replay = log.replay("task-1") + + assert integrity.valid + assert [event.sequence for event in replay.events] == [1, 2, 3] + assert replay.integrity.valid + + +def test_tampered_event_breaks_integrity(tmp_path): + path = tmp_path / "audit.jsonl" + log = AuditLog(path) + log.append(_event("observe")) + log.append(_event("verify")) + lines = path.read_text(encoding="utf-8").splitlines() + payload = json.loads(lines[0]) + payload["payload"]["status"] = "tampered" + lines[0] = json.dumps(payload, sort_keys=True) + path.write_text("\n".join(lines) + "\n", encoding="utf-8") + + assert not log.verify_integrity().valid diff --git a/tests/test_autonomy/__init__.py b/tests/test_autonomy/__init__.py new file mode 100644 index 0000000..a29dab9 --- /dev/null +++ b/tests/test_autonomy/__init__.py @@ -0,0 +1 @@ +"""Autonomy policy tests.""" diff --git a/tests/test_autonomy/test_decision.py b/tests/test_autonomy/test_decision.py new file mode 100644 index 0000000..8a3a195 --- /dev/null +++ b/tests/test_autonomy/test_decision.py @@ -0,0 +1,67 @@ +from __future__ import annotations + +import pytest + +from sqlgraph.autonomy import ( + AuthorizationScope, + AutonomyLevel, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) + + +def _action( + *, + reversible: bool = True, + grounded: bool = True, + scope: AuthorizationScope = AuthorizationScope.SINGLE_L3, + roi: float = 1.0, +) -> GovernanceAction: + return GovernanceAction( + action_type="sql_patch", + evidence_version="task-1@v1#abc", + evidence_grounded=grounded, + reversibility=ReversibilityEvidence( + state_restorable=reversible, + external_effects_controlled=reversible, + rollback_verified=reversible, + ), + authorization_scope=scope, + roi=roi, + ) + + +@pytest.mark.parametrize( + ("reversible", "grounded", "scope", "expected"), + [ + (False, True, AuthorizationScope.CONTINUOUS_L5, AutonomyLevel.L4_APPROVED), + (True, False, AuthorizationScope.SINGLE_L3, AutonomyLevel.L0_OBSERVE), + (True, True, AuthorizationScope.NONE, AutonomyLevel.L2_PROPOSE), + (True, True, AuthorizationScope.SINGLE_L3, AutonomyLevel.L3_BOUNDED), + ], +) +def test_hard_gates_precede_scoring(reversible, grounded, scope, expected): + decision = decide_autonomy( + _action( + reversible=reversible, + grounded=grounded, + scope=scope, + roi=999, + ) + ) + + assert decision.level == expected + if not reversible or not grounded: + assert not decision.scoring_performed + + +def test_continuous_scope_is_separate_from_action_depth(): + decision = decide_autonomy( + _action(scope=AuthorizationScope.CONTINUOUS_L5) + ) + + assert decision.level == AutonomyLevel.L3_BOUNDED + assert decision.within_l5_scope + assert decision.policy_version + assert decision.evidence_version == "task-1@v1#abc" diff --git a/tests/test_builder/test_builder.py b/tests/test_builder/test_builder.py index 8cf986c..96db9af 100644 --- a/tests/test_builder/test_builder.py +++ b/tests/test_builder/test_builder.py @@ -78,10 +78,13 @@ def test_expr_dag_different_output_fields_not_merged(): GROUP BY ad_id """ graph = builder.build_from_sql(sql, name="dedup") - expr_nodes = [n for n in graph.nodes if n.node_type.value == "transform"] - assert len(expr_nodes) == 2 - assert {n.output_name for n in expr_nodes} == {"ctr", "ctr2"} - assert len({n.fingerprint for n in expr_nodes}) == 1 + root_nodes = [ + n for n in graph.nodes + if n.node_type.value == "transform" and n.output_name in {"ctr", "ctr2"} + ] + assert len(root_nodes) == 2 + assert {n.output_name for n in root_nodes} == {"ctr", "ctr2"} + assert len({n.fingerprint for n in root_nodes}) == 1 def test_expr_dag_same_logic_same_output_field_merged_across_sql(): @@ -127,15 +130,17 @@ def test_expr_dag_commutative_order_is_not_normalized(): assert len({n.fingerprint for n in expr_nodes}) == 2 -def test_composite_expression_single_node(): - """复合表达式整体作为一个节点,不再拆成子表达式""" +def test_composite_expression_keeps_root_and_operand_dag(): + """复合表达式保留稳定根节点,同时显式记录内部操作数关系""" builder = GraphBuilder(dialect="spark") graph = builder.build_from_sql( "INSERT OVERWRITE TABLE t SELECT ROUND(SUM(x) / COUNT(*), 4) AS r FROM e", name="single") - expr_nodes = [n for n in graph.nodes if n.node_type.value == "transform"] - assert len(expr_nodes) == 1 - # 不应再产生表达式内部的操作数边 - assert len(graph.get_edges_by_type(EdgeType.EXPR_OPERAND)) == 0 + root_nodes = [ + n for n in graph.nodes + if n.node_type.value == "transform" and n.output_name == "r" + ] + assert len(root_nodes) == 1 + assert graph.get_edges_by_type(EdgeType.EXPR_OPERAND) # 表达式引用的物理列产生计算依赖边(x 一个来源列) assert len(graph.get_edges_by_type(EdgeType.COMPUTE_DEPENDENCY)) >= 1 @@ -259,9 +264,12 @@ def test_build_lateral_view_output_inside_expression_uses_generator_source(): } assert "src.items" in full_columns assert "src.item" not in full_columns - transforms = [n for n in graph.nodes if n.node_type.value == "transform"] - assert len(transforms) == 1 - assert transforms[0].expression == "CONCAT(EXPLODE(src.items), '_x')" + roots = [ + n for n in graph.nodes + if n.node_type.value == "transform" and n.output_name == "item_x" + ] + assert len(roots) == 1 + assert roots[0].expression == "CONCAT(EXPLODE(src.items), '_x')" def test_build_cte_columns_are_connected_across_subqueries(): diff --git a/tests/test_cli/test_governance_cli.py b/tests/test_cli/test_governance_cli.py new file mode 100644 index 0000000..8dc448b --- /dev/null +++ b/tests/test_cli/test_governance_cli.py @@ -0,0 +1,38 @@ +from __future__ import annotations + +from pathlib import Path + +from typer.testing import CliRunner + +from sqlgraph.cli import app + + +ROOT = Path(__file__).parents[2] + + +def test_governance_run_replay_and_verify(tmp_path): + runner = CliRunner() + scenario = ROOT / "examples" / "minimal" / "scenario.yaml" + output = tmp_path / "result" + + run = runner.invoke( + app, + ["governance", "run", str(scenario), "-o", str(output)], + ) + assert run.exit_code == 0, run.stdout + assert "outcome=success" in run.stdout + + verify = runner.invoke( + app, + ["governance", "verify", str(output)], + ) + assert verify.exit_code == 0, verify.stdout + assert "valid=True" in verify.stdout + + replay = runner.invoke( + app, + ["governance", "replay", str(output / "audit.jsonl")], + ) + assert replay.exit_code == 0, replay.stdout + assert "observe" in replay.stdout + assert "learn" in replay.stdout diff --git a/tests/test_evidence/__init__.py b/tests/test_evidence/__init__.py new file mode 100644 index 0000000..f283378 --- /dev/null +++ b/tests/test_evidence/__init__.py @@ -0,0 +1 @@ +"""Evidence engine tests.""" diff --git a/tests/test_evidence/test_engine.py b/tests/test_evidence/test_engine.py new file mode 100644 index 0000000..a3e54d6 --- /dev/null +++ b/tests/test_evidence/test_engine.py @@ -0,0 +1,70 @@ +from __future__ import annotations + +from dataclasses import replace + +from sqlgraph.api import build_graph +from sqlgraph.evidence import EvidenceEngine, EvidenceRequest + + +def _request() -> EvidenceRequest: + return EvidenceRequest( + task_id="task-1", + baseline_id="base_1", + intent="caliber_repair", + anchors=("src", "dst"), + direction="both", + max_depth=1, + coverage_obligations=( + "anchors_resolved", + "lineage_path", + "counterevidence_checked", + "no_unresolved", + ), + ) + + +def test_evidence_bundle_contains_support_counterevidence_and_exclusions(): + graph = build_graph( + "INSERT INTO dst SELECT s.id FROM src s JOIN alternate a ON s.id=a.id; " + "INSERT INTO downstream SELECT id FROM dst;", + dialect="spark", + ) + + bundle = EvidenceEngine(graph).collect(_request()) + + assert bundle.supporting + assert any(item.kind == "alternative_upstream" for item in bundle.counterevidence) + assert bundle.coverage_contract["required"] + assert bundle.subgraph_hash.startswith("evidence_") + + +def test_removed_required_evidence_cannot_remain_sufficient(): + graph = build_graph( + "INSERT INTO dst SELECT id FROM src", + dialect="spark", + ) + engine = EvidenceEngine(graph) + bundle = engine.collect(_request()) + + stripped = replace(bundle, included_edges=()) + decision = engine.assess(stripped) + + assert not decision.sufficient + assert decision.action in {"expand", "degrade", "refuse", "escalate"} + + +def test_expansion_creates_new_version_and_history(): + graph = build_graph( + "INSERT INTO mid SELECT id FROM src; " + "INSERT INTO dst SELECT id FROM mid;", + dialect="spark", + ) + request = replace(_request(), max_depth=0) + engine = EvidenceEngine(graph) + first = engine.collect(request) + + expanded = engine.expand(first, "lineage path not covered") + + assert expanded.version == first.version + 1 + assert expanded.parent_hash == first.subgraph_hash + assert expanded.history[-1]["reason"] == "lineage path not covered" diff --git a/tests/test_graph_advanced/__init__.py b/tests/test_graph_advanced/__init__.py new file mode 100644 index 0000000..94b5742 --- /dev/null +++ b/tests/test_graph_advanced/__init__.py @@ -0,0 +1 @@ +"""Advanced graph contract tests.""" diff --git a/tests/test_graph_advanced/test_drilldown.py b/tests/test_graph_advanced/test_drilldown.py new file mode 100644 index 0000000..7872af8 --- /dev/null +++ b/tests/test_graph_advanced/test_drilldown.py @@ -0,0 +1,40 @@ +from __future__ import annotations + +from sqlgraph.analyze.loader import load_analysis_view +from sqlgraph.analyze.table_graph import build_table_graph +from sqlgraph.api import build_graph +from sqlgraph.lineage import drilldown + + +def test_table_edge_drills_to_statement_columns_and_transforms(): + graph = build_graph( + "INSERT INTO dst " + "SELECT CASE WHEN status='A' THEN amount ELSE 0 END AS value " + "FROM src", + dialect="spark", + ) + + result = drilldown(graph, "src", "dst") + + assert result["found"] + assert result["statements"] + path = next( + item for item in result["column_paths"] + if item["target_column"] == "dst.value" + ) + assert set(path["source_columns"]) == {"src.amount", "src.status"} + assert path["transform"] + + +def test_table_projection_keeps_rebuild_references(): + graph = build_graph( + "INSERT INTO dst SELECT amount * 2 AS value FROM src", + dialect="spark", + ) + + table_graph = build_table_graph(load_analysis_view(graph)) + edge = table_graph.edges[0] + + assert edge.statement_refs + assert edge.column_dependency_ids + assert edge.transform_ids diff --git a/tests/test_graph_advanced/test_statement_lineage.py b/tests/test_graph_advanced/test_statement_lineage.py new file mode 100644 index 0000000..aa01af0 --- /dev/null +++ b/tests/test_graph_advanced/test_statement_lineage.py @@ -0,0 +1,24 @@ +from __future__ import annotations + +from sqlgraph.api import build_graph +from sqlgraph.model import EdgeType, NodeType + + +def test_expression_operands_are_explicit_graph_edges(): + graph = build_graph( + "INSERT INTO d SELECT a + b * c AS result FROM s", + dialect="spark", + ) + + transforms = { + node.id: node + for node in graph.get_nodes_by_type(NodeType.TRANSFORM) + } + operand_edges = graph.get_edges_by_type(EdgeType.EXPR_OPERAND) + + assert operand_edges + assert any( + transforms[edge.source_id].op == "mul" + and transforms[edge.target_id].op == "add" + for edge in operand_edges + ) diff --git a/tests/test_graphrag/__init__.py b/tests/test_graphrag/__init__.py new file mode 100644 index 0000000..065c696 --- /dev/null +++ b/tests/test_graphrag/__init__.py @@ -0,0 +1 @@ +"""GraphRAG grounding tests.""" diff --git a/tests/test_graphrag/test_grounding.py b/tests/test_graphrag/test_grounding.py new file mode 100644 index 0000000..7af49b4 --- /dev/null +++ b/tests/test_graphrag/test_grounding.py @@ -0,0 +1,70 @@ +from __future__ import annotations + +from sqlgraph.api import build_graph +from sqlgraph.evidence import EvidenceEngine, EvidenceRequest +from sqlgraph.graphrag import GroundedAssertion, validate_assertions + + +def _evidence(): + graph = build_graph( + "INSERT INTO dst SELECT id FROM src", + dialect="spark", + ) + request = EvidenceRequest( + task_id="task-1", + baseline_id="base-1", + intent="lineage", + anchors=("src", "dst"), + max_depth=1, + ) + return EvidenceEngine(graph).collect(request) + + +def test_nonexistent_reference_rejects_assertion(): + evidence = _evidence() + report = validate_assertions( + [GroundedAssertion( + statement="dst depends on src", + citations=("edge_missing",), + baseline_id=evidence.baseline_id, + evidence_hash=evidence.subgraph_hash, + )], + evidence, + ) + + assert report.status == "rejected" + assert report.invalid_citations == ("edge_missing",) + + +def test_valid_reference_is_grounded(): + evidence = _evidence() + citation = evidence.supporting[0].citations[0] + report = validate_assertions( + [GroundedAssertion( + statement="dst depends on src", + citations=(citation,), + baseline_id=evidence.baseline_id, + evidence_hash=evidence.subgraph_hash, + )], + evidence, + ) + + assert report.status == "grounded" + assert report.valid_assertions == 1 + + +def test_cross_baseline_reference_is_rejected(): + evidence = _evidence() + citation = evidence.supporting[0].citations[0] + report = validate_assertions( + [GroundedAssertion( + statement="dst depends on src", + citations=(citation,), + baseline_id="base-other", + evidence_hash=evidence.subgraph_hash, + )], + evidence, + ) + + assert report.status == "rejected" + assert report.version_mismatches diff --git a/tests/test_identity/__init__.py b/tests/test_identity/__init__.py new file mode 100644 index 0000000..78ef595 --- /dev/null +++ b/tests/test_identity/__init__.py @@ -0,0 +1 @@ +"""Deterministic identity tests.""" diff --git a/tests/test_identity/test_determinism.py b/tests/test_identity/test_determinism.py new file mode 100644 index 0000000..5153627 --- /dev/null +++ b/tests/test_identity/test_determinism.py @@ -0,0 +1,55 @@ +from __future__ import annotations + +from sqlgraph.api import build_graph +from sqlgraph.model import EdgeType + + +def _all_ids(graph): + return sorted([node.id for node in graph.nodes] + [edge.id for edge in graph.edges]) + + +def _table_pairs(graph): + names = {node.id: node.name for node in graph.nodes} + return { + (names[edge.source_id], names[edge.target_id]) + for edge in graph.edges + if edge.edge_type == EdgeType.TABLE_LINEAGE + } + + +def test_two_builds_have_identical_full_ids(): + sql = "INSERT INTO d SELECT a + b AS c FROM s" + + assert _all_ids(build_graph(sql, dialect="spark")) == _all_ids( + build_graph(sql, dialect="spark") + ) + + +def test_independent_statements_do_not_cross_link(): + graph = build_graph( + "INSERT INTO d1 SELECT * FROM s1; INSERT INTO d2 SELECT * FROM s2;", + dialect="spark", + ) + + assert _table_pairs(graph) == {("s1", "d1"), ("s2", "d2")} + + +def test_table_lineage_keeps_statement_provenance(): + graph = build_graph( + "INSERT INTO d1 SELECT * FROM s1; INSERT INTO d2 SELECT * FROM s2;", + dialect="spark", + ) + + edges = graph.get_edges_by_type(EdgeType.TABLE_LINEAGE) + assert edges + for edge in edges: + assert edge.properties["provenance"] + assert {"sql_id", "stmt_index"} <= edge.properties["provenance"][0].keys() + + +def test_graph_contains_reproducible_environment_and_coverage(): + graph = build_graph("INSERT INTO d SELECT a FROM s", dialect="spark") + + assert graph.metadata["environment"]["identity_rule_version"] + assert graph.metadata["environment"]["dialect"] == "spark" + assert graph.metadata["coverage"]["ok"] == 1 diff --git a/tests/test_integration/test_book_cases.py b/tests/test_integration/test_book_cases.py new file mode 100644 index 0000000..90e441d --- /dev/null +++ b/tests/test_integration/test_book_cases.py @@ -0,0 +1,30 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest + +from sqlgraph.audit import AuditLog +from sqlgraph.reasoning.scenario import run_scenario + + +ROOT = Path(__file__).parents[2] / "examples" / "book_cases" + + +@pytest.mark.parametrize( + ("case", "outcome"), + [ + ("caliber_consistency", "success"), + ("cold_table_retirement", "success"), + ("irreversible_drop", "held_for_human_review"), + ], +) +def test_book_case_exports_replayable_bundle(case, outcome, tmp_path): + result = run_scenario(ROOT / case / "scenario.yaml", tmp_path / case) + + assert result.outcome == outcome + assert AuditLog(result.audit_path).verify_integrity().valid + assert len(result.events) == 7 + if case == "irreversible_drop": + assert result.decision.reversibility_veto + assert result.execution.status == "blocked" diff --git a/tests/test_reasoning/__init__.py b/tests/test_reasoning/__init__.py new file mode 100644 index 0000000..28e0afe --- /dev/null +++ b/tests/test_reasoning/__init__.py @@ -0,0 +1 @@ +"""Governance runner tests.""" diff --git a/tests/test_reasoning/test_runner.py b/tests/test_reasoning/test_runner.py new file mode 100644 index 0000000..2bb93d6 --- /dev/null +++ b/tests/test_reasoning/test_runner.py @@ -0,0 +1,85 @@ +from __future__ import annotations + +import hashlib + +from sqlgraph.actions import ActionEngine, SqlFilePatchAdapter +from sqlgraph.audit import AuditLog +from sqlgraph.autonomy import ( + AuthorizationScope, + GovernanceAction, + ReversibilityEvidence, +) +from sqlgraph.baseline import build_baseline +from sqlgraph.evidence import EvidenceEngine, EvidenceRequest +from sqlgraph.input import SqlSource +from sqlgraph.reasoning import GovernanceRequest, GovernanceRunner +from sqlgraph.verification import VerificationEngine + + +def test_runner_exports_all_seven_steps(tmp_path): + path = tmp_path / "query.sql" + before = "INSERT INTO dst SELECT value * 100 AS value FROM src" + after = "INSERT INTO dst SELECT value AS value FROM src" + path.write_text(before, encoding="utf-8") + source = SqlSource.from_file(str(path)) + baseline = build_baseline(source, dialect="spark") + from sqlgraph.api import build_graph + + graph = build_graph(source, dialect="spark") + evidence_request = EvidenceRequest( + task_id="task-1", + baseline_id=baseline.baseline_id, + intent="caliber_repair", + anchors=("src", "dst"), + max_depth=1, + ) + action = GovernanceAction( + action_type="sql_patch", + evidence_version="pending", + evidence_grounded=True, + reversibility=ReversibilityEvidence(True, True, True), + authorization_scope=AuthorizationScope.SINGLE_L3, + ) + request = GovernanceRequest( + task_id="task-1", + baseline=baseline, + graph=graph, + evidence_request=evidence_request, + action=action, + adapter="sql_file_patch", + operations=({ + "path": str(path), + "before": before, + "after": after, + "expected_sha256": hashlib.sha256(before.encode()).hexdigest(), + },), + source_path=str(path), + source_table="src", + target_table="dst", + dialect="spark", + runtime_observed={ + "checks_passed": True, + "reports_recomputed": True, + "new_alerts": 0, + }, + ) + runner = GovernanceRunner( + EvidenceEngine(graph), + ActionEngine([SqlFilePatchAdapter()]), + VerificationEngine(), + AuditLog(tmp_path / "audit.jsonl"), + ) + + result = runner.run(request) + + assert [event.step for event in result.events] == [ + "observe", + "explain", + "propose", + "authorize", + "execute", + "verify", + "learn", + ] + assert result.outcome == "success" + assert AuditLog(result.audit_path).verify_integrity().valid diff --git a/tests/test_release/__init__.py b/tests/test_release/__init__.py new file mode 100644 index 0000000..1d8582a --- /dev/null +++ b/tests/test_release/__init__.py @@ -0,0 +1 @@ +"""Open-source release regression tests.""" diff --git a/tests/test_release/test_remote_capabilities.py b/tests/test_release/test_remote_capabilities.py new file mode 100644 index 0000000..82c2894 --- /dev/null +++ b/tests/test_release/test_remote_capabilities.py @@ -0,0 +1,20 @@ +"""Regression protection for capabilities already published on GitHub.""" + +from typer.main import get_command + +from sqlgraph.cli import app + + +def test_public_cli_keeps_existing_github_commands(): + commands = get_command(app).commands + + for command in ( + "build", + "stats", + "analyze", + "profile", + "serve", + "playground", + "demo", + ): + assert command in commands diff --git a/tests/test_verification/__init__.py b/tests/test_verification/__init__.py new file mode 100644 index 0000000..621ac31 --- /dev/null +++ b/tests/test_verification/__init__.py @@ -0,0 +1 @@ +"""Independent verification tests.""" diff --git a/tests/test_verification/test_engine.py b/tests/test_verification/test_engine.py new file mode 100644 index 0000000..7559aa4 --- /dev/null +++ b/tests/test_verification/test_engine.py @@ -0,0 +1,79 @@ +from __future__ import annotations + +import hashlib + +from sqlgraph.actions import ActionEngine, ActionRequest, SqlFilePatchAdapter +from sqlgraph.autonomy import ( + AuthorizationScope, + GovernanceAction, + ReversibilityEvidence, + decide_autonomy, +) +from sqlgraph.verification import ( + LayerResult, + VerificationEngine, + VerificationReport, +) + + +def _executed_patch(tmp_path): + path = tmp_path / "query.sql" + before = "INSERT INTO dst SELECT value * 100 AS value FROM src" + after = "INSERT INTO dst SELECT value AS value FROM src" + path.write_text(before, encoding="utf-8") + decision = decide_autonomy(GovernanceAction( + action_type="sql_patch", + evidence_version="task-1@v1#abc", + evidence_grounded=True, + reversibility=ReversibilityEvidence(True, True, True), + authorization_scope=AuthorizationScope.SINGLE_L3, + )) + request = ActionRequest( + task_id="task-1", + baseline_id="base-1", + evidence_version=decision.evidence_version, + decision=decision, + adapter="sql_file_patch", + operations=({ + "path": str(path), + "before": before, + "after": after, + "expected_sha256": hashlib.sha256(before.encode()).hexdigest(), + },), + ) + engine = ActionEngine([SqlFilePatchAdapter()]) + plan = engine.plan(request) + execution = engine.execute(plan) + return plan, execution, path + + +def test_structure_verification_rebuilds_from_changed_source(tmp_path): + plan, execution, path = _executed_patch(tmp_path) + + report = VerificationEngine().verify( + plan, + execution, + source_path=path, + source_table="src", + target_table="dst", + dialect="spark", + runtime_observed={ + "checks_passed": True, + "reports_recomputed": True, + "new_alerts": 0, + }, + ) + + assert report.structure.status == "pass" + assert report.structure.evidence["rebuilt_from_source"] is True + assert report.outcome == "success" + + +def test_any_critical_failure_blocks_success(): + report = VerificationReport( + code=LayerResult("code", "pass"), + structure=LayerResult("structure", "fail"), + runtime=LayerResult("runtime", "pass"), + ) + + assert report.outcome == "failed" diff --git a/tests/test_video_warehouse/__init__.py b/tests/test_video_warehouse/__init__.py new file mode 100644 index 0000000..8a257f8 --- /dev/null +++ b/tests/test_video_warehouse/__init__.py @@ -0,0 +1 @@ +"""视频平台商业化数仓验收测试。""" diff --git a/tests/test_video_warehouse/test_catalog.py b/tests/test_video_warehouse/test_catalog.py new file mode 100644 index 0000000..24b106b --- /dev/null +++ b/tests/test_video_warehouse/test_catalog.py @@ -0,0 +1,66 @@ +from collections import Counter + +from examples.video_commercial_warehouse.catalog import ( + ALL_TABLES, + BASE_TABLES, + TASKS, + layer_counts, + validate_catalog, +) + + +def test_catalog_exact_scale_and_unique_targets(): + assert len(BASE_TABLES) == 32 + assert len(TASKS) == 62 + assert len({task.target for task in TASKS}) == 62 + assert len(ALL_TABLES) == 94 + assert len(set(ALL_TABLES)) == 94 + + +def test_catalog_layer_counts(): + assert layer_counts() == { + "ODS": 18, + "STG": 18, + "DIM": 14, + "DWD": 18, + "DWS": 16, + "ADS": 10, + } + + +def test_catalog_covers_ten_business_domains(): + domains = {table.domain for table in BASE_TABLES.values()} + domains |= {task.domain for task in TASKS} + assert domains == { + "content", + "user", + "traffic", + "recommendation", + "ad_inventory", + "ad_delivery", + "attribution", + "billing", + "experiment", + "risk", + } + + +def test_task_dag_is_acyclic_and_topologically_ordered(): + validate_catalog() + positions = {task.target: index for index, task in enumerate(TASKS)} + for task in TASKS: + for dependency in task.dependencies: + if dependency in positions: + assert positions[dependency] < positions[task.target] + + +def test_each_derived_layer_has_expected_task_count(): + counts = Counter(task.layer for task in TASKS) + assert counts == {"STG": 18, "DWD": 18, "DWS": 16, "ADS": 10} + + +def test_every_task_has_executable_create_table_sql(): + for task in TASKS: + assert f"CREATE OR REPLACE TABLE {task.target} AS\n" in task.sql + assert task.sql.rstrip().endswith(";") + assert "SELECT" in task.sql.upper() diff --git a/tests/test_video_warehouse/test_generate_sql.py b/tests/test_video_warehouse/test_generate_sql.py new file mode 100644 index 0000000..228920e --- /dev/null +++ b/tests/test_video_warehouse/test_generate_sql.py @@ -0,0 +1,44 @@ +from __future__ import annotations + +import hashlib +import json + +from examples.video_commercial_warehouse.generate_sql import generate_sql + + +def _hashes(paths): + return { + path.name: hashlib.sha256(path.read_bytes()).hexdigest() + for path in paths + } + + +def test_generate_sql_is_complete_and_deterministic(tmp_path): + first = generate_sql(tmp_path) + first_hashes = _hashes(first) + second = generate_sql(tmp_path) + second_hashes = _hashes(second) + + assert len(first) == 62 + assert first_hashes == second_hashes + assert first[0].name == "001_stg_recommend_request.sql" + assert first[-1].name == "062_ads_risk_dashboard.sql" + + +def test_manifest_describes_all_tables_tasks_and_dependencies(tmp_path): + generate_sql(tmp_path) + manifest = json.loads((tmp_path / "manifest.json").read_text(encoding="utf-8")) + + assert manifest["table_count"] == 94 + assert manifest["base_table_count"] == 32 + assert manifest["task_count"] == 62 + assert len(manifest["tables"]) == 94 + assert len(manifest["tasks"]) == 62 + assert manifest["layer_counts"] == { + "ADS": 10, + "DIM": 14, + "DWD": 18, + "DWS": 16, + "ODS": 18, + "STG": 18, + } diff --git a/tests/test_video_warehouse/test_governance.py b/tests/test_video_warehouse/test_governance.py new file mode 100644 index 0000000..3e9fdca --- /dev/null +++ b/tests/test_video_warehouse/test_governance.py @@ -0,0 +1,106 @@ +from __future__ import annotations + +from pathlib import Path + +from examples.video_commercial_warehouse.generate_sql import generate_sql +from examples.video_commercial_warehouse.governance import ( + AFTER_CTR_EXPR, + BEFORE_CTR_EXPR, + restore_governance, + run_ctr_governance, +) + + +def _scenario(tmp_path: Path) -> Path: + root = tmp_path / "warehouse" + generate_sql(root) + return root + + +def test_ctr_governance_uses_real_duckdb_results(tmp_path): + root = _scenario(tmp_path) + result = run_ctr_governance(root, tmp_path / "out", profile="smoke") + + assert result["catalog"]["table_count"] == 94 + assert result["catalog"]["task_count"] == 62 + assert result["graph"]["coverage"]["failed"] == 0 + assert result["runtime"]["source"] == "duckdb_query" + assert result["runtime"]["before"]["invalid_ctr_rows"] > 0 + assert result["runtime"]["after"]["invalid_ctr_rows"] == 0 + assert result["runtime"]["after"]["grade_mismatch_rows"] == 0 + assert result["runtime"]["threshold_consistent"] is True + assert result["verify"]["closed_loop_status"] == "success" + assert result["audit"]["outcome"] == "closed" + + +def test_governance_only_reruns_affected_dag(tmp_path): + result = run_ctr_governance( + _scenario(tmp_path), tmp_path / "out", profile="smoke" + ) + assert result["rerun"]["tasks"] == [ + "dws_creative_performance_daily", + "ads_creative_report", + ] + assert result["rerun"]["success_count"] == 2 + assert set(result["impact"]["reachable"]) == {"ads_creative_report"} + + +def test_wrong_downstream_threshold_fails_runtime_verification(tmp_path): + root = _scenario(tmp_path) + report_sql = root / "sql" / "059_ads_creative_report.sql" + text = report_sql.read_text(encoding="utf-8") + text = ( + text.replace("p.ctr >= 0.05", "p.ctr >= 0.50") + .replace("p.ctr >= 0.03", "p.ctr >= 0.30") + .replace("p.ctr >= 0.01", "p.ctr >= 0.10") + ) + report_sql.write_text(text, encoding="utf-8") + + result = run_ctr_governance(root, tmp_path / "out", profile="smoke") + assert result["runtime"]["threshold_consistent"] is False + assert result["verify"]["contract"]["status"] == "fail" + assert result["verify"]["closed_loop_status"] == "failed" + assert result["audit"]["outcome"] == "failed" + + +def test_idempotent_rerun_is_recorded_as_noop(tmp_path): + root = _scenario(tmp_path) + output = tmp_path / "out" + first = run_ctr_governance(root, output, profile="smoke") + second = run_ctr_governance(root, output, profile="smoke") + + assert first["change"]["changed"] is True + assert second["change"]["changed"] is False + execute = next(step for step in second["audit"]["steps"] if step["step"] == "execute") + assert execute["detail"]["executed"] is False + assert execute["detail"]["mode"] == "no_op/already_compliant" + + +def test_sql_fix_is_exact_and_backup_is_preserved(tmp_path): + root = _scenario(tmp_path) + output = tmp_path / "out" + target = root / "sql" / "044_dws_creative_performance_daily.sql" + original = target.read_text(encoding="utf-8") + + result = run_ctr_governance(root, output, profile="smoke") + changed = target.read_text(encoding="utf-8") + assert result["change"]["changed"] is True + assert BEFORE_CTR_EXPR in original + assert BEFORE_CTR_EXPR not in changed + assert AFTER_CTR_EXPR in changed + assert (output / "before" / target.name).read_text(encoding="utf-8") == original + + +def test_restore_updates_sql_database_and_audit_state(tmp_path): + root = _scenario(tmp_path) + output = tmp_path / "out" + run_ctr_governance(root, output, profile="smoke") + + restored = restore_governance(root, output) + target = root / "sql" / "044_dws_creative_performance_daily.sql" + assert BEFORE_CTR_EXPR in target.read_text(encoding="utf-8") + assert restored["scenario"]["status"] == "restored" + assert restored["audit"]["outcome"] == "restored" + assert restored["verify"]["closed_loop_status"] == "incomplete" + assert restored["runtime"]["after"]["invalid_ctr_rows"] > 0 + assert restored["audit"]["steps"][-1]["step"] == "restore" diff --git a/tests/test_video_warehouse/test_report.py b/tests/test_video_warehouse/test_report.py new file mode 100644 index 0000000..f3bd77a --- /dev/null +++ b/tests/test_video_warehouse/test_report.py @@ -0,0 +1,143 @@ +from __future__ import annotations + +from examples.video_commercial_warehouse.generate_sql import generate_sql +from examples.video_commercial_warehouse.governance import run_ctr_governance +from examples.video_commercial_warehouse.report import render_report +from sqlgraph.serve.theme import load_explorer_css + + +def test_report_contains_real_warehouse_evidence(tmp_path): + root = tmp_path / "warehouse" + output = tmp_path / "out" + generate_sql(root) + result = run_ctr_governance(root, output, profile="smoke") + path = render_report(result, output / "warehouse_report.html") + text = path.read_text(encoding="utf-8") + + assert '"table_count":94' in text + assert '"task_count":62' in text + assert '"source":"duckdb_query"' in text + assert "data-node-id" in text + assert "switchView(" in text + assert "setLayer(" in text + assert "setDomain(" in text + assert "selectTable(" in text + assert "https://" not in text + + +def test_report_uses_html_buttons_for_reliable_node_clicks(tmp_path): + root = tmp_path / "warehouse" + output = tmp_path / "out" + generate_sql(root) + result = run_ctr_governance(root, output, profile="smoke") + path = render_report(result, output / "warehouse_report.html") + text = path.read_text(encoding="utf-8") + + assert '
0 + assert result.table_rows["ads_revenue_dashboard"] > 0 + + +def test_smoke_build_is_deterministic(tmp_path): + first = build_warehouse(tmp_path / "first.duckdb", profile="smoke") + second = build_warehouse(tmp_path / "second.duckdb", profile="smoke") + assert first.table_rows == second.table_rows + + with duckdb.connect(str(tmp_path / "first.duckdb"), read_only=True) as con1: + rows1 = con1.execute( + "SELECT * FROM ads_creative_report ORDER BY ad_creative_id, event_date" + ).fetchall() + with duckdb.connect(str(tmp_path / "second.duckdb"), read_only=True) as con2: + rows2 = con2.execute( + "SELECT * FROM ads_creative_report ORDER BY ad_creative_id, event_date" + ).fetchall() + assert rows1 == rows2 + + +def test_profiles_have_declared_scale(): + smoke = PROFILES["smoke"] + demo = PROFILES["demo"] + assert smoke.recommend_requests == 5000 + assert demo.recommend_requests == 200_000 + assert demo.users > smoke.users + assert demo.videos > smoke.videos + + +def test_default_runtime_does_not_overwrite_existing_sql(tmp_path, monkeypatch): + import examples.video_commercial_warehouse.runtime as runtime + + root = tmp_path / "warehouse" + generate_sql(root) + target = root / "sql" / "044_dws_creative_performance_daily.sql" + marker = "-- user-governed\n" + target.write_text(marker + target.read_text(encoding="utf-8"), encoding="utf-8") + monkeypatch.setattr(runtime, "ROOT", root) + + build_warehouse(tmp_path / "warehouse.duckdb", profile="smoke") + assert target.read_text(encoding="utf-8").startswith(marker) diff --git a/tools/benchmark_scale.py b/tools/benchmark_scale.py new file mode 100755 index 0000000..90a300d --- /dev/null +++ b/tools/benchmark_scale.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Generate deterministic SQL scale benchmarks.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import statistics +import sys +import time +import tracemalloc +from dataclasses import asdict, dataclass +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from sqlgraph.api import build_graph +from sqlgraph.input import SqlSource, SqlSourceItem + + +@dataclass(frozen=True) +class ScaleResult: + sql_count: int + parsed: int + elapsed_ms: float + p50_ms: float + p95_ms: float + peak_memory_mb: float + node_count: int + edge_count: int + unknown_count: int + artifact_bytes: int + deterministic: bool + + +def _source(count: int) -> SqlSource: + source = SqlSource() + for index in range(count): + source.add_item(SqlSourceItem( + name=f"task_{index:05d}", + content=( + f"INSERT INTO table_{index + 1:05d} " + f"SELECT id, value + {index} AS value FROM table_{index:05d}" + ), + )) + return source + + +def _digest(graph) -> str: + ids = sorted( + [node.id for node in graph.nodes] + [edge.id for edge in graph.edges] + ) + return hashlib.sha256("\n".join(ids).encode("utf-8")).hexdigest() + + +def run_scale_benchmark(count: int) -> ScaleResult: + sample_times = [] + for index in range(min(count, 20)): + started = time.perf_counter() + build_graph( + f"INSERT INTO d{index} SELECT id FROM s{index}", + dialect="spark", + ) + sample_times.append((time.perf_counter() - started) * 1000) + + source = _source(count) + tracemalloc.start() + started = time.perf_counter() + graph = build_graph(source, dialect="spark") + elapsed_ms = (time.perf_counter() - started) * 1000 + _, peak = tracemalloc.get_traced_memory() + tracemalloc.stop() + second = build_graph(_source(count), dialect="spark") + payload = json.dumps(graph.to_dict(), sort_keys=True, ensure_ascii=True) + unknown_count = sum( + getattr(node, "name", "") == "UNKNOWN" for node in graph.nodes + ) + ordered = sorted(sample_times) + p95_index = max(0, min(len(ordered) - 1, int(len(ordered) * 0.95) - 1)) + return ScaleResult( + sql_count=count, + parsed=graph.metadata["coverage"]["ok"], + elapsed_ms=round(elapsed_ms, 3), + p50_ms=round(statistics.median(sample_times), 3), + p95_ms=round(ordered[p95_index], 3), + peak_memory_mb=round(peak / 1024 / 1024, 3), + node_count=len(graph.nodes), + edge_count=len(graph.edges), + unknown_count=unknown_count, + artifact_bytes=len(payload.encode("utf-8")), + deterministic=_digest(graph) == _digest(second), + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--counts", default="100") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + results = [ + asdict(run_scale_benchmark(int(raw))) + for raw in args.counts.split(",") + ] + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text( + json.dumps(results, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + print(args.output) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/check_capabilities.py b/tools/check_capabilities.py new file mode 100755 index 0000000..f05eecb --- /dev/null +++ b/tools/check_capabilities.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Validate the public capability ledger against the repository.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import jsonschema +import yaml + + +ROOT = Path(__file__).resolve().parents[1] +LEDGER = ROOT / "CAPABILITIES.yaml" +SCHEMA = ROOT / "schemas" / "capabilities-v1.schema.json" + + +def check_capabilities(root: Path = ROOT) -> list[str]: + ledger_path = root / "CAPABILITIES.yaml" + schema_path = root / "schemas" / "capabilities-v1.schema.json" + payload = yaml.safe_load(ledger_path.read_text(encoding="utf-8")) + schema = json.loads(schema_path.read_text(encoding="utf-8")) + errors = [] + try: + jsonschema.validate(payload, schema) + except jsonschema.ValidationError as exc: + errors.append(f"schema: {exc.message}") + return errors + + seen = set() + for capability in payload["capabilities"]: + capability_id = capability["id"] + if capability_id in seen: + errors.append(f"duplicate capability id: {capability_id}") + seen.add(capability_id) + if capability["status"] != "implemented": + continue + for field in ("modules", "tests", "examples"): + if not capability[field]: + errors.append( + f"{capability_id}: implemented capability has no {field}" + ) + for relative in capability[field]: + if not (root / relative).exists(): + errors.append( + f"{capability_id}: missing {field} path {relative}" + ) + return errors + + +def main() -> int: + errors = check_capabilities() + if errors: + for error in errors: + print(error) + return 1 + print("Capability ledger passed.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tools/check_coverage.py b/tools/check_coverage.py new file mode 100644 index 0000000..948810d --- /dev/null +++ b/tools/check_coverage.py @@ -0,0 +1,76 @@ +# tools/check_coverage.py +"""地基层覆盖率闸门(REQ-CLI-02 AC2)。 + +规格书要求:测试覆盖率对**地基层**(graph / identity / lineage)达到约定阈值 +(建议 ≥ 85%)。本脚本读取 `coverage` 生成的 JSON 报告,只统计地基层文件的 +合计覆盖率,低于阈值即退出码 1,用于在 CI 中阻断合并。 + +用法(先跑测试并生成 coverage.json,再校验): + COVERAGE_CORE=sysmon coverage run -m pytest -q + coverage json -o coverage.json + python -m tools.check_coverage --report coverage.json --min 85 + +地基层文件由路径前缀识别:sqlgraph/model/、sqlgraph/identity/、sqlgraph/lineage/。 +""" +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +# 地基层:确定性图模型 + 身份服务 + 血缘下钻(书稿第 2-3 章的确定性地基)。 +FOUNDATION_PREFIXES = ( + "sqlgraph/model/", + "sqlgraph/identity/", + "sqlgraph/lineage/", +) + + +def _is_foundation(path: str) -> bool: + norm = path.replace("\\", "/") + return any(seg in norm for seg in FOUNDATION_PREFIXES) + + +def foundation_coverage(report: dict) -> tuple[int, int, float]: + """从 coverage JSON 汇总地基层的 (已覆盖语句, 总语句, 覆盖率%)。""" + covered = 0 + total = 0 + for path, data in report.get("files", {}).items(): + if not _is_foundation(path): + continue + summary = data.get("summary", {}) + total += summary.get("num_statements", 0) + covered += summary.get("covered_lines", 0) + pct = (covered / total * 100.0) if total else 0.0 + return covered, total, pct + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="地基层覆盖率闸门(REQ-CLI-02 AC2)") + parser.add_argument("--report", default="coverage.json", help="coverage json 报告路径") + parser.add_argument("--min", type=float, default=85.0, help="地基层最低覆盖率阈值(百分比)") + args = parser.parse_args(argv) + + report_path = Path(args.report) + if not report_path.exists(): + print(f"未找到覆盖率报告 {report_path};请先运行 `coverage json -o {report_path}`。") + return 2 + + report = json.loads(report_path.read_text(encoding="utf-8")) + covered, total, pct = foundation_coverage(report) + if total == 0: + print("未在覆盖率报告中找到地基层文件(model/identity/lineage)。") + return 2 + + status = "达标" if pct >= args.min else "未达标" + print(f"地基层覆盖率(model/identity/lineage):{covered}/{total} = {pct:.2f}% " + f"(阈值 {args.min:.0f}%)→ {status}") + if pct < args.min: + print("覆盖率低于阈值,阻断合并(REQ-CLI-02 AC2)。", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/check_layering.py b/tools/check_layering.py new file mode 100644 index 0000000..837afca --- /dev/null +++ b/tools/check_layering.py @@ -0,0 +1,175 @@ +# tools/check_layering.py +"""分层依赖单向性检查(REQ-ARCH-01)。 + +《AI 原生数仓治理》第 2-3 章把「解析 / 图 / 证据」定义为确定性地基,第 4-7 章 +把「推理循环(决策 / 执行)」定义为上层操作化展开。地基不得依赖推理结果, +否则确定性会被上层污染。本脚本把该约束固化为可执行、可纳入 CI 的守卫: + + 上层可以 import 下层;下层禁止 import 上层。 + +用法: + python -m tools.check_layering # 检查真实仓库,反向依赖时退出码 1 + python -m tools.check_layering --json # 结构化输出 + +也可作为库被验收测试调用:``scan_imports`` / ``classify``。 + +层级映射(真实子包 -> 规格书建议的层名 -> 层级序号):规格书建议的模块名 +(adapters/graph/...)与本仓库既有命名(input+parser/model+builder/...)不完全 +一致,这里按语义对齐,并在报告中给出层名,避免读者困惑。 +""" +from __future__ import annotations + +import argparse +import ast +import json +from dataclasses import dataclass +from pathlib import Path + +# 真实子包 -> (层级序号, 规格书层名)。序号越小越靠近确定性地基。 +# 规则:一个子包只能 import 层级序号 <= 自身的子包(含同层)。 +LAYERS: dict[str, tuple[int, str]] = { + # L0 共享基础设施:无业务语义,任何层都可依赖。 + "utils": (0, "shared"), + # L1 确定性地基:数据模型 + 身份服务。 + "model": (1, "graph"), + "identity": (1, "identity"), + # L2-L3 适配器:输入装载 + SQL 解析(graph 的上游、身份的下游)。 + "input": (2, "adapters"), + "parser": (3, "adapters"), + # L4 图构建:把解析结果编译为确定性图。 + "builder": (4, "graph"), + # L5 图派生只读服务:序列化 / 可视化 / 血缘下钻 / 结构指标 / 契约。 + "serialize": (5, "graph-derived"), + "visualize": (5, "graph-derived"), + "lineage": (5, "lineage"), + "metrics": (5, "metrics"), + "contract": (5, "adapters-contract"), + # L6 证据子图。 + "evidence": (6, "evidence"), + # L7 验证与决策(推理循环的判定层)。 + "verify": (7, "verify"), + "autonomy": (7, "autonomy"), + # L8 智能体:编排七步闭环。 + "agent": (8, "agent"), + # L9-L10 入口层:高层 API / playground / CLI。 + "api": (9, "entry"), + "playground": (9, "entry"), + "cli": (10, "cli"), +} + + +@dataclass(frozen=True) +class Violation: + """一条反向依赖:``importer`` 属于下层,却 import 了上层 ``imported``。""" + + importer_pkg: str + imported_pkg: str + importer_level: int + imported_level: int + location: str # "文件:行号" + + def __str__(self) -> str: # pragma: no cover - 仅用于 CLI 展示 + return ( + f"{self.location}: 下层 '{self.importer_pkg}'(L{self.importer_level}) " + f"禁止 import 上层 '{self.imported_pkg}'(L{self.imported_level})" + ) + + +def classify(importer_pkg: str, imported_pkg: str) -> bool: + """判断一条包间依赖是否构成反向依赖(违规)。 + + Returns: + True 表示违规(下层 import 了严格上层);False 表示合法(含同层、 + 依赖下层,或任一包不在受管层级表中而无法判定)。 + """ + if importer_pkg == imported_pkg: + return False + a = LAYERS.get(importer_pkg) + b = LAYERS.get(imported_pkg) + if a is None or b is None: + return False + return b[0] > a[0] + + +def _imported_pkg(module: str, package_name: str) -> str | None: + """从被 import 的模块全名中取出其受管子包名。 + + ``sqlgraph.autonomy.decision`` -> ``autonomy``;``sqlgraph`` 门面 -> None。 + """ + if module == package_name: + return None + prefix = package_name + "." + if not module.startswith(prefix): + return None + return module[len(prefix):].split(".", 1)[0] + + +def _importer_pkg(py_file: Path, root: Path) -> str | None: + """从文件路径推断其所属受管子包名(root 直属文件如 api.py -> 'api')。""" + rel = py_file.relative_to(root) + parts = rel.parts + if len(parts) == 1: # 例如 api.py / cli.py / playground.py + return parts[0][:-3] if parts[0].endswith(".py") else parts[0] + return parts[0] # 例如 autonomy/decision.py -> 'autonomy' + + +def scan_imports(root: Path, package_name: str = "sqlgraph") -> list[Violation]: + """遍历包目录,用 AST 收集内部 import 并返回全部反向依赖。""" + violations: list[Violation] = [] + for py_file in sorted(root.rglob("*.py")): + if "__pycache__" in py_file.parts: + continue + importer = _importer_pkg(py_file, root) + if importer is None: + continue + try: + tree = ast.parse(py_file.read_text(encoding="utf-8"), filename=str(py_file)) + except SyntaxError: + continue + for node in ast.walk(tree): + if isinstance(node, ast.ImportFrom): + if node.module and node.level == 0: + pkg = _imported_pkg(node.module, package_name) + if pkg and classify(importer, pkg): + violations.append(_mk(importer, pkg, py_file, node.lineno)) + elif isinstance(node, ast.Import): + for alias in node.names: + pkg = _imported_pkg(alias.name, package_name) + if pkg and classify(importer, pkg): + violations.append(_mk(importer, pkg, py_file, node.lineno)) + return violations + + +def _mk(importer: str, imported: str, py_file: Path, lineno: int) -> Violation: + return Violation( + importer_pkg=importer, + imported_pkg=imported, + importer_level=LAYERS[importer][0], + imported_level=LAYERS[imported][0], + location=f"{py_file}:{lineno}", + ) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="分层依赖单向性检查(REQ-ARCH-01)") + parser.add_argument("--root", default=None, help="被检查的包目录,默认自动定位 sqlgraph/") + parser.add_argument("--json", action="store_true", help="以 JSON 输出结果") + args = parser.parse_args(argv) + + root = Path(args.root) if args.root else Path(__file__).resolve().parent.parent / "sqlgraph" + violations = scan_imports(root) + + if args.json: + print(json.dumps([v.__dict__ for v in violations], ensure_ascii=False, indent=2)) + elif violations: + print(f"发现 {len(violations)} 条反向依赖(下层 import 上层),违反 REQ-ARCH-01:") + for v in violations: + print(f" - {v}") + else: + print("分层依赖检查通过:无反向依赖(REQ-ARCH-01 AC / 退出码 0)。") + + return 1 if violations else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/check_random_ids.py b/tools/check_random_ids.py new file mode 100644 index 0000000..61e734c --- /dev/null +++ b/tools/check_random_ids.py @@ -0,0 +1,134 @@ +# tools/check_random_ids.py +"""随机 ID 禁令扫描(REQ-ID-01 强制项 / REQ-CLI-02 TC1)。 + +书稿定稿态的两大修正点之一是「确定性 ID」:全仓库节点与边的 ID 必须由内容键 / +语义键派生,禁止随机 UUID/随机数。否则跨构建差分会把随机噪声误判为结构变化。 + +本脚本用 AST 静态扫描 ``sqlgraph/`` 下是否引入了随机性来源: + - ``import uuid`` / ``from uuid import ...`` / ``uuid.uuid1/uuid4`` 等; + - ``import random`` / ``random.*``; + - ``import secrets`` / ``secrets.*``; + - ``os.urandom``。 + +命中即视为违反禁令,退出码 1,阻断 CI(REQ-CLI-02 TC1:提交违反随机 ID 禁令 +的代码,CI 失败)。允许通过行内注释 ``# allow-random`` 显式豁免(例如与身份 +无关的抽样场景),豁免会被记录在报告中,做到「不静默放行」。 + +用法: + python -m tools.check_random_ids + python -m tools.check_random_ids --json +""" +from __future__ import annotations + +import argparse +import ast +import json +from dataclasses import dataclass +from pathlib import Path + +# 被禁止的随机性来源模块名。 +_FORBIDDEN_MODULES = {"uuid", "random", "secrets"} +# 被禁止的属性调用(模块.属性)。 +_FORBIDDEN_ATTRS = { + ("os", "urandom"), +} +_ALLOW_MARKER = "allow-random" + + +@dataclass(frozen=True) +class Finding: + """一处随机性来源命中。""" + + location: str # "文件:行号" + symbol: str # 命中的符号,如 "uuid" / "random.random" / "os.urandom" + allowed: bool # 是否被行内 # allow-random 豁免 + + def __str__(self) -> str: # pragma: no cover - 仅 CLI 展示 + tag = "(已豁免 allow-random)" if self.allowed else "" + return f"{self.location}: 检出随机性来源 '{self.symbol}'{tag}" + + +def _line_allows(source_lines: list[str], lineno: int) -> bool: + """该行是否带有 ``# allow-random`` 豁免标记。""" + if 1 <= lineno <= len(source_lines): + return _ALLOW_MARKER in source_lines[lineno - 1] + return False + + +def scan_file(py_file: Path) -> list[Finding]: + """扫描单个文件的随机性来源命中。""" + findings: list[Finding] = [] + text = py_file.read_text(encoding="utf-8") + lines = text.splitlines() + try: + tree = ast.parse(text, filename=str(py_file)) + except SyntaxError: + return findings + + for node in ast.walk(tree): + symbol: str | None = None + lineno = getattr(node, "lineno", 0) + if isinstance(node, ast.Import): + for alias in node.names: + base = alias.name.split(".", 1)[0] + if base in _FORBIDDEN_MODULES: + findings.append( + Finding(f"{py_file}:{node.lineno}", alias.name, + _line_allows(lines, node.lineno)) + ) + continue + if isinstance(node, ast.ImportFrom): + base = (node.module or "").split(".", 1)[0] + if base in _FORBIDDEN_MODULES: + symbol = f"from {node.module}" + elif isinstance(node, ast.Attribute) and isinstance(node.value, ast.Name): + mod = node.value.id + if (mod, node.attr) in _FORBIDDEN_ATTRS: + symbol = f"{mod}.{node.attr}" + if symbol is not None: + findings.append(Finding(f"{py_file}:{lineno}", symbol, + _line_allows(lines, lineno))) + return findings + + +def scan_tree(root: Path) -> list[Finding]: + """扫描整个包目录,返回全部命中(含已豁免项)。""" + findings: list[Finding] = [] + for py_file in sorted(root.rglob("*.py")): + if "__pycache__" in py_file.parts: + continue + findings.extend(scan_file(py_file)) + return findings + + +def violations(findings: list[Finding]) -> list[Finding]: + """未被豁免的命中即违规。""" + return [f for f in findings if not f.allowed] + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="随机 ID 禁令扫描(REQ-ID-01)") + parser.add_argument("--root", default=None, help="被扫描的包目录,默认 sqlgraph/") + parser.add_argument("--json", action="store_true", help="以 JSON 输出结果") + args = parser.parse_args(argv) + + root = Path(args.root) if args.root else Path(__file__).resolve().parent.parent / "sqlgraph" + findings = scan_tree(root) + bad = violations(findings) + + if args.json: + print(json.dumps([f.__dict__ for f in findings], ensure_ascii=False, indent=2)) + elif bad: + print(f"检出 {len(bad)} 处未豁免的随机性来源,违反 REQ-ID-01(确定性 ID 禁令):") + for f in bad: + print(f" - {f}") + else: + allowed = [f for f in findings if f.allowed] + note = f"({len(allowed)} 处经 allow-random 显式豁免)" if allowed else "" + print(f"随机 ID 扫描通过:地基未引入随机性来源{note}。") + + return 1 if bad else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/check_release_assets.py b/tools/check_release_assets.py new file mode 100755 index 0000000..0a39d13 --- /dev/null +++ b/tools/check_release_assets.py @@ -0,0 +1,44 @@ +#!/usr/bin/env python3 +# Copyright (c) 2026 ByteDance Ltd. and/or its affiliates +# SPDX-License-Identifier: MIT + +"""Reject generated or private artifacts from the Git index.""" + +from __future__ import annotations + +import subprocess +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +FORBIDDEN = ( + ".duckdb", + ".zip", + "comparison-screenshots/", + "generated_illustrations", + "demo_output/", + "dogfood-output/", +) + + +def main() -> int: + tracked = subprocess.check_output( + ["git", "ls-files"], + cwd=ROOT, + text=True, + ).splitlines() + bad = [ + path for path in tracked + if any(marker in path for marker in FORBIDDEN) + ] + if bad: + print("Forbidden release assets:") + for path in bad: + print(f" {path}") + return 1 + print("Release asset check passed.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/ci_checks.sh b/tools/ci_checks.sh new file mode 100755 index 0000000..6e2c847 --- /dev/null +++ b/tools/ci_checks.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash +set -euo pipefail + +PYTHON="${PYTHON:-python}" + +"$PYTHON" -m ruff check sqlgraph tests tools examples +"$PYTHON" tools/check_layering.py +"$PYTHON" tools/check_random_ids.py +"$PYTHON" tools/check_capabilities.py +"$PYTHON" scripts/opensource_guard.py +"$PYTHON" tools/check_release_assets.py +"$PYTHON" -m pytest --cov=sqlgraph --cov-report=json:coverage.json -q +"$PYTHON" tools/check_coverage.py --report coverage.json --min 85 +"$PYTHON" -m build