diff --git a/.agent/demos/CURATED_CATALOG.md b/.agent/demos/CURATED_CATALOG.md deleted file mode 100644 index 2e6f3eff87..0000000000 --- a/.agent/demos/CURATED_CATALOG.md +++ /dev/null @@ -1,54 +0,0 @@ -# Polylogue Demo Catalog - -Generated by `devtools workspace demo-shelf`. - -## _Packet Contract Stub - -- id: `_packet-contract-stub` -- files: 9 (6 readable) - -## Agent Affordance Usage - -- id: `agent-affordance-usage` -- readme: `agent-affordance-usage/README.md` -- files: 13 (13 readable) - -## Agent Forensics - -- id: `agent-forensics` -- readme: `agent-forensics/README.md` -- files: 9 (9 readable) - -## Anti Demo Multi Source Reconstruction - -- id: `anti-demo-multi-source-reconstruction` -- files: 9 (6 readable) - -## Attachment Acquisition Census - -- id: `attachment-acquisition-census` -- readme: `attachment-acquisition-census/README.md` -- analysis: `attachment-acquisition-census/ANALYSIS.md` -- files: 5 (4 readable) - -## Basic Usage — The Features Actually Work - -- id: `basic-usage` -- readme: `basic-usage/README.md` -- files: 11 (11 readable) - -## D1 Receipts - -- id: `d1-receipts` -- files: 9 (6 readable) - -## D4 Behavioral Archaeology - -- id: `d4-behavioral-archaeology` -- files: 9 (6 readable) - -## Handoff-Pack Uplift Experiment - -- id: `uplift-two-arm` -- readme: `uplift-two-arm/README.md` -- files: 20 (20 readable) diff --git a/.agent/demos/MANIFEST.readable.json b/.agent/demos/MANIFEST.readable.json deleted file mode 100644 index 387648fa34..0000000000 --- a/.agent/demos/MANIFEST.readable.json +++ /dev/null @@ -1,492 +0,0 @@ -{ - "contract": "current-curated-demo-set", - "root": ".agent/demos", - "input_closure": { - "mode": "git-tracked", - "included_count": 95, - "excluded_count": 0, - "exclusion_reason_counts": {}, - "exclusion_samples": "bounded samples are emitted in the command JSON payload" - }, - "packaging": "Use devtools workspace read-package for portable readable bundles; this helper only writes indexes.", - "curation_policy": "This is not append-only. Keep the best current demos here; replace, consolidate, or move stale demos out.", - "file_count": 95, - "readable_count": 82, - "files": [ - { - "path": "_packet-contract-stub/NON-CLAIMS.md", - "bytes": 177, - "readable": true - }, - { - "path": "_packet-contract-stub/PROMPT.md", - "bytes": 1108, - "readable": true - }, - { - "path": "_packet-contract-stub/checks.json", - "bytes": 151, - "readable": true - }, - { - "path": "_packet-contract-stub/evidence.ndjson", - "bytes": 258, - "readable": false - }, - { - "path": "_packet-contract-stub/finding.yaml", - "bytes": 266, - "readable": true - }, - { - "path": "_packet-contract-stub/packet.json", - "bytes": 4230, - "readable": true - }, - { - "path": "_packet-contract-stub/queries.ndjson", - "bytes": 74, - "readable": false - }, - { - "path": "_packet-contract-stub/report.md", - "bytes": 1076, - "readable": true - }, - { - "path": "_packet-contract-stub/run.log", - "bytes": 132, - "readable": false - }, - { - "path": "agent-affordance-usage/README.md", - "bytes": 2951, - "readable": true - }, - { - "path": "agent-affordance-usage/affordance-usage.report.json", - "bytes": 45710, - "readable": true - }, - { - "path": "agent-affordance-usage/archive-origin-counts.csv", - "bytes": 167, - "readable": true - }, - { - "path": "agent-affordance-usage/evidence-kind-counts.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-affordance-usage/family-counts.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-affordance-usage/recent-30d-tool-counts.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-affordance-usage/recent-7d-tool-counts.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-affordance-usage/summary.json", - "bytes": 2276, - "readable": true - }, - { - "path": "agent-affordance-usage/surface-classification-summary.csv", - "bytes": 134, - "readable": true - }, - { - "path": "agent-affordance-usage/surface-inventory.csv", - "bytes": 15387, - "readable": true - }, - { - "path": "agent-affordance-usage/tool-by-origin.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-affordance-usage/tool-counts.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-affordance-usage/tool-samples.csv", - "bytes": 1, - "readable": true - }, - { - "path": "agent-forensics/README.md", - "bytes": 4813, - "readable": true - }, - { - "path": "agent-forensics/current/archive-workload.json", - "bytes": 19828, - "readable": true - }, - { - "path": "agent-forensics/current/cost-rollups-timeout.txt", - "bytes": 637, - "readable": true - }, - { - "path": "agent-forensics/current/coverage-origin.json", - "bytes": 8675, - "readable": true - }, - { - "path": "agent-forensics/current/summary.json", - "bytes": 5451, - "readable": true - }, - { - "path": "agent-forensics/current/usage-headline-all.json", - "bytes": 68551, - "readable": true - }, - { - "path": "agent-forensics/current/usage-headline-claude-code.json", - "bytes": 31714, - "readable": true - }, - { - "path": "agent-forensics/current/usage-headline-codex.json", - "bytes": 16763, - "readable": true - }, - { - "path": "agent-forensics/current/usage-timeline-month-origin-model.json", - "bytes": 5540, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/NON-CLAIMS.md", - "bytes": 239, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/PROMPT.md", - "bytes": 1714, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/checks.json", - "bytes": 603, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/evidence.ndjson", - "bytes": 1035, - "readable": false - }, - { - "path": "anti-demo-multi-source-reconstruction/finding.yaml", - "bytes": 529, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/packet.json", - "bytes": 4907, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/queries.ndjson", - "bytes": 713, - "readable": false - }, - { - "path": "anti-demo-multi-source-reconstruction/report.md", - "bytes": 4188, - "readable": true - }, - { - "path": "anti-demo-multi-source-reconstruction/run.log", - "bytes": 1600, - "readable": false - }, - { - "path": "attachment-acquisition-census/ANALYSIS.md", - "bytes": 953, - "readable": true - }, - { - "path": "attachment-acquisition-census/README.md", - "bytes": 2691, - "readable": true - }, - { - "path": "attachment-acquisition-census/census.json", - "bytes": 1292, - "readable": true - }, - { - "path": "attachment-acquisition-census/reconcile-attachment-acquisition-debt.json", - "bytes": 356, - "readable": true - }, - { - "path": "attachment-acquisition-census/regenerate.sh", - "bytes": 8286, - "readable": false - }, - { - "path": "basic-usage/01-find-query.txt", - "bytes": 1204, - "readable": true - }, - { - "path": "basic-usage/02-read.txt", - "bytes": 723, - "readable": true - }, - { - "path": "basic-usage/03-search.txt", - "bytes": 664, - "readable": true - }, - { - "path": "basic-usage/04-resume.txt", - "bytes": 89, - "readable": true - }, - { - "path": "basic-usage/05-cost-usage.json", - "bytes": 53777, - "readable": true - }, - { - "path": "basic-usage/06-lineage.txt", - "bytes": 794, - "readable": true - }, - { - "path": "basic-usage/07-mcp-roundtrip.json", - "bytes": 2390, - "readable": true - }, - { - "path": "basic-usage/08-status-health.debt.json", - "bytes": 9292, - "readable": true - }, - { - "path": "basic-usage/08-status-health.txt", - "bytes": 1306, - "readable": true - }, - { - "path": "basic-usage/COLD_READER_GATE.md", - "bytes": 2945, - "readable": true - }, - { - "path": "basic-usage/README.md", - "bytes": 10073, - "readable": true - }, - { - "path": "d1-receipts/NON-CLAIMS.md", - "bytes": 550, - "readable": true - }, - { - "path": "d1-receipts/PROMPT.md", - "bytes": 3513, - "readable": true - }, - { - "path": "d1-receipts/checks.json", - "bytes": 1012, - "readable": true - }, - { - "path": "d1-receipts/evidence.ndjson", - "bytes": 3735, - "readable": false - }, - { - "path": "d1-receipts/finding.yaml", - "bytes": 651, - "readable": true - }, - { - "path": "d1-receipts/packet.json", - "bytes": 7452, - "readable": true - }, - { - "path": "d1-receipts/queries.ndjson", - "bytes": 1432, - "readable": false - }, - { - "path": "d1-receipts/report.md", - "bytes": 9186, - "readable": true - }, - { - "path": "d1-receipts/run.log", - "bytes": 5131, - "readable": false - }, - { - "path": "d4-behavioral-archaeology/NON-CLAIMS.md", - "bytes": 400, - "readable": true - }, - { - "path": "d4-behavioral-archaeology/PROMPT.md", - "bytes": 1807, - "readable": true - }, - { - "path": "d4-behavioral-archaeology/checks.json", - "bytes": 552, - "readable": true - }, - { - "path": "d4-behavioral-archaeology/evidence.ndjson", - "bytes": 2296, - "readable": false - }, - { - "path": "d4-behavioral-archaeology/finding.yaml", - "bytes": 664, - "readable": true - }, - { - "path": "d4-behavioral-archaeology/packet.json", - "bytes": 10443, - "readable": true - }, - { - "path": "d4-behavioral-archaeology/queries.ndjson", - "bytes": 1358, - "readable": false - }, - { - "path": "d4-behavioral-archaeology/report.md", - "bytes": 5033, - "readable": true - }, - { - "path": "d4-behavioral-archaeology/run.log", - "bytes": 9800, - "readable": false - }, - { - "path": "registry.json", - "bytes": 1268, - "readable": true - }, - { - "path": "uplift-two-arm/README.md", - "bytes": 864, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair1-handoff-pack-output.md", - "bytes": 7265, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair1-raw-ref-output.md", - "bytes": 8123, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair2-handoff-pack-output.md", - "bytes": 2489, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair2-raw-ref-output.md", - "bytes": 2406, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair3-handoff-pack-output.md", - "bytes": 2667, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair3-raw-ref-output.md", - "bytes": 2262, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair4-handoff-pack-output.md", - "bytes": 2236, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair4-raw-ref-output.md", - "bytes": 2158, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair5-handoff-pack-output.md", - "bytes": 2251, - "readable": true - }, - { - "path": "uplift-two-arm/current/arms/pair5-raw-ref-output.md", - "bytes": 1987, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/ground-truth-pair1.json", - "bytes": 1989, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/ground-truth-pair2.json", - "bytes": 874, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/ground-truth-pair3.json", - "bytes": 706, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/ground-truth-pair4.json", - "bytes": 614, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/ground-truth-pair5.json", - "bytes": 736, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/rubric.json", - "bytes": 912, - "readable": true - }, - { - "path": "uplift-two-arm/current/metrics/score.json", - "bytes": 2106, - "readable": true - }, - { - "path": "uplift-two-arm/current/pairs.json", - "bytes": 1849, - "readable": true - }, - { - "path": "uplift-two-arm/current/report.md", - "bytes": 7468, - "readable": true - } - ] -} diff --git a/.agent/demos/README.md b/.agent/demos/README.md deleted file mode 100644 index 706cf3ad23..0000000000 --- a/.agent/demos/README.md +++ /dev/null @@ -1,36 +0,0 @@ -# Polylogue Current Demo Shelf - -Generated by `devtools workspace demo-shelf`. - -This shelf contains the best current Polylogue demos for the active devloop. -It is not append-only. Replace, consolidate, or move stale demos out when a -better demo supersedes them. - -## Current Entries - -- `_packet-contract-stub` — _Packet Contract Stub - - files: 9 (6 readable) -- `agent-affordance-usage` — Agent Affordance Usage - - readme: `agent-affordance-usage/README.md` - - files: 13 (13 readable) -- `agent-forensics` — Agent Forensics - - readme: `agent-forensics/README.md` - - files: 9 (9 readable) -- `anti-demo-multi-source-reconstruction` — Anti Demo Multi Source Reconstruction - - files: 9 (6 readable) -- `attachment-acquisition-census` — Attachment Acquisition Census - - readme: `attachment-acquisition-census/README.md` - - analysis: `attachment-acquisition-census/ANALYSIS.md` - - files: 5 (4 readable) -- `basic-usage` — Basic Usage — The Features Actually Work - - readme: `basic-usage/README.md` - - files: 11 (11 readable) -- `d1-receipts` — D1 Receipts - - files: 9 (6 readable) -- `d4-behavioral-archaeology` — D4 Behavioral Archaeology - - files: 9 (6 readable) -- `uplift-two-arm` — Handoff-Pack Uplift Experiment - - readme: `uplift-two-arm/README.md` - - files: 20 (20 readable) - -Retired demo material belongs under `.agent/archive/retired-demos/`, not here. diff --git a/.agent/demos/SUMMARY_INDEX.json b/.agent/demos/SUMMARY_INDEX.json deleted file mode 100644 index 681a1d9eb8..0000000000 --- a/.agent/demos/SUMMARY_INDEX.json +++ /dev/null @@ -1,68 +0,0 @@ -{ - "root": ".agent/demos", - "summary_count": 2, - "coverage": { - "without_claim": [], - "without_non_claim": [], - "without_proof_fields": [], - "without_caveat_fields": [], - "unsummarized_demos": [ - "_packet-contract-stub", - "anti-demo-multi-source-reconstruction", - "attachment-acquisition-census", - "basic-usage", - "d1-receipts", - "d4-behavioral-archaeology", - "uplift-two-arm" - ] - }, - "records": [ - { - "demo": "agent-affordance-usage", - "summary_path": "agent-affordance-usage/summary.json", - "artifact": "agent-affordance-usage", - "claim": "Polylogue can compare agent affordance usage across normalized action evidence without summing unlike tool-name spellings or provider-specific call shapes.", - "non_claim": "This is not a human-quality utility evaluation of any particular tool family. It measures captured usage evidence, failure signals, and coverage gaps; usefulness still requires qualitative review of session context and outcomes.", - "proof_fields": [ - "proof_report" - ], - "caveat_fields": [ - "caveats" - ], - "archive_root": "/path/to/demo-archive", - "index_schema_version": 54, - "timestamp": "2026-08-02T13:51:38.483581+00:00", - "coverage": { - "claim": true, - "non_claim": true, - "proof_fields": true, - "caveat_fields": true - } - }, - { - "demo": "agent-forensics/current", - "summary_path": "agent-forensics/current/summary.json", - "artifact": "agent-forensics", - "claim": "Polylogue can regenerate a current v54 longitudinal agent-usage forensics packet over a deterministic seeded fixture archive using product analysis surfaces, with provenance-separated token/cost lanes and explicit physical-vs-logical token grains.", - "non_claim": "This packet is not provider billing truth, not an LLM judgment of failure follow-up behavior, not a resurrected standalone forensics script, and not evidence about the operator's real archive; it reads the deterministic demo archive through the same product commands used against a live archive, and points structured-failure follow-up to the current claim-vs-evidence packet.", - "proof_fields": [ - "command_proofs", - "proof_fields" - ], - "caveat_fields": [ - "caveat_fields", - "fixture_scale_caveat", - "usage_headline_caveats" - ], - "archive_root": "/path/to/demo-archive", - "index_schema_version": 54, - "timestamp": "2026-08-02T13:52:20Z", - "coverage": { - "claim": true, - "non_claim": true, - "proof_fields": true, - "caveat_fields": true - } - } - ] -} diff --git a/.agent/demos/_packet-contract-stub/NON-CLAIMS.md b/.agent/demos/_packet-contract-stub/NON-CLAIMS.md deleted file mode 100644 index 45df2054e7..0000000000 --- a/.agent/demos/_packet-contract-stub/NON-CLAIMS.md +++ /dev/null @@ -1,4 +0,0 @@ -# Non-claims - -- This fixture does not prove an archive, search, lineage, memory, or agent capability. -- This fixture does not establish field prevalence, scale, or performance. diff --git a/.agent/demos/_packet-contract-stub/PROMPT.md b/.agent/demos/_packet-contract-stub/PROMPT.md deleted file mode 100644 index f9c82ed65c..0000000000 --- a/.agent/demos/_packet-contract-stub/PROMPT.md +++ /dev/null @@ -1,22 +0,0 @@ -# Packet Contract Stub - -Predeclaration receipt: `artifact:packet-contract-stub-predeclaration`. - -This is a deliberately minimal, hand-authored fixture packet proving the -Demo Finding Packet contract (polylogue-212.7) end to end — it is NOT one of -the real 212 demos (D1/D2/D4/D5/D8/post-hoc-Q&A/anti-demo/foreman-rhetoric). -Those still need their own implementation (each is its own bead); this stub -exists only so `devtools lab policy demo-packet-registry` has one real, -conforming registry entry to validate, and so a future contributor building -a real demo has a concrete worked example of every required file. - -## What a real prompt would say here - -A real demo's PROMPT.md instructs a coding agent to run specific `polylogue` -commands (product primitives — the 212 compositionality rule: shell/python -is glue only) against the seeded demo corpus (`polylogue demo seed`, seed -1843), then package the results into this same packet shape. - -This stub's "claim" is trivial by design: count sessions in the seeded -corpus. Run: `polylogue --format json find` against the seeded corpus and -report the total. diff --git a/.agent/demos/_packet-contract-stub/checks.json b/.agent/demos/_packet-contract-stub/checks.json deleted file mode 100644 index 30123d30c7..0000000000 --- a/.agent/demos/_packet-contract-stub/checks.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "pass": true, - "unsupported_claims": [], - "coverage_notes": "fixture packet for the demo-packet-registry contract; not a real analytical demo" -} diff --git a/.agent/demos/_packet-contract-stub/evidence.ndjson b/.agent/demos/_packet-contract-stub/evidence.ndjson deleted file mode 100644 index 5ddaf9710d..0000000000 --- a/.agent/demos/_packet-contract-stub/evidence.ndjson +++ /dev/null @@ -1,2 +0,0 @@ -{"ref": "artifact:packet-contract-stub-evidence", "cited_for": "Demo Packet v2 receipt root", "verified_via": "committed evidence.ndjson"} -{"ref": "seeded-corpus-seed-1843:session-count", "value": "stub", "note": "fixture row proving evidence.ndjson shape"} diff --git a/.agent/demos/_packet-contract-stub/finding.yaml b/.agent/demos/_packet-contract-stub/finding.yaml deleted file mode 100644 index 3c0c17b131..0000000000 --- a/.agent/demos/_packet-contract-stub/finding.yaml +++ /dev/null @@ -1,6 +0,0 @@ -archive_cursor: seeded-corpus-seed-1843 -measure_version: demo-packet-v2 -commit_sha: ffba3ec0c -sample_frame_predicate: "all sessions in the seeded demo corpus (seed 1843)" -run_date: "2026-07-08" -claim: "the seeded demo corpus has a fixed, reproducible session count" diff --git a/.agent/demos/_packet-contract-stub/packet.json b/.agent/demos/_packet-contract-stub/packet.json deleted file mode 100644 index 56c340f11c..0000000000 --- a/.agent/demos/_packet-contract-stub/packet.json +++ /dev/null @@ -1,131 +0,0 @@ -{ - "baseline": { - "method": "Use the recorded simpler comparison path rather than the structural product path.", - "name": "legacy filename checklist", - "receipts": [ - "artifact:packet-contract-stub-evidence" - ], - "result": "Check only that the legacy packet filenames exist." - }, - "claim": { - "declared_before_execution": true, - "receipts": [ - "artifact:packet-contract-stub-predeclaration" - ], - "scope": "this validator fixture only", - "statement": "The complete committed fixture passes the Demo Packet v2 validation gate.", - "status": "supported" - }, - "controls": { - "missing_evidence": [ - { - "expected": "Removing the receipt artifact must make the packet invalid.", - "id": "missing-receipt", - "observed": { - "valid": false - }, - "passed": true, - "purpose": "Require missing evidence to remain explicit rather than converted into a positive claim.", - "receipts": [ - "artifact:packet-contract-stub-evidence" - ] - } - ], - "negative": [ - { - "expected": "Removing the falsifier must make the packet invalid.", - "id": "missing-falsifier", - "observed": { - "valid": false - }, - "passed": true, - "purpose": "Prevent an adjacent easier signal from being counted as the primary construct.", - "receipts": [ - "artifact:packet-contract-stub-evidence" - ] - } - ] - }, - "falsifier": { - "condition": "The v2 validator accepts the committed packet after its falsifier field is removed.", - "evaluation_method": "Apply the stated condition to the committed evidence and run log.", - "receipts": [ - "artifact:packet-contract-stub-evidence" - ], - "result": "pass", - "triggered": false - }, - "mode": "fixture", - "non_claims": [ - "This fixture does not prove an archive, search, lineage, memory, or agent capability.", - "This fixture does not establish field prevalence, scale, or performance." - ], - "oracle": { - "description": "The normative JSON Schema and packet validator independently enumerate the required contract fields.", - "expected": { - "valid": true - }, - "independent": true, - "method": "Validate the committed packet against docs/schemas/demo-packet-v2.schema.json.", - "receipts": [ - "artifact:packet-contract-stub-evidence" - ] - }, - "packet_id": "packet-contract-stub", - "primary_construct": { - "id": "demo.packet-v2.shape", - "product_primitives": [ - "demo packet registry validator" - ], - "statement": "The committed fixture demonstrates the complete Demo Packet v2 directory and epistemic shape." - }, - "provenance": { - "archive_cursor": "seeded-corpus-seed-1843", - "commit_sha": "ffba3ec0c", - "measure_version": "demo-packet-v2", - "run_date": "2026-07-10", - "sample_frame_predicate": "the committed packet-contract fixture" - }, - "receipts": [ - { - "artifact_path": "evidence.ndjson", - "description": "Committed evidence rows and references for this packet.", - "kind": "artifact", - "ref": "artifact:packet-contract-stub-evidence", - "resolved": true, - "sha256": "cdcfa78a44335e29068683c317cf83c09e85e0414b3271306138b58e404baa70" - }, - { - "artifact_path": "PROMPT.md", - "description": "The committed prompt that states the packet claim before execution.", - "kind": "artifact", - "ref": "artifact:packet-contract-stub-predeclaration", - "resolved": true, - "sha256": "4c0d59f421e04714276ce03b509e585abd14a3bc83cbb6b94597e64dbd118465" - } - ], - "reproduction": { - "commands": [ - "python -m devtools.verify_demo_packet_registry" - ], - "deterministic": true, - "fixture": "seeded-corpus contract fixture", - "private_data": false - }, - "results": { - "measurements": [ - { - "name": "schema_errors", - "receipts": [ - "artifact:packet-contract-stub-evidence" - ], - "unit": "errors", - "value": 0 - } - ], - "status": "pass", - "summary": "The fixture satisfies the complete Demo Packet v2 contract." - }, - "schema_version": "2.0.0", - "title": "Demo Packet contract fixture" -} diff --git a/.agent/demos/_packet-contract-stub/queries.ndjson b/.agent/demos/_packet-contract-stub/queries.ndjson deleted file mode 100644 index f088a22413..0000000000 --- a/.agent/demos/_packet-contract-stub/queries.ndjson +++ /dev/null @@ -1 +0,0 @@ -{"text": "find", "lowered_spec": {"unit": "sessions", "predicate": null}} diff --git a/.agent/demos/_packet-contract-stub/report.md b/.agent/demos/_packet-contract-stub/report.md deleted file mode 100644 index 19bb2d3e5b..0000000000 --- a/.agent/demos/_packet-contract-stub/report.md +++ /dev/null @@ -1,44 +0,0 @@ -# Packet Contract Stub - -## Claim - -The seeded demo corpus (seed 1843) has a fixed, reproducible session count. - -## Corpus - -`polylogue demo seed` output, seed 1843 — deterministic, no private data. - -## Method - -`polylogue --format json find` over the full seeded corpus; count result rows. - -## Findings - -The seeded corpus reproduces the same session count on every regeneration -(this is a fixture proving the packet contract's shape, not a real analytical -finding — see evidence.ndjson for the (stubbed) citation format). - -## Specimens - -See `evidence.ndjson` for the cited rows. - -## Counterexamples - -None — this is a trivial reproducibility check by construction. - -## Limits - -This packet is a fixture for `devtools lab policy demo-packet-registry`, not -a real 212 demo. Do not cite its "claim" externally. - -## Non-claims - -- This fixture does not prove an archive, search, lineage, memory, or agent capability. -- This fixture does not establish field prevalence, scale, or performance. - -## Reproduce - -```bash -polylogue demo seed --seed 1843 -polylogue --format json find -``` diff --git a/.agent/demos/_packet-contract-stub/run.log b/.agent/demos/_packet-contract-stub/run.log deleted file mode 100644 index 861088f7fd..0000000000 --- a/.agent/demos/_packet-contract-stub/run.log +++ /dev/null @@ -1,4 +0,0 @@ -$ polylogue demo seed --seed 1843 -seeded demo corpus at seed 1843 -$ polylogue --format json find -{"sessions": ["stub"], "count": 1} diff --git a/.agent/demos/agent-affordance-usage/README.md b/.agent/demos/agent-affordance-usage/README.md deleted file mode 100644 index fe3289ceae..0000000000 --- a/.agent/demos/agent-affordance-usage/README.md +++ /dev/null @@ -1,64 +0,0 @@ -# Agent Affordance Usage - -Generated: 2026-08-02T13:51:38.483581+00:00 -Archive root: `/path/to/demo-archive` -Index schema: v54 -Action scope: `grouped-tool-name-recent-window` - -## Top Families - - -## Recent Window (7 days) - - -## Surface Inventory Classification - -- cli_command keep: 64 surface(s), observed_actions=0. -- cli_command kill: 54 surface(s), observed_actions=0. -- mcp_tool keep: 4 surface(s), observed_actions=0. -- mcp_tool kill: 6 surface(s), observed_actions=0. - -## Kill Candidates - -These are zero-use non-operator surfaces in the captured archive evidence. They are review candidates, not automatic removals. - -- mcp_tool `context` — zero captured agent use in this archive window; review before removal -- mcp_tool `explain` — zero captured agent use in this archive window; review before removal -- mcp_tool `get` — zero captured agent use in this archive window; review before removal -- mcp_tool `query` — zero captured agent use in this archive window; review before removal -- mcp_tool `read` — zero captured agent use in this archive window; review before removal -- mcp_tool `status` — zero captured agent use in this archive window; review before removal -- cli_command `agent` — zero captured agent use in this archive window; review before removal -- cli_command `agent doctor` — zero captured agent use in this archive window; review before removal -- cli_command `agent install` — zero captured agent use in this archive window; review before removal -- cli_command `agent manifest` — zero captured agent use in this archive window; review before removal -- cli_command `agent manual` — zero captured agent use in this archive window; review before removal -- cli_command `agent status` — zero captured agent use in this archive window; review before removal - -## Interpretation - -- Family-normalized counts avoid treating plugin-prefixed tool names as separate affordances. -- The default action scope is the recent-session window; use --all-time for the intentionally broader scan. -- Recent windows are required for newly-added affordances such as Serena and codebase-memory. -- Failure rates are structured tool-result signals; they identify friction, not necessarily low utility. -- The surface inventory left-joins observed usage against every registered MCP tool and CLI command. -- Operator-only rows are kept even when unused; the classification caveat is part of the data. - -## Notes - -- Default family counts used an indexed grouped tool-name path over blocks. -- Command and input text bodies are not scanned unless --detail-pattern is supplied. -- Samples are omitted on this fast grouped path to avoid materializing every matching action row. - -## Files - -- `family-counts.csv` -- `evidence-kind-counts.csv` -- `tool-counts.csv` -- `tool-by-origin.csv` -- `recent-7d-tool-counts.csv` -- `tool-samples.csv` -- `surface-inventory.csv` -- `surface-classification-summary.csv` -- `affordance-usage.report.json` -- `summary.json` diff --git a/.agent/demos/agent-affordance-usage/affordance-usage.report.json b/.agent/demos/agent-affordance-usage/affordance-usage.report.json deleted file mode 100644 index 744f43f8a1..0000000000 --- a/.agent/demos/agent-affordance-usage/affordance-usage.report.json +++ /dev/null @@ -1,1384 +0,0 @@ -{ - "action_scope": "grouped-tool-name-recent-window", - "archive_root": "/path/to/demo-archive", - "captured_at": "2026-08-02T13:51:38.483581+00:00", - "command": "devtools workspace affordance-usage", - "detail_patterns": [], - "evidence_kind_counts": [], - "family_counts": [], - "index_db": "/path/to/demo-archive/index.db", - "index_schema_version": 54, - "notes": [ - "Default family counts used an indexed grouped tool-name path over blocks.", - "Command and input text bodies are not scanned unless --detail-pattern is supplied.", - "Samples are omitted on this fast grouped path to avoid materializing every matching action row." - ], - "origin_counts": [ - { - "origin": "codex-session", - "sessions": 7 - }, - { - "origin": "claude-code-session", - "sessions": 4 - }, - { - "origin": "chatgpt-export", - "sessions": 3 - }, - { - "origin": "aistudio-drive", - "sessions": 1 - }, - { - "origin": "antigravity-session", - "sessions": 1 - }, - { - "origin": "claude-ai-export", - "sessions": 1 - }, - { - "origin": "gemini-cli-session", - "sessions": 1 - }, - { - "origin": "hermes-session", - "sessions": 1 - } - ], - "patterns": [ - "serena", - "codebase", - "cclsp", - "context7", - "polylogue", - "lynchpin" - ], - "recent_cutoff_ms": 1785073898480, - "recent_tool_counts": [], - "recent_window_days": 7, - "report_version": 3, - "samples": [], - "summary": { - "interpretation": [ - "Family-normalized counts avoid treating plugin-prefixed tool names as separate affordances.", - "The default action scope is the recent-session window; use --all-time for the intentionally broader scan.", - "Recent windows are required for newly-added affordances such as Serena and codebase-memory.", - "Failure rates are structured tool-result signals; they identify friction, not necessarily low utility." - ], - "recent_top_families": [], - "recent_window_days": 7, - "top_families": [] - }, - "surface_inventory": [ - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "context", - "surface_type": "mcp_tool" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "explain", - "surface_type": "mcp_tool" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "get", - "surface_type": "mcp_tool" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "judge", - "surface_type": "mcp_tool" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "maintenance", - "surface_type": "mcp_tool" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "query", - "surface_type": "mcp_tool" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "read", - "surface_type": "mcp_tool" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "run", - "surface_type": "mcp_tool" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "status", - "surface_type": "mcp_tool" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "write", - "surface_type": "mcp_tool" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent doctor", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent install", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent manifest", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent manual", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent status", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agent uninstall", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agents", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agents status", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "agents work-item", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights cost-rollups", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights costs", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights coverage", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights debt", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights phases", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights profiles", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights tags", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights threads", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights timeline", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights tool-usage", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights usage-timeline", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze insights work-events", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze latency", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze pace", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze tools", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze turns", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "analyze usage", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "annotations", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "annotations import", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "annotations join", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "compare", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "config", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "config action-affordances", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "config completions", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "config paths", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "config query-completions", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "continue", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "dashboard", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "delete", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "demo", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "demo receipts", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "demo script", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "demo seed", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "demo tour", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "demo verify", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "facets", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "hooks", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "hooks install", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "hooks status", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "hooks uninstall", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "import", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "init", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "judge", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "manual", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "mark", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "note", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops auth", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops backup", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops debt", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops debt list", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops diagnostics", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops diagnostics codex-title-census", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops diagnostics space", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops diagnostics workload", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops doctor", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed backfill", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed disable", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed enable", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed preflight", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed resolve-failure", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops embed status", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops excise", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops insights", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops insights audit", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops insights export", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops insights hermes-health", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops insights status", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance archive-init", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance archive-plan", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance archive-read", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance assertion-export", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance attachment-acquisition-debt", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance backup-plan", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance blob-gc", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance blob-publications", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance blob-reference-debt", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance blob-reference-prune-orphans", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance blob-reference-recovery-plan", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance blob-reference-replace-from-source", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance embedding-orphan-reconcile", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance embeddings-rescue", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance gc-history", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance migrate-tier", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance plan", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance preview", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance raw-authority-blocker-resolve", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance raw-authority-blockers", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance raw-authority-census", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance raw-authority-detail", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance raw-authority-frontier", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance rebuild-index", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance run", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance status", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops maintenance verify-archive", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops materialize-incident-evidence", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops reconcile-work-effects", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops reset", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops scan-secrets", - "surface_type": "cli_command" - }, - { - "caveat": "operator-only surface; zero captured agent use is not removal evidence", - "classification": "keep", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": true, - "surface_name": "ops status", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "read", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "select", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "status", - "surface_type": "cli_command" - }, - { - "caveat": "zero captured agent use in this archive window; review before removal", - "classification": "kill", - "failure_rate": 0.0, - "observed_actions": 0, - "observed_sessions": 0, - "operator_only_caveat": false, - "surface_name": "tutorial", - "surface_type": "cli_command" - } - ], - "surface_inventory_summary": [ - { - "classification": "keep", - "observed_actions": 0, - "surface_type": "cli_command", - "surfaces": 64 - }, - { - "classification": "kill", - "observed_actions": 0, - "surface_type": "cli_command", - "surfaces": 54 - }, - { - "classification": "keep", - "observed_actions": 0, - "surface_type": "mcp_tool", - "surfaces": 4 - }, - { - "classification": "kill", - "observed_actions": 0, - "surface_type": "mcp_tool", - "surfaces": 6 - } - ], - "tool_by_origin": [], - "tool_counts": [] -} diff --git a/.agent/demos/agent-affordance-usage/archive-origin-counts.csv b/.agent/demos/agent-affordance-usage/archive-origin-counts.csv deleted file mode 100644 index 33b9066300..0000000000 --- a/.agent/demos/agent-affordance-usage/archive-origin-counts.csv +++ /dev/null @@ -1,9 +0,0 @@ -origin,sessions -codex-session,7 -claude-code-session,4 -chatgpt-export,3 -aistudio-drive,1 -antigravity-session,1 -claude-ai-export,1 -gemini-cli-session,1 -hermes-session,1 diff --git a/.agent/demos/agent-affordance-usage/evidence-kind-counts.csv b/.agent/demos/agent-affordance-usage/evidence-kind-counts.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/evidence-kind-counts.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-affordance-usage/family-counts.csv b/.agent/demos/agent-affordance-usage/family-counts.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/family-counts.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-affordance-usage/recent-30d-tool-counts.csv b/.agent/demos/agent-affordance-usage/recent-30d-tool-counts.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/recent-30d-tool-counts.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-affordance-usage/recent-7d-tool-counts.csv b/.agent/demos/agent-affordance-usage/recent-7d-tool-counts.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/recent-7d-tool-counts.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-affordance-usage/summary.json b/.agent/demos/agent-affordance-usage/summary.json deleted file mode 100644 index 0287b1c417..0000000000 --- a/.agent/demos/agent-affordance-usage/summary.json +++ /dev/null @@ -1,67 +0,0 @@ -{ - "archive_root": "/path/to/demo-archive", - "artifact": "agent-affordance-usage", - "caveats": [ - "Counts describe captured action evidence, not independent proof of user benefit.", - "Failure rates are provider-reported tool-result signals where available; missing outcome structure is not success.", - "Recent windows are adoption-sensitive and can legitimately differ from all-time counts.", - "Zero captured agent use is not enough to remove operator-only surfaces; those rows carry an operator-only caveat." - ], - "claim": "Polylogue can compare agent affordance usage across normalized action evidence without summing unlike tool-name spellings or provider-specific call shapes.", - "index_schema_version": 54, - "non_claim": "This is not a human-quality utility evaluation of any particular tool family. It measures captured usage evidence, failure signals, and coverage gaps; usefulness still requires qualitative review of session context and outcomes.", - "proof_report": { - "action_scope": "grouped-tool-name-recent-window", - "detail_patterns": [], - "patterns": [ - "serena", - "codebase", - "cclsp", - "context7", - "polylogue", - "lynchpin" - ], - "recent_top_families": [], - "recent_window_days": 7, - "report_version": 3, - "surface_inventory_summary": [ - { - "classification": "keep", - "observed_actions": 0, - "surface_type": "cli_command", - "surfaces": 64 - }, - { - "classification": "kill", - "observed_actions": 0, - "surface_type": "cli_command", - "surfaces": 54 - }, - { - "classification": "keep", - "observed_actions": 0, - "surface_type": "mcp_tool", - "surfaces": 4 - }, - { - "classification": "kill", - "observed_actions": 0, - "surface_type": "mcp_tool", - "surfaces": 6 - } - ], - "top_families": [] - }, - "source_files": [ - "affordance-usage.report.json", - "family-counts.csv", - "evidence-kind-counts.csv", - "tool-counts.csv", - "tool-by-origin.csv", - "recent-7d-tool-counts.csv", - "tool-samples.csv", - "surface-inventory.csv", - "surface-classification-summary.csv" - ], - "updated_at": "2026-08-02T13:51:38.483581+00:00" -} diff --git a/.agent/demos/agent-affordance-usage/surface-classification-summary.csv b/.agent/demos/agent-affordance-usage/surface-classification-summary.csv deleted file mode 100644 index cb72f8f3a3..0000000000 --- a/.agent/demos/agent-affordance-usage/surface-classification-summary.csv +++ /dev/null @@ -1,5 +0,0 @@ -surface_type,classification,surfaces,observed_actions -cli_command,keep,64,0 -cli_command,kill,54,0 -mcp_tool,keep,4,0 -mcp_tool,kill,6,0 diff --git a/.agent/demos/agent-affordance-usage/surface-inventory.csv b/.agent/demos/agent-affordance-usage/surface-inventory.csv deleted file mode 100644 index 7d349d4b68..0000000000 --- a/.agent/demos/agent-affordance-usage/surface-inventory.csv +++ /dev/null @@ -1,129 +0,0 @@ -surface_type,surface_name,observed_actions,observed_sessions,failure_rate,classification,operator_only_caveat,caveat -mcp_tool,context,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -mcp_tool,explain,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -mcp_tool,get,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -mcp_tool,judge,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -mcp_tool,maintenance,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -mcp_tool,query,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -mcp_tool,read,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -mcp_tool,run,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -mcp_tool,status,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -mcp_tool,write,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,agent,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agent doctor,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agent install,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agent manifest,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agent manual,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agent status,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agent uninstall,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agents,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agents status,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,agents work-item,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights cost-rollups,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights costs,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights coverage,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights debt,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights phases,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights profiles,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights tags,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights threads,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights timeline,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights tool-usage,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights usage-timeline,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze insights work-events,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze latency,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze pace,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze tools,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze turns,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,analyze usage,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,annotations,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,annotations import,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,annotations join,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,compare,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,config,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,config action-affordances,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,config completions,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,config paths,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,config query-completions,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,continue,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,dashboard,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,delete,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,demo,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,demo receipts,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,demo script,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,demo seed,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,demo tour,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,demo verify,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,facets,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,hooks,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,hooks install,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,hooks status,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,hooks uninstall,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,import,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,init,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,judge,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,manual,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,mark,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,note,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,ops,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops auth,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops backup,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops debt,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops debt list,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops diagnostics,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops diagnostics codex-title-census,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops diagnostics space,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops diagnostics workload,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops doctor,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed backfill,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed disable,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed enable,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed preflight,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed resolve-failure,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops embed status,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops excise,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops insights,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops insights audit,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops insights export,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops insights hermes-health,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops insights status,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance archive-init,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance archive-plan,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance archive-read,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance assertion-export,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance attachment-acquisition-debt,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance backup-plan,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance blob-gc,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance blob-publications,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance blob-reference-debt,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance blob-reference-prune-orphans,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance blob-reference-recovery-plan,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance blob-reference-replace-from-source,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance embedding-orphan-reconcile,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance embeddings-rescue,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance gc-history,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance migrate-tier,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance plan,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance preview,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance raw-authority-blocker-resolve,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance raw-authority-blockers,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance raw-authority-census,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance raw-authority-detail,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance raw-authority-frontier,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance rebuild-index,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance run,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance status,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops maintenance verify-archive,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops materialize-incident-evidence,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops reconcile-work-effects,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops reset,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops scan-secrets,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,ops status,0,0,0.0,keep,True,operator-only surface; zero captured agent use is not removal evidence -cli_command,read,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,select,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,status,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal -cli_command,tutorial,0,0,0.0,kill,False,zero captured agent use in this archive window; review before removal diff --git a/.agent/demos/agent-affordance-usage/tool-by-origin.csv b/.agent/demos/agent-affordance-usage/tool-by-origin.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/tool-by-origin.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-affordance-usage/tool-counts.csv b/.agent/demos/agent-affordance-usage/tool-counts.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/tool-counts.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-affordance-usage/tool-samples.csv b/.agent/demos/agent-affordance-usage/tool-samples.csv deleted file mode 100644 index 8b13789179..0000000000 --- a/.agent/demos/agent-affordance-usage/tool-samples.csv +++ /dev/null @@ -1 +0,0 @@ - diff --git a/.agent/demos/agent-forensics/README.md b/.agent/demos/agent-forensics/README.md deleted file mode 100644 index a8ca822f1d..0000000000 --- a/.agent/demos/agent-forensics/README.md +++ /dev/null @@ -1,111 +0,0 @@ -# Agent Forensics - -A longitudinal agent-usage forensics packet, over the deterministic seeded -demo archive (`polylogue demo seed`). - -This packet replaces the retired schema-v23 full-report packet for current -cardinality and headline token/cost claims. It uses product analysis -surfaces, not the deleted standalone `scripts/agent_forensics.py` path. - -**This packet is private-data-free.** A prior version of this packet was -generated against a live operator archive and committed real corpus size, -token totals, per-model spend in USD, and the operator's archive path to -this public repo (polylogue-0bgr). It has been regenerated from scratch -against the deterministic seeded fixture archive; every number below -describes the ~19-session synthetic fixture, not any operator's real usage. - -## What This Proves - -Polylogue can regenerate an agent-usage forensics packet from an archive -using normal analysis commands. The result keeps distinct evidence claims -separate: - -- physical-session archive totals; -- logical-session high-water totals; -- priced vs origin-reported cost lanes; -- current origin coverage; -- month/origin/model usage timeline rows. - -## Current Headline (seeded fixture) - -Generated: 2026-08-02T13:52:20Z -Archive root: `/path/to/demo-archive` -Index schema: v54 - -- physical sessions: 19 -- messages: 71 -- blocks: 121 -- materialized session profiles: 19 -- origin coverage rows: 8 -- usage timeline rows: 5 -- physical-session tokens accounted: 388,872 -- logical-session high-water tokens accounted: 388,872 -- replay-chain gap: 0 -- Claude Code physical/logical tokens: 388,500 / 388,500 -- Codex physical/logical tokens: 0 / 0 (no priced Codex sessions in the fixture) -- stored provider-priced cost: $2.84 -- catalog API-equivalent cost: $2.84 -- logical catalog API-equivalent cost: $2.84 - -## Caveats - -This is not provider billing truth. It does not query provider accounts. Cost -figures are archive/provider-reported lanes plus catalog API-equivalent pricing -where the pricing catalog matches the model. - -This is not a resurrection of the old standalone forensics script. The old -script was intentionally folded into product analysis surfaces. This packet is -the current demo/finding layer over those surfaces. - -This is not evidence about any operator's real usage or spend. The headline -numbers above describe the deterministic seeded fixture only. - -The cost-rollups drilldown command completes immediately on this fixture (see -`current/cost-rollups-timeout.txt`). The prior live-archive version of this -packet recorded a genuine 120-second timeout on that command at live-archive -scale; that is a real product-performance finding, tracked as a separate -follow-up, and does not reproduce on a fixture this small. - -Structured failure follow-up behavior is covered by the current -`claim-vs-evidence` packet. This packet links to that current demo rather than -duplicating bounded failure-follow-up samples here. - -## Regenerate - -```bash -polylogue demo seed --root ./demo-archive --force --with-overlays -export POLYLOGUE_ARCHIVE_ROOT="$PWD/demo-archive" - -polylogue --plain ops diagnostics workload --json \ - > .agent/demos/agent-forensics/current/archive-workload.json - -polylogue --plain analyze usage --detail headline --format json --limit 0 \ - > .agent/demos/agent-forensics/current/usage-headline-all.json - -polylogue --plain analyze usage --detail headline --origin claude-code-session --format json --limit 0 \ - > .agent/demos/agent-forensics/current/usage-headline-claude-code.json - -polylogue --plain analyze usage --detail headline --origin codex-session --format json --limit 0 \ - > .agent/demos/agent-forensics/current/usage-headline-codex.json - -polylogue --plain analyze insights coverage --group-by origin --format json --limit 1000 \ - > .agent/demos/agent-forensics/current/coverage-origin.json - -polylogue --plain analyze insights usage-timeline --group-by month-origin-model --format json --limit 500 \ - > .agent/demos/agent-forensics/current/usage-timeline-month-origin-model.json - -devtools workspace demo-shelf -``` - -Never regenerate this packet with `POLYLOGUE_ARCHIVE_ROOT` pointed at a live -operator archive -- this is a committed public-repo artifact, and doing so is -exactly the mistake polylogue-0bgr fixed. - -## Files - -- `current/summary.json` — claim/non-claim, headline numbers, caveats, command proofs. -- `current/archive-workload.json` — fixture-archive tier/cardinality snapshot. -- `current/usage-headline-*.json` — all-provider and provider-specific usage lanes. -- `current/coverage-origin.json` — origin coverage table. -- `current/usage-timeline-month-origin-model.json` — month/origin/model timeline. -- `current/cost-rollups-timeout.txt` — cost-rollups drilldown proof (completes on the fixture; documents the live-archive timeout finding it originally captured). diff --git a/.agent/demos/agent-forensics/current/archive-workload.json b/.agent/demos/agent-forensics/current/archive-workload.json deleted file mode 100644 index 04356fa0eb..0000000000 --- a/.agent/demos/agent-forensics/current/archive-workload.json +++ /dev/null @@ -1,687 +0,0 @@ -{ - "archive_tiers": { - "complete": true, - "derived_readiness": { - "checked": true, - "counts": { - "action_count": 1, - "action_count_exact": false, - "block_count": 121, - "cost_profile_count": 4, - "lost_source_evidence_count": 0, - "lost_source_evidence_samples": [], - "message_count": 71, - "messages_fts_count": -2, - "messages_fts_exact_counts": false, - "missing_profile_row_count": 0, - "missing_raw_session_count": 0, - "missing_raw_session_samples": [], - "orphan_profile_row_count": 0, - "phase_row_count": -2, - "profile_phase_count_mismatch": 0, - "profile_row_count": 19, - "profile_work_event_count_mismatch": 0, - "raw_link_count": 19, - "raw_materialization_debt_count": 0, - "raw_materialization_debt_group_count": 0, - "session_count": 19, - "session_tag_count": 2, - "text_block_count": 121, - "thread_count": 16, - "thread_session_count": 19, - "work_event_row_count": -2 - }, - "materialization_counts": { - "context_snapshots": 19, - "latency": 19, - "observed_events": 19, - "phases": 19, - "provider_usage": 19, - "runs": 19, - "session_profile": 19, - "thread": 19, - "work_events": 19 - }, - "missing_materialization_counts": { - "latency": 0, - "phases": 0, - "session_profile": 0, - "thread": 0, - "work_events": 0 - }, - "ready": { - "latency_materialization_ready": true, - "messages_fts_ready": true, - "phase_materialization_ready": true, - "profile_counts_ready": true, - "profile_materialization_ready": true, - "profile_rows_ready": true, - "raw_links_ready": true, - "raw_materialization_ready": true, - "thread_materialization_ready": true, - "work_event_materialization_ready": true - }, - "reason": null, - "source_check_available": true, - "surface_readiness": { - "archive_sessions": { - "blockers": [], - "evidence": { - "message_count": 71, - "session_count": 19, - "text_block_count": 121 - }, - "ready": true - }, - "latency_profiles": { - "blockers": [], - "evidence": { - "missing_materialization_count": 0 - }, - "ready": true - }, - "raw_artifacts": { - "blockers": [], - "evidence": { - "lost_source_evidence_count": 0, - "lost_source_evidence_samples": [], - "missing_raw_session_count": 0, - "missing_raw_session_samples": [], - "raw_link_count": 19, - "raw_materialization_debt_count": 0, - "raw_materialization_debt_group_count": 0, - "source_check_available": true - }, - "ready": true - }, - "search": { - "blockers": [], - "evidence": { - "messages_fts_count": -2, - "messages_fts_exact_counts": false, - "text_block_count": 121 - }, - "ready": true - }, - "session_costs": { - "blockers": [], - "evidence": { - "cost_profile_count": 4, - "missing_materialization_count": 0, - "missing_profile_row_count": 0 - }, - "ready": true - }, - "session_profiles": { - "blockers": [], - "evidence": { - "missing_materialization_count": 0, - "missing_profile_row_count": 0, - "orphan_profile_row_count": 0, - "profile_row_count": 19 - }, - "ready": true - }, - "tag_rollups": { - "blockers": [], - "evidence": { - "session_tag_count": 2 - }, - "ready": true - }, - "threads": { - "blockers": [], - "evidence": { - "missing_materialization_count": 0, - "thread_count": 16, - "thread_session_count": 19 - }, - "ready": true - }, - "timeline_phases": { - "blockers": [], - "evidence": { - "missing_materialization_count": 0, - "phase_row_count": -2 - }, - "ready": true - }, - "timeline_work_events": { - "blockers": [], - "evidence": { - "missing_materialization_count": 0, - "work_event_row_count": -2 - }, - "ready": true - }, - "tool_usage": { - "blockers": [], - "evidence": { - "action_count": 1, - "action_count_exact": false - }, - "ready": true - } - } - }, - "expected_count": 5, - "layout_readiness": { - "archive_ready": true, - "blockers": [], - "evidence": { - "blocked_surface_count": 0, - "complete": true, - "derived_readiness_checked": true, - "derived_surface_count": 11, - "expected_count": 5, - "missing_backup_required_count": 0, - "present_count": 5, - "schema_mismatch_count": 0, - "user_overlay_checked": true, - "user_overlay_orphan_session_references": 0 - }, - "state": "archive_ready" - }, - "missing_backup_required": [], - "observed_tier": "index", - "present": true, - "present_count": 5, - "root": "/path/to/demo-archive", - "schema_mismatches": [], - "table_count_mode": "mixed", - "tiers": { - "embeddings": { - "backup_required": true, - "durability": "expensive_rebuild", - "error": null, - "exists": true, - "expected_user_version": 4, - "filename": "embeddings.db", - "integrity": "not_checked", - "observed_by_probe": false, - "path": "/path/to/demo-archive/embeddings.db", - "size_bytes": 4317184, - "table_count_precision": { - "embedding_status": "exact", - "message_embeddings": "unavailable", - "message_embeddings_meta": "unavailable" - }, - "table_counts": { - "embedding_status": 1, - "message_embeddings": -2, - "message_embeddings_meta": -2 - }, - "tier": "embeddings", - "user_version": 4 - }, - "index": { - "backup_required": false, - "durability": "rebuildable", - "error": null, - "exists": true, - "expected_user_version": 54, - "filename": "index.db", - "integrity": "not_checked", - "observed_by_probe": true, - "path": "/path/to/demo-archive/index.db", - "size_bytes": 1449984, - "table_count_precision": { - "attachment_refs": "estimate", - "attachments": "estimate", - "blocks": "estimate", - "insight_materialization": "exact", - "messages": "exact", - "messages_fts_docsize": "unavailable", - "paste_spans": "unavailable", - "repos": "estimate", - "session_commits": "unavailable", - "session_events": "estimate", - "session_links": "estimate", - "session_phases": "unavailable", - "session_profiles": "exact", - "session_repos": "estimate", - "session_tags": "estimate", - "session_work_events": "unavailable", - "session_working_dirs": "estimate", - "sessions": "exact", - "thread_sessions": "estimate", - "threads": "estimate" - }, - "table_counts": { - "attachment_refs": 1, - "attachments": 1, - "blocks": 121, - "insight_materialization": 171, - "messages": 71, - "messages_fts_docsize": -2, - "paste_spans": -2, - "repos": 5, - "session_commits": -2, - "session_events": 4, - "session_links": 3, - "session_phases": -2, - "session_profiles": 19, - "session_repos": 5, - "session_tags": 2, - "session_work_events": -2, - "session_working_dirs": 5, - "sessions": 19, - "thread_sessions": 19, - "threads": 16 - }, - "tier": "index", - "user_version": 54 - }, - "ops": { - "backup_required": false, - "durability": "disposable", - "error": null, - "exists": true, - "expected_user_version": 1, - "filename": "ops.db", - "integrity": "not_checked", - "observed_by_probe": false, - "path": "/path/to/demo-archive/ops.db", - "size_bytes": 266240, - "table_count_precision": { - "convergence_debt": "exact", - "cursor_lag_samples": "exact", - "daemon_events": "exact", - "daemon_stage_events": "exact", - "embedding_catchup_runs": "unavailable", - "ingest_attempts": "exact", - "ingest_cursor": "unavailable", - "otlp_spans": "unavailable", - "otlp_telemetry": "unavailable" - }, - "table_counts": { - "convergence_debt": 0, - "cursor_lag_samples": 0, - "daemon_events": 0, - "daemon_stage_events": 0, - "embedding_catchup_runs": -2, - "ingest_attempts": 0, - "ingest_cursor": -2, - "otlp_spans": -2, - "otlp_telemetry": -2 - }, - "tier": "ops", - "user_version": 1 - }, - "source": { - "backup_required": true, - "durability": "irreplaceable", - "error": null, - "exists": true, - "expected_user_version": 16, - "filename": "source.db", - "integrity": "not_checked", - "observed_by_probe": false, - "path": "/path/to/demo-archive/source.db", - "size_bytes": 344064, - "table_count_precision": { - "blob_refs": "estimate", - "history_sidecars": "unavailable", - "raw_artifacts": "unavailable", - "raw_hook_events": "unavailable", - "raw_sessions": "exact" - }, - "table_counts": { - "blob_refs": 22, - "history_sidecars": -2, - "raw_artifacts": -2, - "raw_hook_events": -2, - "raw_sessions": 21 - }, - "tier": "source", - "user_version": 16 - }, - "user": { - "backup_required": true, - "durability": "human", - "error": null, - "exists": true, - "expected_user_version": 10, - "filename": "user.db", - "integrity": "not_checked", - "observed_by_probe": false, - "path": "/path/to/demo-archive/user.db", - "size_bytes": 253952, - "table_count_precision": { - "assertions": "unavailable", - "session_metadata": "missing", - "session_tags": "missing" - }, - "table_counts": { - "assertions": -2, - "session_metadata": -1, - "session_tags": -1 - }, - "tier": "user", - "user_version": 10 - } - }, - "user_overlay_orphans": { - "checked": true, - "orphan_session_reference_counts": { - "assertion_annotations": 0, - "assertion_corrections": 0, - "assertion_marks": 0, - "assertion_metadata": 0, - "assertion_notes": 0, - "assertion_suppressions": 0, - "assertion_tags": 0 - }, - "reason": null, - "total_orphan_session_references": 0 - } - }, - "attempt_counts": { - "completed": 0, - "failed": 0, - "overlapping_source_paths": [], - "running": 0, - "stale_cursor_writes": 0, - "total": 0 - }, - "automatic_convergence_backlog": { - "checked": true, - "counts": { - "automatic_backlog_total": 0, - "missing_latency_materialization": 0, - "missing_phases_materialization": 0, - "missing_profile_rows": 0, - "missing_session_profile_materialization": 0, - "missing_threads_materialization": 0, - "missing_work_events_materialization": 0, - "retry_debt_unresolved": 0 - }, - "reason": null, - "state": "ready" - }, - "blob_reference_debt": { - "checked": false, - "ok": true, - "reason": "skipped; pass --blob-reference-debt to scan referenced blob files" - }, - "boundary_table_count_mode": "mixed", - "boundary_table_count_precision": { - "blocks": "estimate", - "convergence_debt": "exact", - "cursor_lag_samples": "exact", - "daemon_events": "exact", - "daemon_stage_events": "exact", - "ingest_attempts": "exact", - "live_ingest_attempt": "missing", - "message_embeddings": "missing", - "messages": "exact", - "messages_fts_docsize": "unavailable", - "otlp_telemetry": "unavailable", - "raw_artifacts": "unavailable", - "raw_sessions": "exact", - "repos": "estimate", - "session_commits": "unavailable", - "session_events": "estimate", - "session_links": "estimate", - "session_profiles": "exact", - "session_repos": "estimate", - "sessions": "exact" - }, - "boundary_table_counts": { - "blocks": 121, - "convergence_debt": 0, - "cursor_lag_samples": 0, - "daemon_events": 0, - "daemon_stage_events": 0, - "ingest_attempts": 0, - "live_ingest_attempt": -1, - "message_embeddings": -1, - "messages": 71, - "messages_fts_docsize": -2, - "otlp_telemetry": -2, - "raw_artifacts": -2, - "raw_sessions": 21, - "repos": 5, - "session_commits": -2, - "session_events": 4, - "session_links": 3, - "session_profiles": 19, - "session_repos": 5, - "sessions": 19 - }, - "captured_at": "2026-08-02T13:51:53+00:00", - "convergence_debt": { - "by_stage": [], - "deferred_count": 0, - "failed_count": 0, - "unresolved_count": 0 - }, - "convergence_stage_timings": { - "convergence_time_s": { - "max": 0.0, - "mean": 0.0, - "min": 0.0, - "sum": 0.0 - }, - "parse_time_s": { - "max": 0.0, - "mean": 0.0, - "min": 0.0, - "sum": 0.0 - }, - "per_stage_s": {}, - "read_amplification": { - "max": 0.0, - "mean": 0.0, - "min": 0.0, - "sum": 0.0 - }, - "sample_size": 0 - }, - "cursor_lag_baselines": { - "families": [], - "family_count": 0, - "table_present": true, - "total_sample_count": 0 - }, - "daemon_resource_signal": { - "available": false - }, - "db_path": "/path/to/demo-archive/index.db", - "fts_trigger_state": { - "all_present": true, - "expected": [ - "messages_fts_ai", - "messages_fts_ad", - "messages_fts_au" - ], - "missing": [], - "present": [ - "messages_fts_ad", - "messages_fts_ai", - "messages_fts_au" - ] - }, - "gc_state": { - "generation_count": 0, - "high_water_generation": 0, - "last_completed_at": null, - "table_present": true - }, - "observability_locations": { - "archive_root": "/path/to/demo-archive", - "logical_tables": { - "blocks": { - "db_path": "/path/to/demo-archive/index.db", - "exists": true, - "logical_table": "blocks", - "physical_table": "blocks", - "tier": "index" - }, - "convergence_debt": { - "db_path": "/path/to/demo-archive/ops.db", - "exists": true, - "logical_table": "convergence_debt", - "physical_table": "convergence_debt", - "tier": "ops" - }, - "live_ingest_attempt": { - "db_path": "/path/to/demo-archive/ops.db", - "exists": true, - "logical_table": "live_ingest_attempt", - "physical_table": "ingest_attempts", - "tier": "ops" - }, - "live_ingest_attempt_stage_events": { - "db_path": "/path/to/demo-archive/ops.db", - "exists": true, - "logical_table": "live_ingest_attempt_stage_events", - "physical_table": "daemon_stage_events", - "tier": "ops" - }, - "messages": { - "db_path": "/path/to/demo-archive/index.db", - "exists": true, - "logical_table": "messages", - "physical_table": "messages", - "tier": "index" - }, - "messages_fts": { - "db_path": "/path/to/demo-archive/index.db", - "exists": true, - "logical_table": "messages_fts", - "physical_table": "messages_fts", - "tier": "index" - }, - "raw_sessions": { - "db_path": "/path/to/demo-archive/source.db", - "exists": true, - "logical_table": "raw_sessions", - "physical_table": "raw_sessions", - "tier": "source" - }, - "sessions": { - "db_path": "/path/to/demo-archive/index.db", - "exists": true, - "logical_table": "sessions", - "physical_table": "sessions", - "tier": "index" - } - }, - "tiers": { - "index": "/path/to/demo-archive/index.db", - "ops": "/path/to/demo-archive/ops.db", - "source": "/path/to/demo-archive/source.db" - } - }, - "ok": true, - "query_plans": { - "message_fts_gap_probe": { - "hazards": [], - "plan": [ - "SEARCH b USING INDEX idx_blocks_session_position (session_id=?)", - "SEARCH d USING INTEGER PRIMARY KEY (rowid=?) LEFT-JOIN" - ] - }, - "source_path_lookup": { - "hazards": [], - "plan": [ - "SEARCH r USING INDEX idx_raw_sessions_source_path (source_path=?)", - "SEARCH s USING INDEX idx_sessions_raw_id (raw_id=?)", - "USE TEMP B-TREE FOR DISTINCT" - ], - "storage_route": "archive_file_set" - } - }, - "raw_replay_backlog": { - "adoption_deferred_count": 0, - "already_parsed_count": 0, - "authority_component_count": 0, - "authority_quarantined_count": 0, - "available": true, - "blocked_authority_component_count": 0, - "blocked_candidate_count": 0, - "byte_authority_fragment_count": 0, - "byte_authority_pending_count": 0, - "byte_authority_quarantined_count": 0, - "candidate_count": 0, - "durable_authority_debt_count": 0, - "executable_authority_component_count": 0, - "execute_blob_limit_bytes": 1073741824, - "execution_block_reason": null, - "execution_blocked": false, - "expanded_aggregate_blocked": false, - "expanded_candidate_count": 0, - "expanded_total_blob_bytes": 0, - "max_blob_bytes": 0, - "missing_blob_count": 0, - "missing_blob_source_available_count": 0, - "missing_blob_source_missing_count": 0, - "origin_summary": [], - "oversized_count": 0, - "oversized_stream_safe_count": 0, - "source_path_summary": [], - "top_raw_rows": [], - "total_blob_bytes": 0 - }, - "recent_attempts": [], - "report_version": 18, - "source_path_churn": [], - "sqlite_maintenance": { - "observed_db": "/path/to/demo-archive/index.db", - "tiers": { - "embeddings": { - "exists": true, - "path": "/path/to/demo-archive/embeddings.db", - "planner_stats_present": true, - "sqlite_stat1_rows": 1, - "wal_bytes": 0 - }, - "index": { - "exists": true, - "path": "/path/to/demo-archive/index.db", - "planner_stats_present": true, - "sqlite_stat1_rows": 84, - "wal_bytes": 0 - }, - "ops": { - "exists": true, - "path": "/path/to/demo-archive/ops.db", - "planner_stats_present": false, - "sqlite_stat1_rows": 0, - "wal_bytes": 0 - }, - "source": { - "exists": true, - "path": "/path/to/demo-archive/source.db", - "planner_stats_present": true, - "sqlite_stat1_rows": 14, - "wal_bytes": 0 - }, - "user": { - "exists": true, - "path": "/path/to/demo-archive/user.db", - "planner_stats_present": true, - "sqlite_stat1_rows": 1, - "wal_bytes": 0 - } - }, - "tiers_with_planner_stats": 4, - "total_wal_bytes": 0 - }, - "storage_route_counts": { - "archive_append": 0, - "archive_file_set": 0, - "archive_full": 0, - "other": 0, - "unknown": 0, - "unsupported_polylogue_batch": 0 - }, - "topology_quarantine_state": { - "oldest_quarantined_at": null, - "quarantined_count": 0, - "resolved_count": 3, - "table_present": true, - "unresolved_count": 0 - } -} diff --git a/.agent/demos/agent-forensics/current/cost-rollups-timeout.txt b/.agent/demos/agent-forensics/current/cost-rollups-timeout.txt deleted file mode 100644 index 495dc65473..0000000000 --- a/.agent/demos/agent-forensics/current/cost-rollups-timeout.txt +++ /dev/null @@ -1,4 +0,0 @@ -exit_code=0 -elapsed_s=n/a (completes well under a second on the 19-session seeded fixture) -command=POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze insights cost-rollups --format json --limit 100 -note=The original version of this packet was generated against a live operator archive and this command timed out there (exit_code=124, elapsed_s=120.03) -- a genuine product-performance finding at live-archive scale, tracked separately. That timeout does not reproduce on this tiny synthetic fixture, so this file no longer documents a timeout; it documents that the command completes and returns 10 cost-rollup rows. diff --git a/.agent/demos/agent-forensics/current/coverage-origin.json b/.agent/demos/agent-forensics/current/coverage-origin.json deleted file mode 100644 index 1b2a2b496b..0000000000 --- a/.agent/demos/agent-forensics/current/coverage-origin.json +++ /dev/null @@ -1,264 +0,0 @@ -{ - "status": "ok", - "result": { - "total": 8, - "archive_coverage": [ - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "codex-session", - "origin": "codex-session", - "session_count": 7, - "logical_session_count": 0, - "message_count": 30, - "user_message_count": 9, - "authored_user_message_count": 9, - "assistant_message_count": 17, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 4.285714285714286, - "avg_user_words": 7.333333333333333, - "avg_authored_user_words": 7.333333333333333, - "avg_assistant_words": 7.588235294117647, - "tool_use_count": 9, - "thinking_count": 0, - "total_sessions_with_tools": 4, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 57.14285714285714, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "claude-code-session", - "origin": "claude-code-session", - "session_count": 4, - "logical_session_count": 0, - "message_count": 19, - "user_message_count": 3, - "authored_user_message_count": 2, - "assistant_message_count": 10, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 4.75, - "avg_user_words": 12.0, - "avg_authored_user_words": 10.5, - "avg_assistant_words": 28.0, - "tool_use_count": 7, - "thinking_count": 0, - "total_sessions_with_tools": 2, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 50.0, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "chatgpt-export", - "origin": "chatgpt-export", - "session_count": 3, - "logical_session_count": 0, - "message_count": 7, - "user_message_count": 4, - "authored_user_message_count": 4, - "assistant_message_count": 3, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 2.3333333333333335, - "avg_user_words": 44.5, - "avg_authored_user_words": 44.5, - "avg_assistant_words": 24.333333333333332, - "tool_use_count": 0, - "thinking_count": 0, - "total_sessions_with_tools": 0, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 0.0, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "aistudio-drive", - "origin": "aistudio-drive", - "session_count": 1, - "logical_session_count": 0, - "message_count": 4, - "user_message_count": 2, - "authored_user_message_count": 2, - "assistant_message_count": 2, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 4.0, - "avg_user_words": 7.0, - "avg_authored_user_words": 7.0, - "avg_assistant_words": 5.0, - "tool_use_count": 0, - "thinking_count": 0, - "total_sessions_with_tools": 0, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 0.0, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "antigravity-session", - "origin": "antigravity-session", - "session_count": 1, - "logical_session_count": 0, - "message_count": 2, - "user_message_count": 1, - "authored_user_message_count": 0, - "assistant_message_count": 1, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 2.0, - "avg_user_words": 11.0, - "avg_authored_user_words": null, - "avg_assistant_words": 18.0, - "tool_use_count": 0, - "thinking_count": 0, - "total_sessions_with_tools": 0, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 0.0, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "claude-ai-export", - "origin": "claude-ai-export", - "session_count": 1, - "logical_session_count": 0, - "message_count": 2, - "user_message_count": 1, - "authored_user_message_count": 1, - "assistant_message_count": 1, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 2.0, - "avg_user_words": 7.0, - "avg_authored_user_words": 7.0, - "avg_assistant_words": 10.0, - "tool_use_count": 0, - "thinking_count": 0, - "total_sessions_with_tools": 0, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 0.0, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "gemini-cli-session", - "origin": "gemini-cli-session", - "session_count": 1, - "logical_session_count": 0, - "message_count": 2, - "user_message_count": 1, - "authored_user_message_count": 0, - "assistant_message_count": 1, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 2.0, - "avg_user_words": 15.0, - "avg_authored_user_words": null, - "avg_assistant_words": 18.0, - "tool_use_count": 1, - "thinking_count": 1, - "total_sessions_with_tools": 1, - "total_sessions_with_thinking": 1, - "tool_use_percentage": 100.0, - "thinking_percentage": 100.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - }, - { - "contract_version": 10, - "insight_kind": "archive_coverage", - "group_by": "origin", - "bucket": "hermes-session", - "origin": "hermes-session", - "session_count": 1, - "logical_session_count": 0, - "message_count": 5, - "user_message_count": 1, - "authored_user_message_count": 0, - "assistant_message_count": 2, - "total_cost_usd": 0.0, - "total_duration_ms": 0, - "total_tool_active_duration_ms": 0, - "total_wall_duration_ms": 0, - "total_words": 0, - "avg_messages_per_session": 5.0, - "avg_user_words": 10.0, - "avg_authored_user_words": null, - "avg_assistant_words": 8.0, - "tool_use_count": 1, - "thinking_count": 0, - "total_sessions_with_tools": 1, - "total_sessions_with_thinking": 0, - "tool_use_percentage": 100.0, - "thinking_percentage": 0.0, - "work_event_breakdown": {}, - "repos_active": [], - "origin_breakdown": {}, - "provenance": null - } - ] - } -} diff --git a/.agent/demos/agent-forensics/current/summary.json b/.agent/demos/agent-forensics/current/summary.json deleted file mode 100644 index 6e234a2b91..0000000000 --- a/.agent/demos/agent-forensics/current/summary.json +++ /dev/null @@ -1,107 +0,0 @@ -{ - "artifact": "agent-forensics", - "title": "Agent Forensics (seeded fixture)", - "generated_at": "2026-08-02T13:52:20Z", - "archive_root": "/path/to/demo-archive", - "index_schema_version": 54, - "archive_cardinality": { - "physical_sessions": 19, - "messages": 71, - "blocks": 121, - "session_profiles_materialized": 19, - "origin_rows": 8, - "usage_timeline_rows": 5 - }, - "physical_session_tokens_accounted": 388872, - "logical_session_high_water_tokens_accounted": 388872, - "all_provider_replay_gap_tokens": 0, - "claude_code_physical_session_tokens": 388500, - "claude_code_logical_session_high_water_tokens": 388500, - "claude_code_replay_gap_tokens": 0, - "codex_physical_session_tokens": 0, - "codex_logical_session_high_water_tokens": 0, - "codex_replay_gap_tokens": 0, - "stored_provider_priced_usd": 2.835, - "catalog_api_equivalent_usd": 2.835, - "logical_catalog_api_equivalent_usd": 2.835, - "claim": "Polylogue can regenerate a current v54 longitudinal agent-usage forensics packet over a deterministic seeded fixture archive using product analysis surfaces, with provenance-separated token/cost lanes and explicit physical-vs-logical token grains.", - "non_claim": "This packet is not provider billing truth, not an LLM judgment of failure follow-up behavior, not a resurrected standalone forensics script, and not evidence about the operator's real archive; it reads the deterministic demo archive through the same product commands used against a live archive, and points structured-failure follow-up to the current claim-vs-evidence packet.", - "proof_fields": [ - "archive_cardinality", - "physical_session_tokens_accounted", - "logical_session_high_water_tokens_accounted", - "pricing_lanes", - "coverage-origin.json", - "usage-timeline-month-origin-model.json" - ], - "caveat_fields": [ - "usage_headline_caveats", - "cost_rollups_performance", - "coverage_semantics", - "fixture_scale_caveat" - ], - "fixture_scale_caveat": [ - "This packet was regenerated against the deterministic seeded demo archive (19 sessions, 71 messages), not a live operator archive (polylogue-0bgr: the prior version of this packet was generated against a live archive and committed real corpus size, token totals, and dollar figures to this public repo).", - "Headline totals here are illustrative of report shape and command wiring only; they carry no evidentiary weight about real usage/cost at any scale." - ], - "usage_headline_caveats": [ - "provider usage events, transcript text volume, and model rollups are separate evidence streams", - "this report does not query provider billing and is not a precise cost report", - "headline detail computes session/source/model-rollup totals only; run with detail='full' for provider-event, cumulative, sample, and stale-rollup diagnostics" - ], - "pricing_lanes": [ - { - "provenance": "priced", - "row_count": 1, - "session_count": 1, - "matched_model_row_count": 1, - "unmatched_model_row_count": 0, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "stored_cost_usd": 2.835, - "catalog_api_equivalent_usd": 2.835, - "caveats": [] - } - ], - "logical_pricing_lanes": [ - { - "provenance": "priced", - "row_count": 1, - "session_count": 1, - "matched_model_row_count": 1, - "unmatched_model_row_count": 0, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "stored_cost_usd": 0.0, - "catalog_api_equivalent_usd": 2.835, - "caveats": [] - } - ], - "command_proofs": { - "workload": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain ops diagnostics workload --json", - "usage_all": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze usage --detail headline --format json --limit 0", - "usage_claude_code": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze usage --detail headline --origin claude-code-session --format json --limit 0", - "usage_codex": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze usage --detail headline --origin codex-session --format json --limit 0", - "coverage_origin": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze insights coverage --group-by origin --format json --limit 1000", - "usage_timeline": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze insights usage-timeline --group-by month-origin-model --format json --limit 500" - }, - "cost_rollups_performance": { - "command": "POLYLOGUE_ARCHIVE_ROOT=/path/to/demo-archive polylogue --plain analyze insights cost-rollups --format json --limit 100", - "timeout_s": 120, - "exit_code": 0, - "elapsed_s": null, - "impact": "Completes immediately on the tiny seeded fixture (10 rollup rows; see cost-rollups-timeout.txt). The prior live-archive drilldown timeout this file documented was a real product-performance finding at live-archive scale; it does not reproduce on a fixture this small and is tracked as a separate performance follow-up, not evidence carried by this packet." - } -} diff --git a/.agent/demos/agent-forensics/current/usage-headline-all.json b/.agent/demos/agent-forensics/current/usage-headline-all.json deleted file mode 100644 index f1a7917c75..0000000000 --- a/.agent/demos/agent-forensics/current/usage-headline-all.json +++ /dev/null @@ -1,1775 +0,0 @@ -{ - "archive_root": "/path/to/demo-archive", - "detail_level": "headline", - "coverage_matrix": [ - { - "origin": "claude-code-session", - "provider": "claude-code", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "message_usage" - ], - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "notes": [ - "Exact token telemetry is available only where exported records include message.usage." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "codex-session", - "provider": "codex", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "token_count" - ], - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "notes": [ - "model_context_window is carried on token_count events when the provider supplies it." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "chatgpt-export", - "provider": "chatgpt", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "ChatGPT exports do not carry reliable per-request token counters.", - "cumulative_semantics": "Provider/account UI totals are external summaries and are not reconstructed from transcript text.", - "cache_semantics": "No cache read/write token lanes are available in ChatGPT export rows.", - "notes": [ - "Cost-looking metadata is not treated as exact token telemetry." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "claude-ai-export", - "provider": "claude-ai", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "Claude.ai exports preserve transcript content, not provider usage counters.", - "cumulative_semantics": "No cumulative provider usage window is present in the export shape.", - "cache_semantics": "No cache read/write token lanes are available in Claude.ai export rows.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "aistudio-drive", - "provider": "gemini", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "AI Studio/Gemini exports may carry message-level tokenCount output counters on some records.", - "cumulative_semantics": "No provider cumulative session usage window is available from Drive prompt exports.", - "cache_semantics": "No cache read/write token lanes are available from Drive prompt exports.", - "notes": [ - "Input tokens and cache semantics are missing unless a future export shape supplies them explicitly." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "gemini-cli-session", - "provider": "gemini-cli", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Local Gemini CLI documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "hermes-session", - "provider": "hermes", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Hermes local-agent documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "antigravity-session", - "provider": "antigravity", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "No provider usage telemetry parser is implemented for this origin.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "unknown-export", - "provider": "unknown", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "Unknown exports are parsed for transcript content only.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - } - ], - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 56256, - "output_tokens": 10596, - "cached_input_tokens": 280020, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388872 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 56256, - "output_tokens": 10596, - "cached_input_tokens": 280020, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388872 - }, - "pricing_catalog_provenance": "litellm-model-prices-vendored+polylogue-curated-overrides", - "pricing_catalog_effective_date": "2026-06-27", - "pricing_lanes": [ - { - "provenance": "priced", - "row_count": 1, - "session_count": 1, - "matched_model_row_count": 1, - "unmatched_model_row_count": 0, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "stored_cost_usd": 2.835, - "catalog_api_equivalent_usd": 2.835, - "catalog_priced_subtotal_usd": 2.835, - "subscription_credit_usd": 0.0, - "grain": "physical_session", - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [] - }, - { - "provenance": "unknown", - "row_count": 5, - "session_count": 5, - "matched_model_row_count": 3, - "unmatched_model_row_count": 2, - "usage": { - "input_tokens": 256, - "output_tokens": 96, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 372 - }, - "stored_cost_usd": 0.0, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 0.0, - "subscription_credit_usd": 0.0, - "grain": "physical_session", - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:physical_session:unknown:exact-total-tokens", - "value_state": "known", - "value": 372, - "measurement_authority": [ - "structural" - ], - "weakest_measurement_authority": "structural", - "evidence_refs": [ - "insight:session-model-usage:physical_session:unknown" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 5, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:physical_session:unknown:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:unknown" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:physical_session:unknown", - "reason": "unpriced-model-rows:2" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [ - "missing_price", - "unpriced_cache_read_or_missing_price" - ] - } - ], - "pricing_grain": "physical_session", - "logical_pricing_lanes": [ - { - "provenance": "priced", - "row_count": 1, - "session_count": 1, - "matched_model_row_count": 1, - "unmatched_model_row_count": 0, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "stored_cost_usd": 0.0, - "catalog_api_equivalent_usd": 2.835, - "catalog_priced_subtotal_usd": 2.835, - "subscription_credit_usd": 0.0, - "grain": "logical_session_model_high_water", - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [] - }, - { - "provenance": "unknown", - "row_count": 5, - "session_count": 5, - "matched_model_row_count": 3, - "unmatched_model_row_count": 2, - "usage": { - "input_tokens": 256, - "output_tokens": 96, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 372 - }, - "stored_cost_usd": 0.0, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 0.0, - "subscription_credit_usd": 0.0, - "grain": "logical_session_model_high_water", - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:unknown:exact-total-tokens", - "value_state": "known", - "value": 372, - "measurement_authority": [ - "structural" - ], - "weakest_measurement_authority": "structural", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:unknown" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 5, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:unknown:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:unknown" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:unknown", - "reason": "unpriced-model-rows:2" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [ - "missing_price", - "unpriced_cache_read_or_missing_price" - ] - } - ], - "logical_pricing_grain": "logical_session_model_high_water", - "stored_provider_priced_usd": 2.835, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 2.835, - "logical_catalog_api_equivalent_usd": null, - "logical_catalog_priced_subtotal_usd": 2.835, - "subscription_credit_catalog_provenance": "polylogue-curated-claude-code-subscription-v1", - "subscription_credit_catalog_effective_date": "2026-05-07", - "subscription_credit_usd": 0.0, - "logical_subscription_credit_usd": 0.0, - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:physical-exact-total-tokens:total", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported", - "structural" - ], - "weakest_measurement_authority": "structural", - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced", - "insight:session-model-usage:physical_session:unknown" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 2, - "observed_count": 2, - "supported_count": 2, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "insight:session-model-usage:physical_session:unknown:exact-total-tokens" - ], - "observed_refs": [ - "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "insight:session-model-usage:physical_session:unknown:exact-total-tokens" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - }, - { - "fact_ref": "insight:session-model-usage:physical_session:unknown:exact-total-tokens", - "value_state": "known", - "value": 372, - "measurement_authority": [ - "structural" - ], - "evidence_refs": [ - "insight:session-model-usage:physical_session:unknown" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 5, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:physical-catalog-api-equivalent-cost:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced", - "insight:session-model-usage:physical_session:unknown" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 2, - "observed_count": 2, - "supported_count": 1, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "insight:session-model-usage:physical_session:unknown:catalog-api-equivalent-cost" - ], - "observed_refs": [ - "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "insight:session-model-usage:physical_session:unknown:catalog-api-equivalent-cost" - ], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:physical_session:unknown", - "reason": "unpriced-model-rows:2" - }, - { - "subject_ref": "insight:session-model-usage:physical_session:unknown:catalog-api-equivalent-cost", - "reason": "contribution-coverage-incomplete" - }, - { - "subject_ref": "insight:session-model-usage:physical_session:unknown:catalog-api-equivalent-cost", - "reason": "unsupported-contribution:unknown" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - }, - { - "fact_ref": "insight:session-model-usage:physical_session:unknown:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:unknown" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:physical_session:unknown", - "reason": "unpriced-model-rows:2" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "logical_exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:logical-exact-total-tokens:total", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported", - "structural" - ], - "weakest_measurement_authority": "structural", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced", - "insight:session-model-usage:logical_session_model_high_water:unknown" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 2, - "observed_count": 2, - "supported_count": 2, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "insight:session-model-usage:logical_session_model_high_water:unknown:exact-total-tokens" - ], - "observed_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "insight:session-model-usage:logical_session_model_high_water:unknown:exact-total-tokens" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - }, - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:unknown:exact-total-tokens", - "value_state": "known", - "value": 372, - "measurement_authority": [ - "structural" - ], - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:unknown" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 5, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "logical_catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:logical-catalog-api-equivalent-cost:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced", - "insight:session-model-usage:logical_session_model_high_water:unknown" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 2, - "observed_count": 2, - "supported_count": 1, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "insight:session-model-usage:logical_session_model_high_water:unknown:catalog-api-equivalent-cost" - ], - "observed_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "insight:session-model-usage:logical_session_model_high_water:unknown:catalog-api-equivalent-cost" - ], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:unknown", - "reason": "unpriced-model-rows:2" - }, - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:unknown:catalog-api-equivalent-cost", - "reason": "contribution-coverage-incomplete" - }, - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:unknown:catalog-api-equivalent-cost", - "reason": "unsupported-contribution:unknown" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - }, - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:unknown:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:unknown" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:10.419644+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'unknown'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 5, - "observed_count": 5, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:unknown", - "reason": "unpriced-model-rows:2" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:10.419644+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "origins": [ - { - "origin": "aistudio-drive", - "detail_level": "headline", - "provider": "gemini", - "declared_coverage": "partial", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "message_token_fields", - "request_semantics": "AI Studio/Gemini exports may carry message-level tokenCount output counters on some records.", - "cumulative_semantics": "No provider cumulative session usage window is available from Drive prompt exports.", - "cache_semantics": "No cache read/write token lanes are available from Drive prompt exports.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 4, - "transcript_word_count": 24, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry is partial; request, cumulative, and cache semantics are incomplete" - ] - }, - { - "origin": "antigravity-session", - "detail_level": "headline", - "provider": "antigravity", - "declared_coverage": "unsupported", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "transcript_text_only", - "request_semantics": "No provider usage telemetry parser is implemented for this origin.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 2, - "transcript_word_count": 29, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry unsupported for this origin" - ] - }, - { - "origin": "chatgpt-export", - "detail_level": "headline", - "provider": "chatgpt", - "declared_coverage": "estimate_only", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "transcript_text_estimate", - "request_semantics": "ChatGPT exports do not carry reliable per-request token counters.", - "cumulative_semantics": "Provider/account UI totals are external summaries and are not reconstructed from transcript text.", - "cache_semantics": "No cache read/write token lanes are available in ChatGPT export rows.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 3, - "message_count": 7, - "transcript_word_count": 251, - "raw_session_count": 5, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 2, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "exact provider telemetry unavailable; transcript text counts are estimate-only", - "raw rows without parse errors are acquired but not materialized; usage coverage is incomplete until index.db is rebuilt" - ] - }, - { - "origin": "claude-ai-export", - "detail_level": "headline", - "provider": "claude-ai", - "declared_coverage": "estimate_only", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "transcript_text_estimate", - "request_semantics": "Claude.ai exports preserve transcript content, not provider usage counters.", - "cumulative_semantics": "No cumulative provider usage window is present in the export shape.", - "cache_semantics": "No cache read/write token lanes are available in Claude.ai export rows.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 2, - "transcript_word_count": 17, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "exact provider telemetry unavailable; transcript text counts are estimate-only" - ] - }, - { - "origin": "claude-code-session", - "detail_level": "headline", - "provider": "claude-code", - "declared_coverage": "exact", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "provider_reported_usage", - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 4, - "message_count": 19, - "transcript_word_count": 398, - "raw_session_count": 4, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 1, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "some sessions have no provider usage event rows; transcript words and model rollups cover different evidence" - ] - }, - { - "origin": "codex-session", - "detail_level": "headline", - "provider": "codex", - "declared_coverage": "exact", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "provider_reported_usage", - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 7, - "message_count": 30, - "transcript_word_count": 229, - "raw_session_count": 7, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "some sessions have no provider usage event rows; transcript words and model rollups cover different evidence" - ] - }, - { - "origin": "gemini-cli-session", - "detail_level": "headline", - "provider": "gemini-cli", - "declared_coverage": "partial", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "message_token_fields", - "request_semantics": "Local Gemini CLI documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 2, - "transcript_word_count": 33, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 160, - "output_tokens": 64, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 244 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 160, - "output_tokens": 64, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 244 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry is partial; request, cumulative, and cache semantics are incomplete" - ] - }, - { - "origin": "hermes-session", - "detail_level": "headline", - "provider": "hermes", - "declared_coverage": "partial", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "message_token_fields", - "request_semantics": "Hermes local-agent documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 5, - "transcript_word_count": 39, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 96, - "output_tokens": 32, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 128 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 96, - "output_tokens": 32, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 128 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry is partial; request, cumulative, and cache semantics are incomplete" - ] - } - ], - "caveats": [ - "provider usage events, transcript text volume, and model rollups are separate evidence streams", - "this report does not query provider billing and is not a precise cost report", - "headline detail computes session/source/model-rollup totals only; run with detail='full' for provider-event, cumulative, sample, and stale-rollup diagnostics", - "catalog_api_equivalent_usd is unknown because at least one model row has no catalog price; catalog_priced_subtotal_usd preserves the known priced subset" - ] -} diff --git a/.agent/demos/agent-forensics/current/usage-headline-claude-code.json b/.agent/demos/agent-forensics/current/usage-headline-claude-code.json deleted file mode 100644 index 84e1a82465..0000000000 --- a/.agent/demos/agent-forensics/current/usage-headline-claude-code.json +++ /dev/null @@ -1,826 +0,0 @@ -{ - "archive_root": "/path/to/demo-archive", - "detail_level": "headline", - "coverage_matrix": [ - { - "origin": "claude-code-session", - "provider": "claude-code", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "message_usage" - ], - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "notes": [ - "Exact token telemetry is available only where exported records include message.usage." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "codex-session", - "provider": "codex", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "token_count" - ], - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "notes": [ - "model_context_window is carried on token_count events when the provider supplies it." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "chatgpt-export", - "provider": "chatgpt", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "ChatGPT exports do not carry reliable per-request token counters.", - "cumulative_semantics": "Provider/account UI totals are external summaries and are not reconstructed from transcript text.", - "cache_semantics": "No cache read/write token lanes are available in ChatGPT export rows.", - "notes": [ - "Cost-looking metadata is not treated as exact token telemetry." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "claude-ai-export", - "provider": "claude-ai", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "Claude.ai exports preserve transcript content, not provider usage counters.", - "cumulative_semantics": "No cumulative provider usage window is present in the export shape.", - "cache_semantics": "No cache read/write token lanes are available in Claude.ai export rows.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "aistudio-drive", - "provider": "gemini", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "AI Studio/Gemini exports may carry message-level tokenCount output counters on some records.", - "cumulative_semantics": "No provider cumulative session usage window is available from Drive prompt exports.", - "cache_semantics": "No cache read/write token lanes are available from Drive prompt exports.", - "notes": [ - "Input tokens and cache semantics are missing unless a future export shape supplies them explicitly." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "gemini-cli-session", - "provider": "gemini-cli", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Local Gemini CLI documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "hermes-session", - "provider": "hermes", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Hermes local-agent documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "antigravity-session", - "provider": "antigravity", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "No provider usage telemetry parser is implemented for this origin.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "unknown-export", - "provider": "unknown", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "Unknown exports are parsed for transcript content only.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - } - ], - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "pricing_catalog_provenance": "litellm-model-prices-vendored+polylogue-curated-overrides", - "pricing_catalog_effective_date": "2026-06-27", - "pricing_lanes": [ - { - "provenance": "priced", - "row_count": 1, - "session_count": 1, - "matched_model_row_count": 1, - "unmatched_model_row_count": 0, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "stored_cost_usd": 2.835, - "catalog_api_equivalent_usd": 2.835, - "catalog_priced_subtotal_usd": 2.835, - "subscription_credit_usd": 0.0, - "grain": "physical_session", - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [] - } - ], - "pricing_grain": "physical_session", - "logical_pricing_lanes": [ - { - "provenance": "priced", - "row_count": 1, - "session_count": 1, - "matched_model_row_count": 1, - "unmatched_model_row_count": 0, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "stored_cost_usd": 0.0, - "catalog_api_equivalent_usd": 2.835, - "catalog_priced_subtotal_usd": 2.835, - "subscription_credit_usd": 0.0, - "grain": "logical_session_model_high_water", - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [] - } - ], - "logical_pricing_grain": "logical_session_model_high_water", - "stored_provider_priced_usd": 2.835, - "catalog_api_equivalent_usd": 2.835, - "catalog_priced_subtotal_usd": 2.835, - "logical_catalog_api_equivalent_usd": 2.835, - "logical_catalog_priced_subtotal_usd": 2.835, - "subscription_credit_catalog_provenance": "polylogue-curated-claude-code-subscription-v1", - "subscription_credit_catalog_effective_date": "2026-05-07", - "subscription_credit_usd": 0.0, - "logical_subscription_credit_usd": 0.0, - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:physical-exact-total-tokens:total", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:physical_session:priced:exact-total-tokens" - ], - "observed_refs": [ - "insight:session-model-usage:physical_session:priced:exact-total-tokens" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:physical-catalog-api-equivalent-cost:total", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost" - ], - "observed_refs": [ - "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "logical_exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:logical-exact-total-tokens:total", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens" - ], - "observed_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "value_state": "known", - "value": 388500, - "measurement_authority": [ - "provider-reported" - ], - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "logical_catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:logical-catalog-api-equivalent-cost:total", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost" - ], - "observed_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "value_state": "known", - "value": 2.835, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "temporal": { - "observed_at": "2026-08-02T13:52:20.867273+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-08-02T13:52:20.867273+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "origins": [ - { - "origin": "claude-code-session", - "detail_level": "headline", - "provider": "claude-code", - "declared_coverage": "exact", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "provider_reported_usage", - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 4, - "message_count": 19, - "transcript_word_count": 398, - "raw_session_count": 4, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 1, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "some sessions have no provider usage event rows; transcript words and model rollups cover different evidence" - ] - } - ], - "caveats": [ - "provider usage events, transcript text volume, and model rollups are separate evidence streams", - "this report does not query provider billing and is not a precise cost report", - "headline detail computes session/source/model-rollup totals only; run with detail='full' for provider-event, cumulative, sample, and stale-rollup diagnostics" - ] -} diff --git a/.agent/demos/agent-forensics/current/usage-headline-codex.json b/.agent/demos/agent-forensics/current/usage-headline-codex.json deleted file mode 100644 index b1cebd6fae..0000000000 --- a/.agent/demos/agent-forensics/current/usage-headline-codex.json +++ /dev/null @@ -1,402 +0,0 @@ -{ - "archive_root": "/path/to/demo-archive", - "detail_level": "headline", - "coverage_matrix": [ - { - "origin": "claude-code-session", - "provider": "claude-code", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "message_usage" - ], - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "notes": [ - "Exact token telemetry is available only where exported records include message.usage." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "codex-session", - "provider": "codex", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "token_count" - ], - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "notes": [ - "model_context_window is carried on token_count events when the provider supplies it." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "chatgpt-export", - "provider": "chatgpt", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "ChatGPT exports do not carry reliable per-request token counters.", - "cumulative_semantics": "Provider/account UI totals are external summaries and are not reconstructed from transcript text.", - "cache_semantics": "No cache read/write token lanes are available in ChatGPT export rows.", - "notes": [ - "Cost-looking metadata is not treated as exact token telemetry." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "claude-ai-export", - "provider": "claude-ai", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "Claude.ai exports preserve transcript content, not provider usage counters.", - "cumulative_semantics": "No cumulative provider usage window is present in the export shape.", - "cache_semantics": "No cache read/write token lanes are available in Claude.ai export rows.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "aistudio-drive", - "provider": "gemini", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "AI Studio/Gemini exports may carry message-level tokenCount output counters on some records.", - "cumulative_semantics": "No provider cumulative session usage window is available from Drive prompt exports.", - "cache_semantics": "No cache read/write token lanes are available from Drive prompt exports.", - "notes": [ - "Input tokens and cache semantics are missing unless a future export shape supplies them explicitly." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "gemini-cli-session", - "provider": "gemini-cli", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Local Gemini CLI documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "hermes-session", - "provider": "hermes", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Hermes local-agent documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "antigravity-session", - "provider": "antigravity", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "No provider usage telemetry parser is implemented for this origin.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "unknown-export", - "provider": "unknown", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "Unknown exports are parsed for transcript content only.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - } - ], - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "pricing_catalog_provenance": "litellm-model-prices-vendored+polylogue-curated-overrides", - "pricing_catalog_effective_date": "2026-06-27", - "pricing_lanes": [], - "pricing_grain": "physical_session", - "logical_pricing_lanes": [], - "logical_pricing_grain": "logical_session_model_high_water", - "stored_provider_priced_usd": 0, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 0, - "logical_catalog_api_equivalent_usd": null, - "logical_catalog_priced_subtotal_usd": 0, - "subscription_credit_catalog_provenance": "polylogue-curated-claude-code-subscription-v1", - "subscription_credit_catalog_effective_date": "2026-05-07", - "subscription_credit_usd": 0, - "logical_subscription_credit_usd": 0, - "observed_at": "2026-08-02T13:52:31.038402+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:physical-exact-total-tokens:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [], - "weakest_measurement_authority": null, - "evidence_refs": [], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:31.038402+00:00", - "time_source": null, - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "inferred-partial", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 0, - "observed_count": 0, - "supported_count": 0, - "complete": false, - "coverage_ratio": null, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "unavailable", - "evaluated_at": "2026-08-02T13:52:31.038402+00:00", - "cause": "no-freshness-evidence", - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:physical-catalog-api-equivalent-cost:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [], - "weakest_measurement_authority": null, - "evidence_refs": [], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:31.038402+00:00", - "time_source": null, - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "inferred-partial", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 0, - "observed_count": 0, - "supported_count": 0, - "complete": false, - "coverage_ratio": null, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "unavailable", - "evaluated_at": "2026-08-02T13:52:31.038402+00:00", - "cause": "no-freshness-evidence", - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "logical_exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:logical-exact-total-tokens:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [], - "weakest_measurement_authority": null, - "evidence_refs": [], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:31.038402+00:00", - "time_source": null, - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "inferred-partial", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 0, - "observed_count": 0, - "supported_count": 0, - "complete": false, - "coverage_ratio": null, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "unavailable", - "evaluated_at": "2026-08-02T13:52:31.038402+00:00", - "cause": "no-freshness-evidence", - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "logical_catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:logical-catalog-api-equivalent-cost:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [], - "weakest_measurement_authority": null, - "evidence_refs": [], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-08-02T13:52:31.038402+00:00", - "time_source": null, - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "inferred-partial", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 0, - "observed_count": 0, - "supported_count": 0, - "complete": false, - "coverage_ratio": null, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "unavailable", - "evaluated_at": "2026-08-02T13:52:31.038402+00:00", - "cause": "no-freshness-evidence", - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "origins": [ - { - "origin": "codex-session", - "detail_level": "headline", - "provider": "codex", - "declared_coverage": "exact", - "coverage_state": "headline_not_audited", - "coverage_basis": "headline detail does not scan provider usage events or stale rollup diagnostics; model rollup totals are still computed from session_model_usage", - "evidence_stream": "provider_reported_usage", - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 7, - "message_count": 30, - "transcript_word_count": 229, - "raw_session_count": 7, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "some sessions have no provider usage event rows; transcript words and model rollups cover different evidence" - ] - } - ], - "caveats": [ - "provider usage events, transcript text volume, and model rollups are separate evidence streams", - "this report does not query provider billing and is not a precise cost report", - "headline detail computes session/source/model-rollup totals only; run with detail='full' for provider-event, cumulative, sample, and stale-rollup diagnostics" - ] -} diff --git a/.agent/demos/agent-forensics/current/usage-timeline-month-origin-model.json b/.agent/demos/agent-forensics/current/usage-timeline-month-origin-model.json deleted file mode 100644 index 99fa7e0c9f..0000000000 --- a/.agent/demos/agent-forensics/current/usage-timeline-month-origin-model.json +++ /dev/null @@ -1,178 +0,0 @@ -{ - "status": "ok", - "result": { - "total": 5, - "usage_timeline": [ - { - "contract_version": 10, - "insight_kind": "usage_timeline", - "semantic_tier": "evidence", - "group_by": "month-origin-model", - "bucket": "2024-02", - "origin": "aistudio-drive", - "model_name": "models/gemini-2.5-pro models/gemini", - "normalized_model": "gemini-2.5-pro", - "session_count": 1, - "event_count": 0, - "usage": { - "input_tokens": 0, - "output_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0, - "total_tokens": 0 - }, - "reasoning_output_tokens": 0, - "stored_cost_usd": 0.0, - "subscription_credits": 0.0, - "cost_provenance_counts": { - "unknown": 1 - }, - "provenance": { - "materializer_version": 0, - "materialized_at": "2026-08-02T13:52:52.183374+00:00", - "source_updated_at": null, - "source_sort_key": 1706934756990.0, - "input_high_water_mark": null, - "input_high_water_mark_source": null, - "time_confidence": "unknown" - } - }, - { - "contract_version": 10, - "insight_kind": "usage_timeline", - "semantic_tier": "evidence", - "group_by": "month-origin-model", - "bucket": "2024-11", - "origin": "claude-code-session", - "model_name": "claude-opus-4-8", - "normalized_model": "claude-opus-4-8", - "session_count": 1, - "event_count": 0, - "usage": { - "input_tokens": 0, - "output_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0, - "total_tokens": 0 - }, - "reasoning_output_tokens": 0, - "stored_cost_usd": 2.835, - "subscription_credits": 0.0, - "cost_provenance_counts": { - "priced": 1 - }, - "provenance": { - "materializer_version": 0, - "materialized_at": "2026-08-02T13:52:52.183374+00:00", - "source_updated_at": null, - "source_sort_key": 1730589655737.0, - "input_high_water_mark": null, - "input_high_water_mark_source": null, - "time_confidence": "unknown" - } - }, - { - "contract_version": 10, - "insight_kind": "usage_timeline", - "semantic_tier": "evidence", - "group_by": "month-origin-model", - "bucket": "2026-07", - "origin": "chatgpt-export", - "model_name": "gpt-5-demo", - "normalized_model": "gpt-5", - "session_count": 2, - "event_count": 0, - "usage": { - "input_tokens": 0, - "output_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0, - "total_tokens": 0 - }, - "reasoning_output_tokens": 0, - "stored_cost_usd": 0.0, - "subscription_credits": 0.0, - "cost_provenance_counts": { - "unknown": 2 - }, - "provenance": { - "materializer_version": 0, - "materialized_at": "2026-08-02T13:52:52.183374+00:00", - "source_updated_at": null, - "source_sort_key": 1783158903000.0, - "input_high_water_mark": null, - "input_high_water_mark_source": null, - "time_confidence": "unknown" - } - }, - { - "contract_version": 10, - "insight_kind": "usage_timeline", - "semantic_tier": "evidence", - "group_by": "month-origin-model", - "bucket": "2026-07", - "origin": "gemini-cli-session", - "model_name": "gemini-2.5-demo", - "normalized_model": "gemini-2.5-demo", - "session_count": 1, - "event_count": 1, - "usage": { - "input_tokens": 160, - "output_tokens": 64, - "cache_read_tokens": 20, - "cache_write_tokens": 0, - "total_tokens": 0 - }, - "reasoning_output_tokens": 0, - "stored_cost_usd": 0.0, - "subscription_credits": 0.0, - "cost_provenance_counts": { - "unknown": 1 - }, - "provenance": { - "materializer_version": 0, - "materialized_at": "2026-08-02T13:52:52.183374+00:00", - "source_updated_at": null, - "source_sort_key": 1783159084000.0, - "input_high_water_mark": null, - "input_high_water_mark_source": null, - "time_confidence": "unknown" - } - }, - { - "contract_version": 10, - "insight_kind": "usage_timeline", - "semantic_tier": "evidence", - "group_by": "month-origin-model", - "bucket": "2026-07", - "origin": "hermes-session", - "model_name": "nous-hermes-demo", - "normalized_model": "nous-hermes-demo", - "session_count": 1, - "event_count": 0, - "usage": { - "input_tokens": 0, - "output_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0, - "total_tokens": 0 - }, - "reasoning_output_tokens": 0, - "stored_cost_usd": 0.0, - "subscription_credits": 0.0, - "cost_provenance_counts": { - "unknown": 1 - }, - "provenance": { - "materializer_version": 0, - "materialized_at": "2026-08-02T13:52:52.183374+00:00", - "source_updated_at": null, - "source_sort_key": 1783159204000.0, - "input_high_water_mark": null, - "input_high_water_mark_source": null, - "time_confidence": "unknown" - } - } - ] - } -} diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/NON-CLAIMS.md b/.agent/demos/anti-demo-multi-source-reconstruction/NON-CLAIMS.md deleted file mode 100644 index 90c76d940d..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/NON-CLAIMS.md +++ /dev/null @@ -1,4 +0,0 @@ -# Non-claims - -- This refusal does not claim that cross-source reconstruction is impossible in a federated Sinex-backed architecture. -- This packet does not measure production source coverage or the quality of any external activity system. diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/PROMPT.md b/.agent/demos/anti-demo-multi-source-reconstruction/PROMPT.md deleted file mode 100644 index 61bab51810..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/PROMPT.md +++ /dev/null @@ -1,35 +0,0 @@ -# The Honesty Anti-Demo: Minute-by-Minute Multi-Source Reconstruction - -Predeclaration receipt: `artifact:anti-demo-multi-source-reconstruction-predeclaration`. - -This demo's success is refusal. Attempt the tempting claim below against the -archive; the correct outcome is a `checks.json` with `verdict: -"not_supported"`, naming every missing modality/ref and the bead that would -supply it (or stating honestly that no such bead exists yet). - -## The tempting claim - -"Reconstruct exactly what the operator was doing, minute-by-minute, on a -given day — correlating AI chat sessions with desktop window focus, shell -commands typed outside any captured agent session, and browser tabs viewed — -into one fused timeline." - -## Why this is tempting - -Polylogue already answers adjacent, real questions from the same archive: -per-session git commit correlation (`session_commits`, confidence-scored), -per-repo session grouping (`session_repos`), and within-session tool-call -timing. It would be easy to imply these compose into a full desktop -timeline. They do not, and the gap is structural, not a missing query. - -## What to check before claiming success - -1. Query `session_commits`/`session_repos` for the day in question — confirm - what granularity they actually offer (session-to-repo/commit linkage, - not minute-level desktop activity). -2. Search the archive schema for any table holding desktop window-focus, - raw shell history, or browser-tab telemetry — confirm none exists in - Polylogue's own tiers (source/index/embeddings/user/ops). -3. Do not substitute a different, easier claim and declare success; do not - quietly narrow the claim's wording after failing to support it. Report - the exact gap. diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/checks.json b/.agent/demos/anti-demo-multi-source-reconstruction/checks.json deleted file mode 100644 index a2e455dac2..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/checks.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "pass": true, - "verdict": "not_supported", - "unsupported_claims": [ - "minute-by-minute cross-source (chat + desktop window focus + shell history + browser tabs) reconstruction of operator activity for a given day" - ], - "coverage_notes": "pass=true means the refusal itself is correctly evidenced (schema-grep confirms no window-focus/shell-history/browser-tab table exists in any Polylogue tier), not that the analytical claim succeeded. No bead currently owns cross-system (Polylogue+Lynchpin) timeline fusion -- stated plainly rather than citing an invented or loosely-related bead ref." -} diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/evidence.ndjson b/.agent/demos/anti-demo-multi-source-reconstruction/evidence.ndjson deleted file mode 100644 index 7fea6f7ea5..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/evidence.ndjson +++ /dev/null @@ -1,4 +0,0 @@ -{"ref": "artifact:anti-demo-multi-source-reconstruction-evidence", "cited_for": "Demo Packet v2 receipt root", "verified_via": "committed evidence.ndjson"} -{"ref": "polylogue/storage/sqlite/archive_tiers/index.py:517", "cited_for": "session_commits table exists: session-grained git correlation, confidence-scored, not minute-level", "verified_via": "grep -n 'CREATE TABLE IF NOT EXISTS session_commits' -A 10 polylogue/storage/sqlite/archive_tiers/index.py"} -{"ref": "polylogue/storage/sqlite/archive_tiers/index.py:506", "cited_for": "session_repos table exists: session-to-repo/branch linkage, not minute-level", "verified_via": "grep -n 'CREATE TABLE IF NOT EXISTS session_repos' polylogue/storage/sqlite/archive_tiers/index.py"} -{"ref": "polylogue/storage/sqlite/archive_tiers/*.py", "cited_for": "no window-focus/shell-history/browser-tab telemetry table exists in any Polylogue tier", "verified_via": "grep -rln 'window_focus|shell_history|browser_tab|activitywatch' polylogue/storage/sqlite/archive_tiers/*.py (zero matches)"} diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/finding.yaml b/.agent/demos/anti-demo-multi-source-reconstruction/finding.yaml deleted file mode 100644 index 11a59be224..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/finding.yaml +++ /dev/null @@ -1,7 +0,0 @@ -archive_cursor: polylogue-demo-seed-fixture-world -measure_version: demo-packet-v2 -commit_sha: e5a68ec51 -sample_frame_predicate: "polylogue archive schema (source/index/embeddings/user/ops tiers) plus the seeded demo corpus; the claim is refused on structural schema grounds, not corpus size" -run_date: "2026-07-09" -claim: "REFUSED: a minute-by-minute, cross-source (chat + desktop window focus + shell + browser) reconstruction of operator activity for a given day is not supported by the current archive" -verdict: not_supported diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/packet.json b/.agent/demos/anti-demo-multi-source-reconstruction/packet.json deleted file mode 100644 index e1e5570aee..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/packet.json +++ /dev/null @@ -1,132 +0,0 @@ -{ - "baseline": { - "method": "Use the recorded simpler comparison path rather than the structural product path.", - "name": "plausible narrative completion", - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-evidence" - ], - "result": "Infer missing ambient activity from nearby transcript text." - }, - "claim": { - "declared_before_execution": true, - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-predeclaration" - ], - "scope": "the current Polylogue archive tiers at the recorded revision", - "statement": "Minute-by-minute fusion of transcript, window-focus, shell-history, and browser-tab evidence is not supported by this Polylogue schema.", - "status": "not_supported" - }, - "controls": { - "missing_evidence": [ - { - "expected": "No matching ambient source table must produce a refusal, not an empty successful timeline.", - "id": "absent-source-control", - "observed": { - "status": "not_supported" - }, - "passed": true, - "purpose": "Require missing evidence to remain explicit rather than converted into a positive claim.", - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-evidence" - ] - } - ], - "negative": [ - { - "expected": "Existing session-to-repository and session-to-commit rows must not be mislabeled as minute-level ambient evidence.", - "id": "existing-git-correlation", - "observed": { - "ambient_timeline": false - }, - "passed": true, - "purpose": "Prevent an adjacent easier signal from being counted as the primary construct.", - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-evidence" - ] - } - ] - }, - "falsifier": { - "condition": "A current archive-tier table is found that stores minute-grained window-focus, independent shell-history, and browser-tab observations.", - "evaluation_method": "Apply the stated condition to the committed evidence and run log.", - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-evidence" - ], - "result": "pass", - "triggered": false - }, - "mode": "anti-demo", - "non_claims": [ - "This refusal does not claim that cross-source reconstruction is impossible in a federated Sinex-backed architecture.", - "This packet does not measure production source coverage or the quality of any external activity system." - ], - "oracle": { - "description": "The archive DDL independently determines which source domains can exist in Polylogue.", - "expected": { - "ambient_sources_present": false - }, - "independent": true, - "method": "Inspect every archive-tier CREATE TABLE declaration for the required ambient source domains.", - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-evidence" - ] - }, - "packet_id": "anti-demo-multi-source-reconstruction", - "primary_construct": { - "id": "evidence.refusal", - "product_primitives": [ - "schema inspection", - "evidence gap reporting" - ], - "statement": "The product refuses a cross-source reconstruction when required source domains are absent." - }, - "provenance": { - "archive_cursor": "polylogue-demo-seed-fixture-world", - "commit_sha": "e5a68ec51", - "measure_version": "demo-packet-v2", - "run_date": "2026-07-10", - "sample_frame_predicate": "all Polylogue archive-tier DDL files at the recorded revision" - }, - "receipts": [ - { - "artifact_path": "evidence.ndjson", - "description": "Committed evidence rows and references for this packet.", - "kind": "artifact", - "ref": "artifact:anti-demo-multi-source-reconstruction-evidence", - "resolved": true, - "sha256": "4e71bc7bda5121b7e933faf014b025ef66ad659f53cd1e0a3adda3c8e9ef1108" - }, - { - "artifact_path": "PROMPT.md", - "description": "The committed prompt that states the packet claim before execution.", - "kind": "artifact", - "ref": "artifact:anti-demo-multi-source-reconstruction-predeclaration", - "resolved": true, - "sha256": "ec64079d33664dad1fdfc4af633687b2dbadeff9f28463e6c24d1b773332c03f" - } - ], - "reproduction": { - "commands": [ - "python -m devtools.verify_demo_packet_registry" - ], - "deterministic": true, - "fixture": "current archive-tier DDL and deterministic demo seed", - "private_data": false - }, - "results": { - "measurements": [ - { - "name": "required_ambient_source_tables", - "receipts": [ - "artifact:anti-demo-multi-source-reconstruction-evidence" - ], - "unit": "tables", - "value": 0 - } - ], - "status": "pass", - "summary": "The claim is refused on structural schema grounds and the narrower existing correlations are named." - }, - "schema_version": "2.0.0", - "title": "Multi-source reconstruction refusal" -} diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/queries.ndjson b/.agent/demos/anti-demo-multi-source-reconstruction/queries.ndjson deleted file mode 100644 index d4b100043a..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/queries.ndjson +++ /dev/null @@ -1,3 +0,0 @@ -{"text": "grep -n \"CREATE TABLE\" polylogue/storage/sqlite/archive_tiers/index.py | grep -iE \"session_commits|session_repos\"", "lowered_spec": {"kind": "schema-grep", "target": "session-to-external-activity correlation tables"}} -{"text": "grep -n \"CREATE TABLE\" polylogue/storage/sqlite/archive_tiers/index.py | grep -iE \"window|focus|shell_history|browser_tab|activitywatch\"", "lowered_spec": {"kind": "schema-grep", "target": "desktop/shell/browser telemetry tables (expected: none)"}} -{"text": "grep -rln \"window_focus|shell_history|browser_tab|activitywatch\" polylogue/storage/sqlite/archive_tiers/*.py", "lowered_spec": {"kind": "schema-grep", "target": "cross-tier confirmation (expected: none)"}} diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/report.md b/.agent/demos/anti-demo-multi-source-reconstruction/report.md deleted file mode 100644 index a5b3cf6d35..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/report.md +++ /dev/null @@ -1,90 +0,0 @@ -# The Honesty Anti-Demo: Minute-by-Minute Multi-Source Reconstruction - -## Claim - -REFUSED. "Reconstruct exactly what the operator was doing, minute-by-minute, -on a given day — correlating AI chat sessions with desktop window focus, -shell commands typed outside any captured agent session, and browser tabs -viewed — into one fused timeline" is **not supported** by the current -Polylogue archive. - -## Corpus - -Polylogue's own schema (`polylogue/storage/sqlite/archive_tiers/index.py` -and sibling tier DDL files) plus the seeded demo corpus. This is a -structural refusal, not a corpus-size limitation — the claim fails on -schema grounds and would fail identically against the full live archive. - -## Method - -1. Identify what session-to-external-activity correlation actually exists - in the schema. -2. Search for any table carrying desktop window-focus, raw shell history, - or browser-tab telemetry across every archive tier. -3. Refuse the claim if neither exists at the required granularity, naming - the specific gap rather than narrowing the claim quietly. - -## Findings - -**What exists**: `session_commits` (per-session git commit correlation, -`detection_type` one of `time_window|file_overlap|explicit_ref| -origin_reported`, confidence-scored `0..1`) and `session_repos` -(session-to-repo/branch linkage). Both are real, and both are *session*- -grained — they say "this session touched this repo/commit," never "this -number of shell commands ran between 14:02 and 14:03." - -**What does not exist**: no table in any Polylogue tier (source.db, -index.db, embeddings.db, user.db, ops.db) holds desktop window-focus -events, raw shell command history independent of a captured agent session, -or browser tab/navigation telemetry. Confirmed by direct schema grep across -every `archive_tiers/*.py` DDL file (see `run.log`) — zero matches. - -**Where this data actually lives**: window-focus (ActivityWatch), shell -history (Atuin), and browser history are captured by a *separate* system -(sinity-lynchpin), not by Polylogue. Polylogue is scoped to AI session -archives; cross-system correlation with Lynchpin's telemetry is a distinct, -undecided product question, not a missing query on data Polylogue already -holds. - -## Specimens - -See `evidence.ndjson` for the exact grep commands and their (zero-match) -output. - -## Counterexamples - -None — the claim's negative result *is* the finding. There is no adjacent -easier claim substituted here; the demo does not quietly narrow "minute-by- -minute multi-source" down to "session-to-commit" and declare success. - -## Limits - -- This refusal is about Polylogue's own architecture as of this commit. If - a future bead decides to ingest Lynchpin telemetry into Polylogue (or - build a federated cross-system query layer), this verdict would need to - be re-run and could change. -- **No existing bead currently owns cross-system (Polylogue+Lynchpin) - timeline fusion.** This is itself an honest finding: rather than invent a - plausible-sounding bead reference, this report states plainly that the - capability gap has no tracked owner yet. The closest adjacent, already- - scoped primitives are `session_commits`/`session_repos` (session-grained - git correlation, already shipped) and the general query-objects/result-set - direction work (`polylogue-rxdo` family) that could eventually be a - foundation for richer cross-source composition, but neither is a - reference for "the bead that supplies minute-by-minute cross-system - reconstruction" because no such bead exists. - -## Non-claims - -- This refusal does not claim that cross-source reconstruction is impossible in a federated Sinex-backed architecture. -- This packet does not measure production source coverage or the quality of any external activity system. - -## Reproduce - -```bash -grep -n "CREATE TABLE" polylogue/storage/sqlite/archive_tiers/index.py | grep -iE "session_commits|session_repos" -grep -n "CREATE TABLE" polylogue/storage/sqlite/archive_tiers/index.py | grep -iE "window|focus|shell_history|browser_tab|activitywatch" -grep -rln "window_focus|shell_history|browser_tab|activitywatch" polylogue/storage/sqlite/archive_tiers/*.py -``` - -See `run.log` for the exact recorded output. diff --git a/.agent/demos/anti-demo-multi-source-reconstruction/run.log b/.agent/demos/anti-demo-multi-source-reconstruction/run.log deleted file mode 100644 index a299b32c66..0000000000 --- a/.agent/demos/anti-demo-multi-source-reconstruction/run.log +++ /dev/null @@ -1,25 +0,0 @@ -=== Step 1: what session-to-external-activity correlation actually exists === -$ grep -n "CREATE TABLE" polylogue/storage/sqlite/archive_tiers/index.py | grep -iE "session_commits|session_repos" -506:CREATE TABLE IF NOT EXISTS session_repos ( -517:CREATE TABLE IF NOT EXISTS session_commits ( - -session_commits columns (per-session git correlation, confidence-scored, NOT minute-level): -517:CREATE TABLE IF NOT EXISTS session_commits ( -518- session_id TEXT NOT NULL REFERENCES sessions(session_id) ON DELETE CASCADE, -519- commit_sha TEXT NOT NULL, -520- repo_id TEXT REFERENCES repos(repo_id) ON DELETE CASCADE, -521- detection_type TEXT NOT NULL CHECK(detection_type IN ('time_window', 'file_overlap', 'explicit_ref', 'origin_reported')), -522- method TEXT, -523- confidence REAL NOT NULL CHECK(confidence BETWEEN 0 AND 1), -524- evidence_json TEXT NOT NULL DEFAULT '{{}}', -525- created_at_ms INTEGER NOT NULL, -526- PRIMARY KEY(session_id, commit_sha) -527-) STRICT; - -=== Step 2: search for any desktop/shell/browser telemetry table in Polylogue's own schema === -$ grep -n "CREATE TABLE" polylogue/storage/sqlite/archive_tiers/index.py | grep -iE "window|focus|shell_history|browser_tab|activitywatch" -(no output above: no such table exists in polylogue/storage/sqlite/archive_tiers/index.py) - -=== Step 3: confirm no other tier (source/embeddings/user/ops) carries this either === -$ grep -rln "window_focus\|shell_history\|browser_tab\|activitywatch" polylogue/storage/sqlite/archive_tiers/*.py -(no matches -- confirmed absent across all tiers) diff --git a/.agent/demos/attachment-acquisition-census/census.json b/.agent/demos/attachment-acquisition-census/census.json index 938ef7855f..da49a29adc 100644 --- a/.agent/demos/attachment-acquisition-census/census.json +++ b/.agent/demos/attachment-acquisition-census/census.json @@ -44,4 +44,4 @@ "declared_byte_sum": 53, "missing_blob_ref_count": 0 } -} \ No newline at end of file +} diff --git a/.agent/demos/attachment-acquisition-census/regenerate.sh b/.agent/demos/attachment-acquisition-census/regenerate.sh index 7dc12b7afa..3f82c04d3b 100755 --- a/.agent/demos/attachment-acquisition-census/regenerate.sh +++ b/.agent/demos/attachment-acquisition-census/regenerate.sh @@ -156,7 +156,7 @@ payload = { }, } -(demo_root / "census.json").write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8") +(demo_root / "census.json").write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") lines = [ "# Attachment Acquisition Census", diff --git a/.agent/demos/basic-usage/01-find-query.txt b/.agent/demos/basic-usage/01-find-query.txt deleted file mode 100644 index da18f16a1e..0000000000 --- a/.agent/demos/basic-usage/01-find-query.txt +++ /dev/null @@ -1,6 +0,0 @@ -$ polylogue --origin codex-session find "sessions where origin:codex-session" then select --format json -[{"id":"codex-session:demo-receipts","origin":"codex-session","title":"Fix the clock-sensitive test and prove the suite passes.","date":"2026-07-04"},{"id":"codex-session:demo-anti-grep","origin":"codex-session","title":"Explain error budgets without running a command.","date":"2026-07-04"},{"id":"codex-session:demo-terminal-error","origin":"codex-session","title":"Run the command and stop if it fails.","date":"2026-07-04"},{"id":"codex-session:demo-lineage-subagent","origin":"codex-session","title":"demo-lineage-subagent","date":"2026-07-04"},{"id":"codex-session:demo-lineage-fork","origin":"codex-session","title":"Map the demo lineage base context.","date":"2026-07-04"},{"id":"codex-session:demo-lineage-parent","origin":"codex-session","title":"Map the demo lineage base context.","date":"2026-07-04"},{"id":"codex-session:demo-00","origin":"codex-session","title":"Could you review this code for potential issues?","date":"2024-01-23"}] - -$ polylogue find "actions where tool:bash | group by origin | count" -session.origin=claude-code-session count=6 -session.origin=codex-session count=3 diff --git a/.agent/demos/basic-usage/02-read.txt b/.agent/demos/basic-usage/02-read.txt deleted file mode 100644 index 2d7882e81c..0000000000 --- a/.agent/demos/basic-usage/02-read.txt +++ /dev/null @@ -1,34 +0,0 @@ -$ polylogue find id:codex-session:demo-receipts then read --view transcript -# Fix the clock-sensitive test and prove the suite passes. - -`codex-session:demo-receipts` - -## user -Fix the clock-sensitive test and prove the suite passes. - -## assistant - - -## tool -{"metadata": {"exit_code": 1}, "output": "F tests/test_clock.py::test_uses_monotonic_clock\n1 failed in 0.18s"} - -## assistant -All tests pass. The clock fix is complete. - -## user -The receipt disagrees. Correct the fixture and verify again. - -## assistant - - -## tool -{"metadata": {"exit_code": 0}, "output": "Done!"} - -## assistant - - -## tool -{"metadata": {"exit_code": 0}, "output": ". 1 passed in 0.16s"} - -## assistant -Verified after the correction: 1 passed in 0.16s. diff --git a/.agent/demos/basic-usage/03-search.txt b/.agent/demos/basic-usage/03-search.txt deleted file mode 100644 index dc6f645d3b..0000000000 --- a/.agent/demos/basic-usage/03-search.txt +++ /dev/null @@ -1,2 +0,0 @@ -$ polylogue find "clock" then select --format json -[{"id":"chatgpt-export:cross-material-duplicate-01","origin":"chatgpt-export","title":"Flaky clock test fix summary","date":"2026-07-04"},{"id":"chatgpt-export:cross-material-duplicate-02","origin":"chatgpt-export","title":"Flaky clock test fix summary","date":"2026-07-04"},{"id":"claude-code-session:demo-lineage-compaction-parent","origin":"claude-code-session","title":"Fix the flaky clock test before continuing the demo lineage audit.","date":"2026-07-04"},{"id":"codex-session:demo-receipts","origin":"codex-session","title":"Fix the clock-sensitive test and prove the suite passes.","date":"2026-07-04"}] diff --git a/.agent/demos/basic-usage/04-resume.txt b/.agent/demos/basic-usage/04-resume.txt deleted file mode 100644 index 89e92670ac..0000000000 --- a/.agent/demos/basic-usage/04-resume.txt +++ /dev/null @@ -1,2 +0,0 @@ -$ polylogue find id:codex-session:demo-receipts then continue -codex resume demo-receipts diff --git a/.agent/demos/basic-usage/05-cost-usage.json b/.agent/demos/basic-usage/05-cost-usage.json deleted file mode 100644 index e2474ecf41..0000000000 --- a/.agent/demos/basic-usage/05-cost-usage.json +++ /dev/null @@ -1,1380 +0,0 @@ -$ polylogue analyze usage --format json (excerpt: logical_pricing_lanes) - - "archive_root": "/realm/tmp/polylogue-demo-refresh-1784526507", - "detail_level": "full", - "coverage_matrix": [ - { - "origin": "claude-code-session", - "provider": "claude-code", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "message_usage" - ], - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "notes": [ - "Exact token telemetry is available only where exported records include message.usage." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "codex-session", - "provider": "codex", - "status": "exact", - "evidence_stream": "provider_reported_usage", - "event_types": [ - "token_count" - ], - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "notes": [ - "model_context_window is carried on token_count events when the provider supplies it." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "chatgpt-export", - "provider": "chatgpt", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "ChatGPT exports do not carry reliable per-request token counters.", - "cumulative_semantics": "Provider/account UI totals are external summaries and are not reconstructed from transcript text.", - "cache_semantics": "No cache read/write token lanes are available in ChatGPT export rows.", - "notes": [ - "Cost-looking metadata is not treated as exact token telemetry." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "claude-ai-export", - "provider": "claude-ai", - "status": "estimate_only", - "evidence_stream": "transcript_text_estimate", - "event_types": [], - "request_semantics": "Claude.ai exports preserve transcript content, not provider usage counters.", - "cumulative_semantics": "No cumulative provider usage window is present in the export shape.", - "cache_semantics": "No cache read/write token lanes are available in Claude.ai export rows.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "aistudio-drive", - "provider": "gemini", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "AI Studio/Gemini exports may carry message-level tokenCount output counters on some records.", - "cumulative_semantics": "No provider cumulative session usage window is available from Drive prompt exports.", - "cache_semantics": "No cache read/write token lanes are available from Drive prompt exports.", - "notes": [ - "Input tokens and cache semantics are missing unless a future export shape supplies them explicitly." - ], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "gemini-cli-session", - "provider": "gemini-cli", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Local Gemini CLI documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "hermes-session", - "provider": "hermes", - "status": "partial", - "evidence_stream": "message_token_fields", - "event_types": [], - "request_semantics": "Hermes local-agent documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "antigravity-session", - "provider": "antigravity", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "No provider usage telemetry parser is implemented for this origin.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - }, - { - "origin": "unknown-export", - "provider": "unknown", - "status": "unsupported", - "evidence_stream": "transcript_text_only", - "event_types": [], - "request_semantics": "Unknown exports are parsed for transcript content only.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "notes": [], - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence" - } - ], - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 56256, - "output_tokens": 10596, - "cached_input_tokens": 280020, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388872 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 56256, - "output_tokens": 10596, - "cached_input_tokens": 280020, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388872 - }, - "pricing_catalog_provenance": "litellm-model-prices-vendored+polylogue-curated-overrides", - "pricing_catalog_effective_date": "2026-06-27", - "pricing_lanes": [ - { - "provenance": "priced", - "row_count": 6, - "session_count": 6, - "matched_model_row_count": 3, - "unmatched_model_row_count": 3, - "usage": { - "input_tokens": 56256, - "output_tokens": 10596, - "cached_input_tokens": 280020, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388872 - }, - "stored_cost_usd": 2.835, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 2.835, - "subscription_credit_usd": 0.0, - "grain": "physical_session", - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 6, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:physical_session:priced", - "reason": "unpriced-model-rows:3" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [ - "missing_price" - ] - } - ], - "pricing_grain": "physical_session", - "logical_pricing_lanes": [ - { - "provenance": "priced", - "row_count": 6, - "session_count": 6, - "matched_model_row_count": 3, - "unmatched_model_row_count": 3, - "usage": { - "input_tokens": 56256, - "output_tokens": 10596, - "cached_input_tokens": 280020, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388872 - }, - "stored_cost_usd": 0.0, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 2.835, - "subscription_credit_usd": 0.0, - "grain": "logical_session_model_high_water", - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 6, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:priced", - "reason": "unpriced-model-rows:3" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [], - "conflicts": [] - }, - "caveats": [ - "missing_price" - ] - } - ], - "logical_pricing_grain": "logical_session_model_high_water", - "stored_provider_priced_usd": 2.835, - "catalog_api_equivalent_usd": null, - "catalog_priced_subtotal_usd": 2.835, - "logical_catalog_api_equivalent_usd": null, - "logical_catalog_priced_subtotal_usd": 2.835, - "subscription_credit_catalog_provenance": "polylogue-curated-claude-code-subscription-v1", - "subscription_credit_catalog_effective_date": "2026-05-07", - "subscription_credit_usd": 0.0, - "logical_subscription_credit_usd": 0.0, - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:physical-exact-total-tokens:total", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:physical_session:priced:exact-total-tokens" - ], - "observed_refs": [ - "insight:session-model-usage:physical_session:priced:exact-total-tokens" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:physical_session:priced:exact-total-tokens", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported" - ], - "evidence_refs": [ - "insight:session-model-usage:physical_session:priced" - ], - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 6, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:physical-catalog-api-equivalent-cost:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 0, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost" - ], - "observed_refs": [ - "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost" - ], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:physical_session:priced", - "reason": "unpriced-model-rows:3" - }, - { - "subject_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "reason": "contribution-coverage-incomplete" - }, - { - "subject_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "reason": "unsupported-contribution:unknown" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:physical_session:priced:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:physical_session:priced" - ], - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "physical_session session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:physical_session:priced", - "reason": "unpriced-model-rows:3" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "logical_exact_total_tokens_evidence": { - "family": "usage.pricing_lane_exact_total_tokens", - "fact_ref": "insight:provider-usage:logical-exact-total-tokens:total", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported" - ], - "weakest_measurement_authority": "provider-reported", - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-exact-total-tokens:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 1, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens" - ], - "observed_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens" - ], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:exact-total-tokens", - "value_state": "known", - "value": 388872, - "measurement_authority": [ - "provider-reported" - ], - "evidence_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 6, - "complete": true, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "logical_catalog_api_equivalent_evidence": { - "family": "usage.pricing_lane_catalog_api_equivalent_cost", - "fact_ref": "insight:provider-usage:logical-catalog-api-equivalent-cost:total", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "weakest_measurement_authority": "catalog-derived", - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "definition_ref": "insight:usage-lane-catalog-api-equivalent-cost:v1", - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical provider usage pricing lanes", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 1, - "observed_count": 1, - "supported_count": 0, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost" - ], - "observed_refs": [ - "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost" - ], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:priced", - "reason": "unpriced-model-rows:3" - }, - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "reason": "contribution-coverage-incomplete" - }, - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "reason": "unsupported-contribution:unknown" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - }, - "calibrated_confidence": null, - "contributions": [ - { - "fact_ref": "insight:session-model-usage:logical_session_model_high_water:priced:catalog-api-equivalent-cost", - "value_state": "unknown", - "value": null, - "measurement_authority": [ - "catalog-derived" - ], - "evidence_refs": [ - "insight:pricing-catalog:2026-06-27", - "insight:session-model-usage:logical_session_model_high_water:priced" - ], - "temporal": { - "observed_at": "2026-07-20T05:56:19.734297+00:00", - "time_source": "materialization_ts", - "time_confidence": "unknown", - "frame_start": null, - "frame_end": null - }, - "enumeration": "census", - "coverage": { - "intended_frame": "logical_session_model_high_water session_model_usage rows for 'priced'", - "grain": "pricing_lane", - "denominator": "session_model_usage rows in the declared lane", - "intended_count": 6, - "observed_count": 6, - "supported_count": 3, - "complete": false, - "coverage_ratio": 1.0, - "intended_refs": [], - "observed_refs": [], - "exclusions": [ - { - "subject_ref": "insight:session-model-usage:logical_session_model_high_water:priced", - "reason": "unpriced-model-rows:3" - } - ] - }, - "freshness": { - "state": "fresh", - "evaluated_at": "2026-07-20T05:56:19.734297+00:00", - "cause": null, - "last_good_at": null, - "last_good_evidence_refs": [] - } - } - ], - "conflicts": [] - }, - "origins": [ - { - "origin": "aistudio-drive", - "detail_level": "full", - "provider": "gemini", - "declared_coverage": "partial", - "coverage_state": "partial_telemetry_unobserved", - "coverage_basis": "this origin can carry partial message token fields, but none are materialized in model rollups", - "evidence_stream": "message_token_fields", - "request_semantics": "AI Studio/Gemini exports may carry message-level tokenCount output counters on some records.", - "cumulative_semantics": "No provider cumulative session usage window is available from Drive prompt exports.", - "cache_semantics": "No cache read/write token lanes are available from Drive prompt exports.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 4, - "transcript_word_count": 25, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 1, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry is partial; request, cumulative, and cache semantics are incomplete" - ] - }, - { - "origin": "antigravity-session", - "detail_level": "full", - "provider": "antigravity", - "declared_coverage": "unsupported", - "coverage_state": "unsupported", - "coverage_basis": "no reliable provider token telemetry is supported for this origin", - "evidence_stream": "transcript_text_only", - "request_semantics": "No provider usage telemetry parser is implemented for this origin.", - "cumulative_semantics": "No cumulative provider usage window is available.", - "cache_semantics": "No cache read/write token lanes are available.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 2, - "transcript_word_count": 29, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry unsupported for this origin" - ] - }, - { - "origin": "chatgpt-export", - "detail_level": "full", - "provider": "chatgpt", - "declared_coverage": "estimate_only", - "coverage_state": "acquired_not_materialized", - "coverage_basis": "source.db has raw rows without parse errors that are not represented in index.db sessions", - "evidence_stream": "transcript_text_estimate", - "request_semantics": "ChatGPT exports do not carry reliable per-request token counters.", - "cumulative_semantics": "Provider/account UI totals are external summaries and are not reconstructed from transcript text.", - "cache_semantics": "No cache read/write token lanes are available in ChatGPT export rows.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 3, - "message_count": 7, - "transcript_word_count": 222, - "raw_session_count": 5, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 2, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 2, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [ - "e0dbd23201249caeabc286e0f8da7c152a9cb399ff2b693a8c63f3e1cc330588", - "fb0151b0cb130e0634ef263bdf7f6a98bd54d24ab36b3409e99be3b997f1a91d" - ], - "sample_stale_rollup_sessions": [], - "caveats": [ - "exact provider telemetry unavailable; transcript text counts are estimate-only", - "raw rows without parse errors are acquired but not materialized; usage coverage is incomplete until index.db is rebuilt" - ] - }, - { - "origin": "claude-ai-export", - "detail_level": "full", - "provider": "claude-ai", - "declared_coverage": "estimate_only", - "coverage_state": "estimate_only", - "coverage_basis": "exact provider token telemetry is unavailable; transcript text counts remain estimate-only evidence", - "evidence_stream": "transcript_text_estimate", - "request_semantics": "Claude.ai exports preserve transcript content, not provider usage counters.", - "cumulative_semantics": "No cumulative provider usage window is present in the export shape.", - "cache_semantics": "No cache read/write token lanes are available in Claude.ai export rows.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 2, - "transcript_word_count": 17, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "exact provider telemetry unavailable; transcript text counts are estimate-only" - ] - }, - { - "origin": "claude-code-session", - "detail_level": "full", - "provider": "claude-code", - "declared_coverage": "exact", - "coverage_state": "missing_provider_telemetry", - "coverage_basis": "this origin supports exact provider telemetry, but no provider usage event rows are materialized", - "evidence_stream": "provider_reported_usage", - "request_semantics": "Claude Code message.usage rows are per message/request observations.", - "cumulative_semantics": "Session rollups are derived by summing message usage rows; Claude Code does not supply a separate cumulative session high-water event.", - "cache_semantics": "cache_read_input_tokens and cache_creation_input_tokens are preserved as cached_input and cache_write lanes and are not folded into generic input/output.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 4, - "message_count": 19, - "transcript_word_count": 398, - "raw_session_count": 4, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 1, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "exact provider telemetry is supported for this origin but no usage events are materialized", - "some sessions have no provider usage event rows; transcript words and model rollups cover different evidence" - ] - }, - { - "origin": "codex-session", - "detail_level": "full", - "provider": "codex", - "declared_coverage": "exact", - "coverage_state": "missing_provider_telemetry", - "coverage_basis": "this origin supports exact provider telemetry, but no provider usage event rows are materialized", - "evidence_stream": "provider_reported_usage", - "request_semantics": "Codex last_token_usage is request/current-window telemetry and can be summed by request when present.", - "cumulative_semantics": "Codex total_token_usage is cumulative and session-global; rollups take the latest total per session to avoid double-counting.", - "cache_semantics": "cached_input_tokens and cache write/cache creation aliases are preserved as separate lanes and are not folded into generic input/output.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 7, - "message_count": 30, - "transcript_word_count": 229, - "raw_session_count": 7, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 0, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "exact provider telemetry is supported for this origin but no usage events are materialized", - "some sessions have no provider usage event rows; transcript words and model rollups cover different evidence" - ] - }, - { - "origin": "gemini-cli-session", - "detail_level": "full", - "provider": "gemini-cli", - "declared_coverage": "partial", - "coverage_state": "partial_provider_telemetry", - "coverage_basis": "message-level token fields are materialized, but exact provider request/cumulative semantics are incomplete", - "evidence_stream": "message_token_fields", - "request_semantics": "Local Gemini CLI documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 2, - "transcript_word_count": 33, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 1, - "provider_event_count": 1, - "token_count_event_count": 0, - "message_usage_event_count": 1, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 1, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 160, - "output_tokens": 64, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 160, - "output_tokens": 64, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 244 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 160, - "output_tokens": 64, - "cached_input_tokens": 20, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 244 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry is partial; request, cumulative, and cache semantics are incomplete" - ] - }, - { - "origin": "hermes-session", - "detail_level": "full", - "provider": "hermes", - "declared_coverage": "partial", - "coverage_state": "partial_provider_telemetry", - "coverage_basis": "message-level token fields are materialized, but exact provider request/cumulative semantics are incomplete", - "evidence_stream": "message_token_fields", - "request_semantics": "Hermes local-agent documents may carry generic usage/tokens dictionaries per message.", - "cumulative_semantics": "No provider cumulative session usage window is available.", - "cache_semantics": "Generic cache_read/cache_write keys are preserved when present, but their provider semantics are not independently verified.", - "rebuild_guidance": "rebuild index.db from source.db/raw archives so usage events and rollups are materialized from source evidence", - "session_count": 1, - "message_count": 5, - "transcript_word_count": 39, - "raw_session_count": 1, - "raw_parse_error_count": 0, - "acquired_not_materialized_count": 0, - "provider_event_session_count": 0, - "provider_event_count": 0, - "token_count_event_count": 0, - "message_usage_event_count": 0, - "zero_token_event_count": 0, - "missing_model_event_count": 0, - "multi_model_session_count": 0, - "priced_model_row_count": 1, - "origin_reported_model_row_count": 0, - "estimated_model_row_count": 0, - "stale_rollup_session_count": 0, - "provider_request_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "provider_cumulative_usage": { - "input_tokens": 0, - "output_tokens": 0, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 0 - }, - "model_rollup_grain": "physical_session", - "model_rollup_usage": { - "input_tokens": 96, - "output_tokens": 32, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 128 - }, - "logical_model_rollup_grain": "logical_session_model_high_water", - "logical_model_rollup_usage": { - "input_tokens": 96, - "output_tokens": 32, - "cached_input_tokens": 0, - "cache_write_tokens": 0, - "reasoning_output_tokens": 0, - "total_tokens": 128 - }, - "sample_missing_model_sessions": [], - "sample_zero_token_sessions": [], - "sample_acquired_not_materialized_raw_ids": [], - "sample_stale_rollup_sessions": [], - "caveats": [ - "provider usage telemetry is partial; request, cumulative, and cache semantics are incomplete" - ] - } - ], - "caveats": [ - "provider usage events, transcript text volume, and model rollups are separate evidence streams", - "this report does not query provider billing and is not a precise cost report", - "catalog_api_equivalent_usd is unknown because at least one model row has no catalog price; catalog_priced_subtotal_usd preserves the known priced subset" - ] -} diff --git a/.agent/demos/basic-usage/06-lineage.txt b/.agent/demos/basic-usage/06-lineage.txt deleted file mode 100644 index 06c487330b..0000000000 --- a/.agent/demos/basic-usage/06-lineage.txt +++ /dev/null @@ -1,31 +0,0 @@ -$ polylogue find id:codex-session:demo-lineage-fork then read --view transcript -# Map the demo lineage base context. - -`codex-session:demo-lineage-fork` - -## user -Map the demo lineage base context. - -## assistant -I have the base context and can branch the analysis. - -## user -Now take the forked branch and audit construct validity. - -## assistant -The fork diverges into demo corpus construct checks. - -$ polylogue find id:codex-session:demo-lineage-parent then read --view transcript -# Map the demo lineage base context. - -`codex-session:demo-lineage-parent` - -## user -Map the demo lineage base context. - -## assistant -I have the base context and can branch the analysis. - -## assistant -Delegating a topology check to a focused subagent. -Subagent completed. Session: codex-session:demo-lineage-subagent diff --git a/.agent/demos/basic-usage/07-mcp-roundtrip.json b/.agent/demos/basic-usage/07-mcp-roundtrip.json deleted file mode 100644 index 1d05b01bcd..0000000000 --- a/.agent/demos/basic-usage/07-mcp-roundtrip.json +++ /dev/null @@ -1,82 +0,0 @@ -{ - "query_call": { - "tool": "query", - "arguments": { - "expression": "clock", - "projection": "sessions", - "limit": 3 - } - }, - "query_first_hit_session_id": "chatgpt-export:cross-material-duplicate-01", - "query_first_hit_snippet": "", - "query_result_hit_count": 3, - "read_call": { - "tool": "read", - "arguments": { - "ref": "session:chatgpt-export:cross-material-duplicate-01" - } - }, - "read_result": { - "mode": "ref-resolution", - "ref": "session:chatgpt-export:cross-material-duplicate-01", - "normalized_ref": "session:chatgpt-export:cross-material-duplicate-01", - "kind": "session", - "resolved": true, - "payload_kind": "session-summary", - "payload": { - "id": "chatgpt-export:cross-material-duplicate-01", - "origin": "chatgpt-export", - "title": "Flaky clock test fix summary", - "message_count": 2, - "target_ref": { - "target_type": "session", - "target_id": "chatgpt-export:cross-material-duplicate-01", - "session_id": "chatgpt-export:cross-material-duplicate-01", - "message_id": null, - "block_index": null, - "identity_key": "session:chatgpt-export:cross-material-duplicate-01" - }, - "anchor": "session-chatgpt-export-cross-material-duplicate-01", - "actions": { - "open": { - "enabled": true, - "state": "enabled", - "disabled_reason": null, - "repair_path": null, - "inspect_path": null - }, - "copy_link": { - "enabled": true, - "state": "enabled", - "disabled_reason": null, - "repair_path": null, - "inspect_path": null - }, - "annotate": { - "enabled": true, - "state": "enabled", - "disabled_reason": null, - "repair_path": null, - "inspect_path": null - } - }, - "created_at": "2026-07-04T09:55:00Z", - "updated_at": "2026-07-04T09:55:03Z" - }, - "title": "Flaky clock test fix summary", - "summary": "2 messages", - "object_refs": [ - "session:chatgpt-export:cross-material-duplicate-01" - ], - "evidence_refs": [], - "caveats": [], - "actions": [ - { - "label": "read", - "command": "polylogue find id:chatgpt-export:cross-material-duplicate-01 then read --format json", - "href": null, - "enabled": true - } - ] - } -} \ No newline at end of file diff --git a/.agent/demos/basic-usage/08-status-health.debt.json b/.agent/demos/basic-usage/08-status-health.debt.json deleted file mode 100644 index e494373ec3..0000000000 --- a/.agent/demos/basic-usage/08-status-health.debt.json +++ /dev/null @@ -1,245 +0,0 @@ -{ - "mode": "archive-debt-list", - "generated_at": "2026-07-18T13:21:51.704665+00:00", - "archive_root": "/tmp/polylogue-basic-usage-demo", - "rows": [ - { - "debt_ref": "debt:embedding:catchup:backlog", - "kind": "embedding", - "stage": "catchup", - "subject_ref": "embedding:pending", - "severity": "warning", - "status": "actionable", - "owner": "daemon", - "summary": "14 session(s) pending embedding catch-up", - "details": "Pending messages: 34; stale messages: 0.", - "evidence_refs": [ - "archive-tier:/tmp/polylogue-basic-usage-demo/embeddings.db" - ], - "caveats": [], - "actions": [ - { - "label": "Run embedding backfill", - "command": [ - "polylogue", - "ops", - "embed", - "backfill" - ] - } - ] - }, - { - "debt_ref": "debt:fts:messages_fts", - "kind": "fts", - "stage": "index-freshness", - "subject_ref": "fts:messages_fts", - "severity": "warning", - "status": "actionable", - "owner": "ops", - "summary": "messages_fts is not query-ready", - "details": "freshness not ready", - "evidence_refs": [ - "archive-tier:/tmp/polylogue-basic-usage-demo/index.db" - ], - "caveats": [], - "actions": [ - { - "label": "Run daemon convergence", - "command": [ - "polylogued", - "run" - ] - } - ] - }, - { - "debt_ref": "debt:assertion-candidate:assertion-finding:adb1c19ac49d5b835056e68eb12b9389656b83c6eb01d5f09eb7ad6dd5cf7edd", - "kind": "assertion-candidate", - "stage": "candidate-judgment", - "subject_ref": "assertion:assertion-finding:adb1c19ac49d5b835056e68eb12b9389656b83c6eb01d5f09eb7ad6dd5cf7edd", - "severity": "info", - "status": "actionable", - "owner": "user", - "summary": "Candidate assertion awaits judgment for analysis:claim-vs-evidence-2026-07-04", - "details": "The bounded origin-stratified sample inspected 3,752 Claude Code, 1,241 Codex, and seven Claude AI structured failures.", - "observed_at": "2023-11-14T22:16:45+00:00", - "evidence_refs": [ - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#calibration", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#definition", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#sample_frame", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#totals", - "file:.agent/demos/claim-vs-evidence/public-summary.json", - "file:docs/findings/claim-vs-evidence.md", - "assertion:assertion-finding:adb1c19ac49d5b835056e68eb12b9389656b83c6eb01d5f09eb7ad6dd5cf7edd" - ], - "caveats": [], - "actions": [ - { - "label": "Review candidate assertions", - "command": [ - "polylogue", - "mark", - "candidates", - "list", - "--format", - "json" - ] - } - ] - }, - { - "debt_ref": "debt:assertion-candidate:assertion-finding:cbb4967fc2d33ce231978b76949205856a68d8306da46a610f478cb4244d2ad8", - "kind": "assertion-candidate", - "stage": "candidate-judgment", - "subject_ref": "assertion:assertion-finding:cbb4967fc2d33ce231978b76949205856a68d8306da46a610f478cb4244d2ad8", - "severity": "info", - "status": "actionable", - "owner": "user", - "summary": "Candidate assertion awaits judgment for analysis:claim-vs-evidence-2026-07-04", - "details": "The same sample's silent-continuation lower bounds were 22.3% for consequential handlers, 27.1% for benign-recovery handlers, and 53.9% for other handlers.", - "observed_at": "2023-11-14T22:16:45+00:00", - "evidence_refs": [ - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#calibration", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#definition", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#sample_frame", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#totals", - "file:.agent/demos/claim-vs-evidence/public-summary.json", - "file:docs/findings/claim-vs-evidence.md", - "assertion:assertion-finding:cbb4967fc2d33ce231978b76949205856a68d8306da46a610f478cb4244d2ad8" - ], - "caveats": [], - "actions": [ - { - "label": "Review candidate assertions", - "command": [ - "polylogue", - "mark", - "candidates", - "list", - "--format", - "json" - ] - } - ] - }, - { - "debt_ref": "debt:assertion-candidate:assertion-finding:e5da9c88d1c236d15648fc52ad7898d8a6f7018277e84c624bdc9e7c7d037689", - "kind": "assertion-candidate", - "stage": "candidate-judgment", - "subject_ref": "assertion:assertion-finding:e5da9c88d1c236d15648fc52ad7898d8a6f7018277e84c624bdc9e7c7d037689", - "severity": "info", - "status": "actionable", - "owner": "user", - "summary": "Candidate assertion awaits judgment for analysis:claim-vs-evidence-2026-07-04", - "details": "In one bounded private-archive sample, 1,205 of 5,000 inspected structured failures were followed by silent continuation on the next assistant turn, a 24.1% lower bound.", - "observed_at": "2023-11-14T22:16:45+00:00", - "evidence_refs": [ - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#calibration", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#definition", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#sample_frame", - "file:.agent/demos/claim-vs-evidence/claim-vs-evidence.report.json#totals", - "file:.agent/demos/claim-vs-evidence/public-summary.json", - "file:docs/findings/claim-vs-evidence.md", - "assertion:assertion-finding:e5da9c88d1c236d15648fc52ad7898d8a6f7018277e84c624bdc9e7c7d037689" - ], - "caveats": [], - "actions": [ - { - "label": "Review candidate assertions", - "command": [ - "polylogue", - "mark", - "candidates", - "list", - "--format", - "json" - ] - } - ] - }, - { - "debt_ref": "debt:assertion-candidate:assertion-note:3be837fd2faa70afd6139351ee2868b033179a1aa6fbd01daed8711f7a5ea72c", - "kind": "assertion-candidate", - "stage": "candidate-judgment", - "subject_ref": "assertion:assertion-note:3be837fd2faa70afd6139351ee2868b033179a1aa6fbd01daed8711f7a5ea72c", - "severity": "info", - "status": "actionable", - "owner": "user", - "summary": "Candidate assertion awaits judgment for session:claude-code-session:63705dcc-f3e5-4378-8118-8bc21e53bbb6", - "details": "The demo pytest session is the approved fixture for search, read, and assertion overlays.", - "observed_at": "2023-11-14T22:16:41+00:00", - "evidence_refs": [ - "session:claude-code-session:63705dcc-f3e5-4378-8118-8bc21e53bbb6", - "assertion:assertion-note:3be837fd2faa70afd6139351ee2868b033179a1aa6fbd01daed8711f7a5ea72c" - ], - "caveats": [], - "actions": [ - { - "label": "Review candidate assertions", - "command": [ - "polylogue", - "mark", - "candidates", - "list", - "--format", - "json" - ] - } - ] - }, - { - "debt_ref": "debt:assertion-candidate:demo-assertion:decision:pytest-triage", - "kind": "assertion-candidate", - "stage": "candidate-judgment", - "subject_ref": "assertion:demo-assertion:decision:pytest-triage", - "severity": "info", - "status": "actionable", - "owner": "user", - "summary": "Candidate assertion awaits judgment for session:claude-code-session:63705dcc-f3e5-4378-8118-8bc21e53bbb6", - "details": "Use the Claude Code demo session as the stable pytest triage example.", - "observed_at": "2023-11-14T22:16:44+00:00", - "evidence_refs": [ - "session:claude-code-session:63705dcc-f3e5-4378-8118-8bc21e53bbb6", - "assertion:demo-assertion:decision:pytest-triage" - ], - "caveats": [], - "actions": [ - { - "label": "Review candidate assertions", - "command": [ - "polylogue", - "mark", - "candidates", - "list", - "--format", - "json" - ] - } - ] - } - ], - "totals": { - "total": 7, - "critical": 0, - "warning": 2, - "info": 5, - "actionable": 7, - "blocked": 0, - "classified": 0, - "affected_total": 0, - "affected_critical": 0, - "affected_warning": 0, - "affected_info": 0, - "affected_actionable": 0, - "affected_blocked": 0, - "affected_open": 0, - "affected_classified": 0 - }, - "caveats": [ - "Rows are composed from existing readiness projections; exact FTS reconciliation requires --exact-fts." - ] -} diff --git a/.agent/demos/basic-usage/08-status-health.txt b/.agent/demos/basic-usage/08-status-health.txt deleted file mode 100644 index 62f6b431ea..0000000000 --- a/.agent/demos/basic-usage/08-status-health.txt +++ /dev/null @@ -1,22 +0,0 @@ -$ polylogue status --daemon-url http://127.0.0.1:1 # forces the direct-archive fallback path a reader without a running daemon would see - -Archive (daemon not running) - Database: index.db - Schema tiers: present=source, index, embeddings, user, ops - Archive tier detail: source v13/13 ok, raw_sessions=21; index v42/42 ok, sessions=19; embeddings v4/4 ok, embedding_status=1; user v10/10 ok; ops v1/1 ok, ingest_attempts=0 - SQLite maintenance: WAL 0 KB, planner stats=source,index,embeddings,user - Archive readiness: unchecked (direct_status_default_skips_exact_archive_readiness) - Raw materialization: 2 debt row(s), 0 critical, 0 warning, 0 blocked - Raw frontier: healthy - Facade routes: 139/140 archive-ready, 0 unsupported - CLI routes: 5/5 archive-ready, 0 unsupported - Archive routing paths: ingest=archive -> source.db,index.db, rebuild_index=index_db; 0 blockers, tiers=source.db,index.db,embeddings.db,user.db,ops.db - Archive route ownership: facade=index:87,none:1,source:5,user:47; cli=source:2,user:3 - Assertion candidate queue: stale-pending, 5 pending, oldest=978.3d - Sessions: 19 - Messages: 71 - Raw records: 21 - FTS indexed: daemon status unavailable - Embeddings: partial/partial, ready; 2 msgs, 1/19 convs (5.9%), 16 pending convs - - Run polylogued run to start the daemon. diff --git a/.agent/demos/basic-usage/COLD_READER_GATE.md b/.agent/demos/basic-usage/COLD_READER_GATE.md deleted file mode 100644 index 3adc50652b..0000000000 --- a/.agent/demos/basic-usage/COLD_READER_GATE.md +++ /dev/null @@ -1,50 +0,0 @@ -# Cold-Reader Gate - -Give a fresh reader only this directory and ask: - -```text -Using only the files in this directory, list the eight Polylogue features -this demo claims to prove work end-to-end, state which surface (CLI, MCP) -each was exercised through, and identify the one walkthrough whose output -was deliberately captured against a non-default daemon URL — and why. -``` - -## Expected Passing Answer - -- Names all eight walkthroughs: find (fielded query + pipeline aggregate), - read (exact-ref transcript), search (FTS with provenance), resume - (continuation command generation), cost (disjoint token/cost lanes), - lineage (composed fork read), MCP (search → get_session_summary round-trip), - status/health (archive readiness without a daemon). -- States that seven walkthroughs use the `polylogue` CLI directly through the - real subprocess test route and one (MCP) uses a real stdio JSON-RPC - client/server exchange, not the CLI. -- Identifies walkthrough 8 (status/health) as the one run with - `--daemon-url http://127.0.0.1:1`, and explains this forces the - direct-archive fallback path so the captured output reproduces - identically for a reader with no daemon running at all. -- Notices the demo runs against the deterministic seeded archive - (`polylogue demo seed`), not the operator's live archive, and states why - (public repository; the README's own "Why the demo archive" section). -- Does not treat any of the eight numeric results (token counts, hit counts, - session counts) as a claim about a real corpus — they are properties of - the fixed deterministic demo archive. - -## Replacement Verification Mapping - -| Former check | Replacement | Production route exercised | -| --- | --- | --- | -| find query | `test_find_query_covers_fielded_filter_and_pipeline_aggregate` | Query-first CLI parser, SQL query plan, and select renderer | -| read | `test_read_renders_the_seeded_transcript` | Query-first CLI read/transcript renderer | -| search | `test_search_spans_multiple_origins` | CLI lexical search and result serialization | -| resume | `test_continue_generates_a_resume_command` | CLI continuation action | -| cost | `test_usage_reports_disjoint_token_lanes` | CLI usage insight and JSON renderer | -| lineage | `test_lineage_read_composes_parent_prefix_and_child_tail` | CLI read path and lineage recomposition | -| MCP round-trip | `test_mcp_query_and_get_round_trip` | Production stdio MCP query/get dispatcher | -| status/health | `test_status_reports_direct_archive_fallback_when_daemon_is_unreachable` | CLI direct-archive status fallback | - -The replacement test seeds through `polylogue.demo.seed_demo_archive`, runs the -CLI through `tests.infra.cli_subprocess.run_cli`, and invokes the production -MCP stdio route. `polylogue demo verify --require-overlays` remains the -structural demo contract check. The captured outputs remain private-data-free -fixtures for cold reading; they are not a second QA command surface. diff --git a/.agent/demos/basic-usage/README.md b/.agent/demos/basic-usage/README.md deleted file mode 100644 index f2c586d80e..0000000000 --- a/.agent/demos/basic-usage/README.md +++ /dev/null @@ -1,241 +0,0 @@ -# Basic Usage — The Features Actually Work - -Generated: 2026-07-20 -Archive: Polylogue's deterministic demo archive (`polylogue demo seed`), index schema v42 at capture time. - -Eight short, real walkthroughs proving ordinary Polylogue features work -end-to-end, not just in unit tests. Every command below was actually run; every -output is the real result, copied verbatim (only line-wrapping added for -readability). Each walkthrough states exactly what it proves and nothing more. - -**Why the demo archive, not the live one:** every command here is real and -copy-pasteable, but it runs against the private-data-free deterministic demo -archive rather than the operator's live archive, for two reasons: (1) this -repository is public — the live archive is the operator's actual work -history and its content must not appear here; (2) at the time this suite was -built, the live archive's derived index was in a known, unrelated degraded -state (see `docs/findings/claim-vs-evidence.md`) that would have made any -live numbers non-representative anyway. The same commands work identically -against a real archive — only the data differs. - -Reproduce the whole archive first: - -```bash -export POLYLOGUE_ARCHIVE_ROOT=/tmp/polylogue-basic-usage-demo -polylogue demo seed --root "$POLYLOGUE_ARCHIVE_ROOT" --force --with-overlays --format json -polylogue demo verify --root "$POLYLOGUE_ARCHIVE_ROOT" --require-overlays --format json -``` - -Then every command below runs unmodified with that `POLYLOGUE_ARCHIVE_ROOT` exported. - -## 1. Find — fielded query + pipeline aggregate - -File: [`01-find-query.txt`](01-find-query.txt) - -``` -polylogue --origin codex-session find "sessions where origin:codex-session" then select --format json -polylogue find "actions where tool:bash | group by origin | count" -``` - -The first command is a fielded DSL query (`origin:codex-session`) returning -the matched session refs. The second is a pipeline query: it scopes to -`actions` (not `sessions`), filters by tool name, groups by origin, and -aggregates — the DSL's `sessions where … | s where … | group by … | -count` grammar (`archive/query/expression.py`), not a bespoke script. - -**What this proves:** the query-first CLI's fielded predicates and multi-stage -pipeline grammar both execute against a real archive and return real, -distinct result shapes (a JSON row list vs. a grouped count table). - -## 2. Read — exact-ref transcript - -File: [`02-read.txt`](02-read.txt) - -``` -polylogue find id:codex-session:demo-receipts then read --view transcript -``` - -An exact-ref query (`id:`) piped into `read --view transcript`. The -transcript itself is worth reading closely: the assistant claims "All tests -pass" immediately after a tool result shows `exit_code: 1` and a failing -test — a small, self-contained illustration of exactly the claim-vs-evidence -gap this lane's Phase 0/1 work investigates, captured as normal archive -content rather than manufactured for the point. - -**What this proves:** exact-ref resolution and the `transcript` render view -work end-to-end, and the resulting text is faithful to the underlying -structured tool-result evidence (not reconstructed from prose). - -## 3. Search — FTS hit with snippet and provenance - -File: [`03-search.txt`](03-search.txt) - -``` -polylogue find "clock" then select --format json -``` - -An unfielded, quoted free-text query runs through the same FTS5 index this -CLI's `find` verb is backed by (`archive/query/expression.py`'s `near:`/bare -free-text lowering). Every hit carries a stable session ref -(`origin:native_id`), so a caller can always resolve a search hit back to its -exact source. (The MCP `search` tool, demonstrated separately in section 7, -additionally returns a highlighted snippet and message-level ref per hit — -the CLI's `find` surface intentionally keeps this compact for terminal use.) - -**What this proves:** free-text search resolves across origins (a ChatGPT -export, a Claude Code session, and a Codex session all matched "clock" from -genuinely different underlying text) and every result is provenance-bearing, -not a bag of disconnected snippets. - -## 4. Resume — continuation command generation - -File: [`04-resume.txt`](04-resume.txt) - -``` -polylogue find id:codex-session:demo-receipts then continue -``` - -**What this proves:** Polylogue can generate the exact runtime-native resume -invocation (`codex resume `) for an archived session, not just read it — -continuity, not just archaeology. - -## 5. Cost — disjoint token/cost accounting - -File: [`05-cost-usage.json`](05-cost-usage.json) - -``` -polylogue analyze usage --format json -``` - -The full output is long (it deliberately keeps provider-event rows, origin -cumulative counters, and per-model rollups as separate evidence streams -rather than collapsing them); the load-bearing excerpt is -`logical_pricing_lanes`, which reports, per pricing provenance: - -```json -{ - "provenance": "priced", - "row_count": 4, - "usage": { - "input_tokens": 56000, - "output_tokens": 10500, - "cached_input_tokens": 280000, - "cache_write_tokens": 42000, - "reasoning_output_tokens": 0, - "total_tokens": 388500 - }, - "catalog_priced_subtotal_usd": 2.835 -} -``` - -**What this proves:** cost/usage accounting keeps input, output, cached, and -cache-write token lanes disjoint (never silently folded together, matching -the documented Codex-inclusive-token/Claude-cache-token pitfalls) and -reports pricing provenance (`priced` vs. `estimate_only`) per row rather than -one blended number. - -## 6. Lineage — composed fork read - -File: [`06-lineage.txt`](06-lineage.txt) - -``` -polylogue find id:codex-session:demo-lineage-fork then read --view transcript -polylogue find id:codex-session:demo-lineage-parent then read --view transcript -``` - -The fork session's transcript already contains the parent's two turns -("Map the demo lineage base context." / "I have the base context and can -branch the analysis.") composed ahead of its own divergent tail — reading -the fork alone gives the full coherent conversation, not just the child's -new turns. `demo-lineage-parent`'s own transcript is included for direct -comparison: the fork's first two turns are byte-identical to the parent's, -proving genuine prefix composition rather than duplicated storage that -happens to render similarly. - -**What this proves:** forked sessions are read as parent-prefix + child-tail -composition (the archive's `session_links` lineage model), not stored or -displayed as disconnected fragments. - -## 7. MCP — search → get_session_summary round-trip - -File: [`07-mcp-roundtrip.json`](07-mcp-roundtrip.json) - -A real MCP client/server exchange over stdio JSON-RPC (the same protocol any -MCP-speaking agent client uses), driven by -`devtools/continuity_replay.py`'s `StdioMCPContinuityRoute` against the demo -archive: - -```python -async with StdioMCPContinuityRoute(archive_root) as route: - search_result = await route.invoke("search", {"query": "clock", "limit": 3}) - # search_result["hits"][0]["session"]["id"] == "chatgpt-export:cross-material-duplicate-01" - summary = await route.invoke("get_session_summary", {"id": "chatgpt-export:cross-material-duplicate-01"}) -``` - -The MCP surface is mid-rewrite by a parallel lane (#3056 retired the -`archive_list_sessions`/`archive_search_sessions` aliases the same week this -suite was built), so this demo deliberately exercises the two canonical -tools least likely to move — `search` and `get_session_summary` — rather -than the full tool catalog. See `docs/mcp-reference.md` for the current -complete tool list. - -**What this proves:** the MCP surface answers the identical question the CLI -answers (search → resolve a session summary) through a real protocol -round-trip, not a mocked handler — this is the continuity surface real agent -clients (not humans at a terminal) actually use. - -## 8. Status/health — daemon and archive readiness - -Files: [`08-status-health.txt`](08-status-health.txt), -[`08-status-health.debt.json`](08-status-health.debt.json) - -``` -polylogue status --daemon-url http://127.0.0.1:1 -polylogue ops debt list --format json -``` - -(`--daemon-url` points at an unreachable port so this demo reproduces -identically for a reader with no daemon running at all — `polylogue status` -normally talks to a live `polylogued` daemon's HTTP API first and only falls -back to bounded local SQLite checks when the daemon is unreachable.) - -`ops debt list` names concrete, actionable freshness/convergence gaps rather -than a pass/fail flag — in this demo archive: 14 sessions pending embedding -catch-up, `messages_fts` not yet query-ready, and (because this suite seeds -`--with-overlays`) a candidate finding assertion from the claim-vs-evidence -demo overlay still awaiting operator judgment, each with its own evidence -ref and a copy-pasteable remediation command. - -**What this proves:** archive health is queryable without a running daemon — -per-tier schema versions, row counts, raw-materialization debt, and FTS/ -embedding backfill progress are all real, structural facts read directly -from the five SQLite tiers, not decorative placeholder text — and debt is -reported as named, actionable rows, not a single opaque health flag. - -## Regenerate everything - -```bash -export POLYLOGUE_ARCHIVE_ROOT=/tmp/polylogue-basic-usage-demo -polylogue demo seed --root "$POLYLOGUE_ARCHIVE_ROOT" --force --with-overlays --format json -polylogue demo verify --root "$POLYLOGUE_ARCHIVE_ROOT" --require-overlays --format json -devtools test tests/integration/test_basic_usage_cli.py -``` - -The integration tests re-run each CLI walkthrough against a fresh seed through -the real subprocess entry point. `polylogue demo verify` checks the seeded -archive's structural contract. See [`COLD_READER_GATE.md`](COLD_READER_GATE.md) -for the replacement mapping and fresh-reader verification prompt. - -## Files - -- `01-find-query.txt` -- `02-read.txt` -- `03-search.txt` -- `04-resume.txt` -- `05-cost-usage.json` -- `06-lineage.txt` -- `07-mcp-roundtrip.json` -- `08-status-health.txt` -- `08-status-health.debt.json` -- `COLD_READER_GATE.md` -- `README.md` — this file. diff --git a/.agent/demos/d1-receipts/NON-CLAIMS.md b/.agent/demos/d1-receipts/NON-CLAIMS.md deleted file mode 100644 index e52d1257ce..0000000000 --- a/.agent/demos/d1-receipts/NON-CLAIMS.md +++ /dev/null @@ -1,5 +0,0 @@ -# Non-claims - -- This packet does not prove every sentence in PR #3282's body is independently verified -- only the four claims explicitly checked in report.md are scored; claim 4 is explicitly scored `not_supported`. -- This packet does not establish that `session_refs` correctly resolves every PR reference archive-wide -- only that it resolves this one case with structural evidence. -- This packet does not reproduce on the public seed corpus (seed 1843); it requires read-only access to the live archive and the `Sinity/polylogue` GitHub history. diff --git a/.agent/demos/d1-receipts/PROMPT.md b/.agent/demos/d1-receipts/PROMPT.md deleted file mode 100644 index 3e23fb53d6..0000000000 --- a/.agent/demos/d1-receipts/PROMPT.md +++ /dev/null @@ -1,67 +0,0 @@ -# D1 "The Receipts": Claim-vs-Evidence on a Real Merged PR - -Predeclaration receipt: `artifact:d1-receipts-predeclaration`. - -Pick a real merged, agent-authored PR from this repository. Resolve it to -its authoring/dispatch session **structurally** — via `session_refs` -(kind=`pull_request`), not by regex-scanning message prose or a time-window -heuristic. Then check specific sentences from the PR body against that -session's own recorded tool_use/tool_result blocks: does the evidence -actually support the claim, or is the claim resting on the PR body's own -prose with nothing underneath it? - -Product primitives only: `session_refs` (the typed evidence table wired by -PR #3425/#3431), `polylogue read --view correlation`, and structural SQL -reads over `blocks`/`session_refs` for citation (mirroring the exact -read-only query style PR #3392 and PR #3282 themselves used in their own -Verification sections — this demo does not invent a new access pattern). - -## Steps - -1. Resolve PR → session structurally: - ```sql - SELECT session_id, repo, ref_number, url - FROM session_refs - WHERE kind = 'pull_request' AND repo = 'Sinity/polylogue' AND ref_number = 3282; - ``` - Cross-check the same resolution through the CLI's own read surface: - `polylogue find "id:" then read --view correlation --format json` - (this is the surface PR #3425/#3431 wired `session_refs` into — - `insights/session_commit.py:build_correlation_result` and - `insights/correlation_view.py`). - -2. Fetch the PR body from GitHub (`gh pr view 3282 --json body`) and pull - out individually falsifiable sentences — not the whole prose block, each - claim on its own. - -3. For each claim, search the resolved session's own `blocks` rows - (`tool_use`/`tool_result`, joined by `tool_id`) for structural evidence: - an exact command, an exact exit code, an exact pytest summary line. A - claim with no matching block is marked **not independently verified in - this session** — never silently upgraded to "supported" because the PR - body asserts it. - -4. Render the two columns: claimed sentence | observed block evidence - (drillable via the cited `block:` ref), with an explicit status per row. - -## Note on this run - -This demo's session turned out to be a **merge-conductor** session: its own -`blocks` are almost entirely `Bash` (53 of 56 tool_use blocks) plus 3 `Read` -calls — zero `Edit`/`Write` tool_use. The actual file edits for PR #3282 -happened in separately dispatched worker sessions across multiple git -worktrees (`/realm/worktrees/polylogue-membership-head*`); this session -orchestrates `git`, `gh pr create`, and `devtools test`/`devtools verify` -invocations across those worktrees and stitches the result into one PR. - -This is itself a real, useful finding, not a inconvenience to hide: the PR -body's own Verification section names a 7-file `devtools test` invocation -("`devtools test tests/unit/sources/test_live_batch_support.py ...` — all -passing, **see individual commit messages for per-commit pass counts**") — -its own parenthetical admits the aggregate command was never run as one -shot. Searching this session's blocks confirms it: the 7-file string only -appears inside the `gh pr create --body` tool_input (i.e. inside the PR body -text itself), never as an actual invoked command. That specific claim is -marked **not independently verified in this session** in `report.md` and -`checks.json` — precisely the honesty discipline this packet exists to -enforce, applied to itself. diff --git a/.agent/demos/d1-receipts/checks.json b/.agent/demos/d1-receipts/checks.json deleted file mode 100644 index 08d536a977..0000000000 --- a/.agent/demos/d1-receipts/checks.json +++ /dev/null @@ -1,7 +0,0 @@ -{ - "pass": true, - "unsupported_claims": [ - "PR #3282 claim 4: the 7-file devtools test invocation named in the Verification section is not independently verified in the resolved session -- no matching tool_use block exists outside the gh-pr-create body text itself." - ], - "coverage_notes": "4 claims from PR #3282's body were checked structurally against the session_refs-resolved authoring/dispatch session's own blocks. 3 of 4 are supported by exact tool_use/tool_result evidence (gh pr create body/URL match, devtools verify --quick per-step exit codes, the rebuild_index.py test+commit pair). Claim 4 is explicitly and correctly scored not_supported rather than assumed true from the PR body's own prose -- this is the intended outcome of the packet's method, not a defect. The resolved session is also shown to be a merge-conductor (53 Bash + 3 Read tool_use blocks, 0 Edit/Write) rather than the direct file-editing session, which is documented as a real finding in report.md rather than hidden." -} diff --git a/.agent/demos/d1-receipts/evidence.ndjson b/.agent/demos/d1-receipts/evidence.ndjson deleted file mode 100644 index 5035c3869b..0000000000 --- a/.agent/demos/d1-receipts/evidence.ndjson +++ /dev/null @@ -1,11 +0,0 @@ -{"ref": "artifact:d1-receipts-evidence", "cited_for": "Demo Packet v2 receipt root", "verified_via": "committed evidence.ndjson"} -{"ref": "session:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39", "cited_for": "session_refs kind=pull_request resolves this session to Sinity/polylogue#3282", "verified_via": "sqlite3 index.db: SELECT session_id,repo,ref_number,url FROM session_refs WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282"} -{"ref": "block:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39:ae9a4788-4bf6-4b89-b01e-90f10e622981:0", "cited_for": "the gh pr create tool_use --body text is byte-identical to the PR body later fetched live via gh pr view 3282, including the exact bullet claims checked below", "verified_via": "diff of tool_input command against `gh pr view 3282 --repo Sinity/polylogue --json body`"} -{"ref": "block:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39:a98d94d0-831f-4cf6-946b-40bf99179283:0", "cited_for": "gh pr create tool_result: https://github.com/Sinity/polylogue/pull/3282 -- confirms this session actually opened PR #3282, not merely referenced it", "verified_via": "tool_result text (single URL line)"} -{"ref": "block:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39:bdacfb31-ecc7-4491-be2f-891f8bfb888b:0", "cited_for": "tool_use invoking `timeout 180 devtools verify --quick`, checking the PR claim 'devtools verify --quick -- pass'", "verified_via": "tool_input command text"} -{"ref": "block:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39:9786b513-9ad0-4e25-ad3e-e18a0f60220f:0", "cited_for": "tool_result: structured verify-run JSON, every step exit=0, total_duration_s=32.99, exit_code=0 -- SUPPORTS the claim via structure, not a trusted pass/fail word in prose", "verified_via": "tool_result JSON body (per-step name/duration_s/exit array)"} -{"ref": "block:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39:ebe785a2-de91-4de3-9b28-43bd1a8e9596:0", "cited_for": "tool_use invoking devtools test against tests/unit/maintenance/test_rebuild_index_bulk_build.py (+4 more files), checking the Solution-section claim 'rebuild_index bulk FTS materialization checkpoints progress'", "verified_via": "tool_input command text"} -{"ref": "block:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39:59a1be2d-7881-4d64-997d-d055dd61aa74:0", "cited_for": "tool_result: pytest summary '123 passed in 8.05s', ok (12.2s) -- SUPPORTS the rebuild_index claim", "verified_via": "tool_result text (pytest summary line)"} -{"ref": "polylogue-6mvg", "cited_for": "the PR body's own tracking-item reference ('Ref polylogue-6mvg'), cited verbatim inside the gh pr create --body text", "verified_via": "block:...ae9a4788... tool_input"} -{"ref": "action.tool_use.count=53,Bash", "cited_for": "56 total tool_use blocks in this session: 53 Bash, 3 Read, 0 Edit, 0 Write -- this is a merge-conductor/orchestration session, not the direct file-editing session", "verified_via": "sqlite3 index.db: SELECT tool_name, count(*) FROM blocks WHERE session_id=... AND block_type='tool_use' GROUP BY tool_name"} -{"ref": "action.negative-control.missing-7-file-devtools-test-invocation", "cited_for": "counterexample: the PR body's exact 7-file `devtools test tests/unit/sources/test_live_batch_support.py ...` string appears ONLY inside the gh-pr-create --body text (i.e. inside the PR body itself), never as an actually-invoked command in this session's blocks -- 0 matching tool_use rows when the gh-pr-create block is excluded", "verified_via": "sqlite3 index.db: SELECT count(*) FROM blocks WHERE session_id=... AND block_type='tool_use' AND tool_input LIKE '%test_live_batch_support.py%' AND tool_input NOT LIKE '%gh pr create%' -- returns 0"} diff --git a/.agent/demos/d1-receipts/finding.yaml b/.agent/demos/d1-receipts/finding.yaml deleted file mode 100644 index c288a450ab..0000000000 --- a/.agent/demos/d1-receipts/finding.yaml +++ /dev/null @@ -1,6 +0,0 @@ -archive_cursor: "live-archive:/realm/db/polylogue (read-only, file:...?mode=ro)" -measure_version: demo-packet-v2 -commit_sha: 59744a30bf461a587cc679ee62a432e8cd2cf82a -sample_frame_predicate: "session_refs WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282, resolving to claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39" -run_date: "2026-07-31" -claim: "session_refs typed pull_request evidence resolves a real merged PR to its authoring/dispatch session, and specific PR-body verification sentences can be checked against that session's own recorded blocks -- with unsupported claims marked as such, not silently believed" diff --git a/.agent/demos/d1-receipts/packet.json b/.agent/demos/d1-receipts/packet.json deleted file mode 100644 index fb5c35aebe..0000000000 --- a/.agent/demos/d1-receipts/packet.json +++ /dev/null @@ -1,135 +0,0 @@ -{ - "schema_version": "2.0.0", - "packet_id": "d1-receipts", - "title": "D1 receipts: claim-vs-evidence on a real merged PR", - "mode": "private", - "primary_construct": { - "id": "correlation.session_refs.pr_link", - "statement": "session_refs typed pull_request evidence (Claude Code's own pr-link sidecar record) resolves a real merged PR to its authoring/dispatch session with structural confidence, and that session's own tool_use/tool_result blocks let each PR-body verification claim be checked against real evidence instead of trusted prose.", - "product_primitives": [ - "session_refs (storage table)", - "polylogue read --view correlation", - "insights/session_commit.py:build_correlation_result", - "structural SQL reads over blocks/session_refs for citation" - ] - }, - "claim": { - "statement": "session_refs resolves PR #3282 to claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39, and 3 of 4 checked PR-body claims are supported by that session's own recorded blocks; the fourth is correctly scored not_supported rather than assumed.", - "declared_before_execution": true, - "scope": "one real merged PR (Sinity/polylogue#3282) and its session_refs-resolved authoring/dispatch session, on the live archive at /realm/db/polylogue", - "status": "supported", - "receipts": ["artifact:d1-receipts-predeclaration", "artifact:d1-receipts-evidence"] - }, - "oracle": { - "description": "The SQL queries and gh CLI output cited are independent of the report prose and re-runnable against the same live archive and GitHub history.", - "independent": true, - "method": "Re-run the reproduce commands in report.md and compare cited block text / query results against evidence.ndjson.", - "expected": {"queries_return_matching_rows": true, "negative_control_returns_zero": true}, - "receipts": ["artifact:d1-receipts-evidence"] - }, - "baseline": { - "name": "regex/time-window PR reference scan (the pre-#3425 default correlation path)", - "method": "Scan message text for #NNN patterns within a time window around the session, with no structural resolution to a specific typed evidence row.", - "result": "The same read --view correlation call surfaces a disagreements entry: the regex heuristic independently found 13 additional PR numbers in this session's message text that are NOT corroborated by typed session_refs evidence -- demonstrating why the typed path is authoritative and the heuristic path is demoted to a disagreement signal, not silently trusted.", - "receipts": ["artifact:d1-receipts-evidence"] - }, - "controls": { - "negative": [ - { - "id": "gh-pr-create-body-url-match-control", - "purpose": "Prevent a stale or edited PR body from being treated as automatically representative of the session's own recorded claim text.", - "expected": {"gh_pr_create_tool_result_url": "https://github.com/Sinity/polylogue/pull/3282"}, - "observed": {"gh_pr_create_tool_result_url": "https://github.com/Sinity/polylogue/pull/3282"}, - "passed": true, - "receipts": ["artifact:d1-receipts-evidence"] - } - ], - "missing_evidence": [ - { - "id": "unverified-7-file-devtools-test-claim", - "purpose": "Require a PR-body claim's supporting evidence to be structurally present in the resolved session, not assumed from the PR body's own prose.", - "expected": {"status": "not_supported_when_no_matching_tool_use_block_exists"}, - "observed": {"status": "not_supported", "matching_tool_use_rows_excluding_gh_pr_create": 0}, - "passed": true, - "receipts": ["artifact:d1-receipts-evidence"] - } - ] - }, - "falsifier": { - "condition": "The gh pr create tool_use body text does not byte-match the live-fetched PR #3282 body, or the devtools verify --quick tool_result JSON contains any step with exit != 0, or the negative-control count in evidence.ndjson is nonzero.", - "evaluation_method": "Apply the stated condition to the committed evidence and re-run the queries in report.md's Reproduce section against the live archive.", - "triggered": false, - "result": "pass", - "receipts": ["artifact:d1-receipts-evidence"] - }, - "results": { - "status": "pass", - "summary": "3 of 4 checked PR-body claims are structurally supported by the session_refs-resolved session's own blocks; the 4th is correctly scored not_supported, and the session is shown to be a merge-conductor session (0 Edit/Write tool_use blocks) rather than the direct file-editing session -- both are real, documented findings, not hidden.", - "measurements": [ - { - "name": "claims_checked", - "value": 4, - "unit": "claims", - "receipts": ["artifact:d1-receipts-evidence"] - }, - { - "name": "claims_supported", - "value": 3, - "unit": "claims", - "receipts": ["artifact:d1-receipts-evidence"] - }, - { - "name": "claims_not_independently_verified", - "value": 1, - "unit": "claims", - "receipts": ["artifact:d1-receipts-evidence"] - }, - { - "name": "tool_use_blocks_edit_or_write", - "value": 0, - "unit": "blocks", - "receipts": ["artifact:d1-receipts-evidence"] - } - ] - }, - "non_claims": [ - "This packet does not prove every sentence in PR #3282's body is independently verified -- only the four claims explicitly checked are scored.", - "This packet does not establish that session_refs correctly resolves every PR reference archive-wide -- only that it resolves this one case with structural evidence.", - "This packet does not reproduce on the public seed corpus (seed 1843); it requires read-only access to the live archive and the Sinity/polylogue GitHub history -- the public-corpus D1 variant is not built by this packet." - ], - "receipts": [ - { - "ref": "artifact:d1-receipts-evidence", - "kind": "artifact", - "description": "Committed evidence rows and block/session citations for this packet.", - "artifact_path": "evidence.ndjson", - "resolved": true, - "sha256": "1370ede81fa593585b216ed9d4f834093cf50cdd9a2e634467f4c267b3628fad" - }, - { - "ref": "artifact:d1-receipts-predeclaration", - "kind": "artifact", - "description": "The committed prompt that states the packet claim before execution.", - "artifact_path": "PROMPT.md", - "resolved": true, - "sha256": "9860720832a1ed3460a8cb9196b1a0682f9223487821c569817e8f5da3098ecd" - } - ], - "reproduction": { - "fixture": "live archive /realm/db/polylogue (read-only) + Sinity/polylogue GitHub history", - "deterministic": false, - "private_data": true, - "commands": [ - "sqlite3 \"file:/realm/db/polylogue/index.db?mode=ro\" \"SELECT session_id, repo, ref_number, url FROM session_refs WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282\"", - "polylogue find \"id:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39\" then read --view correlation --format json", - "gh pr view 3282 --repo Sinity/polylogue --json body" - ] - }, - "provenance": { - "archive_cursor": "live-archive:/realm/db/polylogue (read-only, file:...?mode=ro)", - "measure_version": "demo-packet-v2", - "commit_sha": "59744a30bf461a587cc679ee62a432e8cd2cf82a", - "sample_frame_predicate": "session_refs WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282", - "run_date": "2026-07-31" - } -} diff --git a/.agent/demos/d1-receipts/queries.ndjson b/.agent/demos/d1-receipts/queries.ndjson deleted file mode 100644 index a12b8a2619..0000000000 --- a/.agent/demos/d1-receipts/queries.ndjson +++ /dev/null @@ -1,5 +0,0 @@ -{"text": "SELECT session_id, repo, ref_number, url FROM session_refs WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282", "lowered_spec": {"unit": "session_ref", "predicate_kind": "structural_equality", "table": "session_refs"}} -{"text": "find \"id:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39\" then read --view correlation --format json", "lowered_spec": {"unit": "session", "entry": "id", "view": "correlation"}} -{"text": "SELECT tool_name, count(*) FROM blocks WHERE session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' AND block_type='tool_use' GROUP BY tool_name ORDER BY 2 DESC", "lowered_spec": {"unit": "block", "pipeline_stages": ["group:tool_name", "count"]}} -{"text": "SELECT tu.block_id, tr.text FROM blocks tu JOIN blocks tr ON tr.tool_id=tu.tool_id AND tr.block_type='tool_result' AND tr.session_id=tu.session_id WHERE tu.session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' AND tu.tool_input LIKE '%devtools verify --quick%'", "lowered_spec": {"unit": "block", "predicate_kind": "join_tool_use_to_tool_result"}} -{"text": "SELECT count(*) FROM blocks WHERE session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' AND block_type='tool_use' AND tool_input LIKE '%test_live_batch_support.py%' AND tool_input NOT LIKE '%gh pr create%'", "lowered_spec": {"unit": "block", "predicate_kind": "negative_control_count", "expected_result": 0}} diff --git a/.agent/demos/d1-receipts/report.md b/.agent/demos/d1-receipts/report.md deleted file mode 100644 index a3de3d1bf0..0000000000 --- a/.agent/demos/d1-receipts/report.md +++ /dev/null @@ -1,153 +0,0 @@ -# D1 "The Receipts": Claim-vs-Evidence on a Real Merged PR - -This file is a Demo Finding Packet artifact (`devtools/demo_packet.py` -`PACKET_FILENAMES` contract), not an agent session summary. It is consumed -by `devtools lab policy demo-packet-registry` and read by future operators -reproducing this demo -- it is checked-in repo content, not a report to the -orchestrating agent. - -## Claim - -`session_refs` typed `pull_request` evidence resolves a real merged PR to -its authoring/dispatch session, and specific sentences from that PR's body -can be checked against the session's own recorded `blocks` -- with claims -that have no matching evidence marked as such, not silently trusted. - -## Corpus - -The live archive (`/realm/db/polylogue`, read-only), scoped to one session: -`claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39`, resolved via -`session_refs WHERE kind='pull_request' AND repo='Sinity/polylogue' AND -ref_number=3282`. This is the real, merged PR -[Sinity/polylogue#3282](https://github.com/Sinity/polylogue/pull/3282) -("perf(storage): defer FTS repair off the live-ingest write path"). - -## Method - -1. Resolved PR #3282 to a session structurally through `session_refs` (the - table PR #3425 populated and PR #3431 wired into - `insights/session_commit.py:build_correlation_result` / - `insights/correlation_view.py`'s `read --view correlation` surface -- - not a regex/time-window guess). -2. Fetched the live PR body via `gh pr view 3282 --json body`. -3. For each individually falsifiable claim in that body, searched the - resolved session's `blocks` table (`tool_use`/`tool_result` joined by - `tool_id`) for matching structural evidence. -4. Recorded each claim's status: `supported` (matching block evidence - found) or `not_supported` (no matching block, regardless of what the PR - prose says). - -## Findings - -Claim-vs-evidence table (full block citations in `evidence.ndjson`): - -| # | PR #3282 claim | Evidence found in the session | Status | -|---|---|---|---| -| 1 | This session authored/opened the PR | `tool_use` block runs `gh pr create --title "perf(storage): defer FTS repair off the live-ingest write path" --body "..."` with body text byte-identical to the live-fetched PR body; `tool_result` returns `https://github.com/Sinity/polylogue/pull/3282` | **supported** | -| 2 | "`devtools verify --quick` -- pass (ruff format/check, mypy, render all, topology/layering/...)" | `tool_use` runs `timeout 180 devtools verify --quick`; `tool_result` is a structured run-JSON with every step's `exit` field `0` (17 steps enumerated, `total_duration_s: 32.99`, top-level `exit_code: 0`) | **supported** (structural -- the exit codes, not a trusted "pass" word) | -| 3 | "`rebuild_index` bulk FTS materialization checkpoints progress (base for #3281, rebased here after that merge)" | `tool_use` runs `devtools test tests/unit/maintenance/test_rebuild_index_bulk_build.py` (+4 more files) in `/realm/worktrees/polylogue-membership-head-provenance`; `tool_result` pytest summary: `123 passed in 8.05s`; a following `git commit` in the same worktree stages exactly `polylogue/maintenance/rebuild_index.py` -- the one file this line's claim is about, matching the PR's own file diff (`polylogue/maintenance/rebuild_index.py 1 1`) | **supported** | -| 4 | "`devtools test tests/unit/sources/test_live_batch_support.py tests/unit/sources/test_live_catchup_planning.py tests/unit/storage/test_revision_replay.py tests/unit/storage/test_fts_identity_ledger.py tests/unit/storage/test_fts_repair_sql.py tests/unit/storage/test_bulk_fts_prefix_reextract.py tests/unit/daemon/test_daemon_cli.py -- all passing (see individual commit messages for per-commit pass counts)" | That exact 7-file string appears **only** inside the `gh pr create --body` tool_input (i.e. inside the PR body text itself) -- 0 rows when searching this session's `tool_use` blocks for the string with the `gh pr create` block excluded | **not independently verified in this session** | - -## Specimens - -See `evidence.ndjson` for the full block-id citations behind each row -above, including the exact `tool_result` text for rows 2 and 3. - -## Counterexamples - -**Finding 4 is a real, structurally-confirmed gap, not an artifact of -sloppy search.** The PR body's own parenthetical for that claim -- -"see individual commit messages for per-commit pass counts" -- already -admits the aggregate 7-file invocation was never run as one shot; this -session's block evidence confirms it structurally: the string is prose -inside the PR body draft, never an executed command. This is the intended -behavior of a claim-vs-evidence packet: a claim the PR body asserts in -prose, with no matching structural evidence in the resolved session, must -render as unsupported -- not silently upgraded because the surrounding -claims (1-3) checked out. - -**This session is a merge-conductor, not the file-editing session.** -`SELECT tool_name, count(*) ... GROUP BY tool_name` over this session's 56 -`tool_use` blocks returns `Bash=53, Read=3` -- zero `Edit`/`Write` blocks. -The PR's actual code changes were authored in separately dispatched worker -sessions across several git worktrees -(`/realm/worktrees/polylogue-membership-head*`); this session orchestrates -`git`/`gh`/`devtools` across them and opens the PR. `session_refs` correctly -resolves PR #3282 to *this* session (the one that ran `gh pr create`), which -is the right target for "which session can I ask about this PR's own -claims" -- but it is not the right target for "which session edited file -X", a different (currently unresolved by this packet) question. - -## Limits - -- This packet checks 4 claims from one PR's body, not every sentence. It is - a method demonstration (structural claim-vs-evidence resolution through - `session_refs`), not an audit of PR #3282's full body. -- This is the **live-archive operator variant** only. The epic's own design - (`polylogue-212`) calls for two variants per demo: a public seeded-corpus - reproduction (seed 1843) and a live-archive operator variant. `session_refs` - `pull_request` rows are a real, provider-native Claude Code capability - (pr-link sidecar records) that the deterministic seed fixture does not - currently populate, so the public variant is not built by this packet -- - named as remaining scope in the owning bead (`polylogue-xyel`) rather than - claimed done here. -- The multi-worktree merge-conductor pattern found in Finding 4/Counterexamples - means `session_refs`'s PR-to-session resolution answers "which session - opened this PR", not "which session wrote this specific line of this - specific file" -- a real, useful distinction this packet surfaces but does - not resolve further (that would need session-to-commit-to-worktree - lineage, which `polylogue-cijx.1`'s notes document as a separate, still- - open problem for the durable `session_commits` table, unrelated to the - `session_refs` mechanism this packet exercises). - -## Non-claims - -- This packet does not prove every sentence in PR #3282's body is - independently verified -- only the four claims explicitly checked above - are scored; claim 4 is explicitly scored `not_supported`. -- This packet does not establish that `session_refs` correctly resolves - every PR reference archive-wide -- only that it resolves this one case - with structural evidence. -- This packet does not reproduce on the public seed corpus (seed 1843); it - requires read-only access to the live archive and the `Sinity/polylogue` - GitHub history. - -## Reproduce - -```bash -# 1. resolve PR -> session -sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT session_id, repo, ref_number, url FROM session_refs \ - WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282" - -# 2. cross-check through the CLI's own correlation surface -POLYLOGUE_ARCHIVE_ROOT=/realm/db/polylogue POLYLOGUE_FORCE_PLAIN=1 \ - polylogue find "id:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39" \ - then read --view correlation --format json - -# 3. fetch the live PR body -gh pr view 3282 --repo Sinity/polylogue --json body - -# 4. tool_name distribution (merge-conductor finding) -sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT tool_name, count(*) FROM blocks \ - WHERE session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND block_type='tool_use' GROUP BY tool_name ORDER BY 2 DESC" - -# 5. the devtools verify --quick evidence (claim 2) -sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT tr.text FROM blocks tu JOIN blocks tr \ - ON tr.tool_id=tu.tool_id AND tr.block_type='tool_result' AND tr.session_id=tu.session_id \ - WHERE tu.session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND tu.tool_input LIKE '%devtools verify --quick%'" - -# 6. the negative-control count (claim 4 -- must return 0) -sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT count(*) FROM blocks \ - WHERE session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND block_type='tool_use' AND tool_input LIKE '%test_live_batch_support.py%' \ - AND tool_input NOT LIKE '%gh pr create%'" -``` - -See `evidence.ndjson` for every cited ref and `checks.json` for the -pass/fail summary. diff --git a/.agent/demos/d1-receipts/run.log b/.agent/demos/d1-receipts/run.log deleted file mode 100644 index 15407a9ff6..0000000000 --- a/.agent/demos/d1-receipts/run.log +++ /dev/null @@ -1,113 +0,0 @@ -=== 1. resolve PR -> session (structural, session_refs) === -$ sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT session_id, repo, ref_number, url FROM session_refs \ - WHERE kind='pull_request' AND repo='Sinity/polylogue' AND ref_number=3282" -claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39|Sinity/polylogue|3282|https://github.com/Sinity/polylogue/pull/3282 - -=== 2. cross-check through the CLI's own correlation surface (production read path) === -$ POLYLOGUE_ARCHIVE_ROOT=/realm/db/polylogue POLYLOGUE_FORCE_PLAIN=1 \ - polylogue find "id:claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39" \ - then read --view correlation --format json -(pr_refs excerpt, one of six duplicate-window matches; source=typed_session_ref proves the - typed session_refs evidence resolved this, not the regex heuristic) -{ - "owner": "Sinity", - "repo": "polylogue", - "number": 3282, - "kind": "pr", - "url": "https://github.com/Sinity/polylogue/pull/3282", - "raw_match": "https://github.com/Sinity/polylogue/pull/3282", - "message_id": null, - "source": "typed_session_ref", - "object_ref": "github-pr:Sinity/polylogue#3282" -} -disagreements: 1 entry -- the regex heuristic path independently found PR numbers -[3212, 3213, 3214, 3215, 3216, 3217, 3262, 3263, 3264, 3271, 3272, 3278, 3281] in message -text that are NOT corroborated by typed session_refs evidence for this session -- surfaced -as a disagreement rather than silently merged into the typed result. - -=== 3. fetch the live PR body === -$ gh pr view 3282 --repo Sinity/polylogue --json body -(full body in evidence.ndjson citation for block ae9a4788-...; byte-identical to the - gh pr create --body tool_input recorded in the session) - -=== 4. tool_name distribution over this session's tool_use blocks (merge-conductor finding) === -$ sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT tool_name, count(*) FROM blocks \ - WHERE session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND block_type='tool_use' GROUP BY tool_name ORDER BY 2 DESC" -Bash|53 -Read|3 - -=== 5. devtools verify --quick evidence (claim 2) === -$ sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT tr.text FROM blocks tu JOIN blocks tr \ - ON tr.tool_id=tu.tool_id AND tr.block_type='tool_result' AND tr.session_id=tu.session_id \ - WHERE tu.session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND tu.tool_input LIKE '%devtools verify --quick%'" -(tail of the structured run-JSON result) - { - "name": "verify docs-coverage", - "duration_s": 2.62, - "exit": 0, - "run_id": "20260726T185346Z-quick-611274-f5c7c756" - }, - { - "name": "verify test-infra-currency", - "duration_s": 0.41, - "exit": 0, - "run_id": "20260726T185346Z-quick-611274-f5c7c756" - }, - { - "name": "verify test-clock-hygiene", - "duration_s": 2.48, - "exit": 0, - "run_id": "20260726T185346Z-quick-611274-f5c7c756" - }, - { - "name": "verify pytest-timeout-overrides", - "duration_s": 4.16, - "exit": 0, - "run_id": "20260726T185346Z-quick-611274-f5c7c756" - }, - { - "name": "verify degrade-loudly", - "duration_s": 1.27, - "exit": 0, - "run_id": "20260726T185346Z-quick-611274-f5c7c756" - } - ], - "total_duration_s": 32.99, - "exit_code": 0 -} -(every step in this run's full JSON has "exit": 0; 17 steps total) - -=== 6. rebuild_index test evidence (claim 3) === -$ sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT tr.text FROM blocks tu JOIN blocks tr \ - ON tr.tool_id=tu.tool_id AND tr.block_type='tool_result' AND tr.session_id=tu.session_id \ - WHERE tu.session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND tu.tool_input LIKE '%test_rebuild_index_bulk_build.py tests/unit/storage/test_planner_statistics_seed.py tests/unit/storage/test_revision_replay.py%'" -2 workers [123 items] -........................................................................ [ 58%] -................................................... [100%] -============================= 123 passed in 8.05s ============================== -ok (12.2s) - -=== 7. negative control: the 7-file devtools test invocation (claim 4) === -$ sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT count(*) FROM blocks \ - WHERE session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND block_type='tool_use' AND tool_input LIKE '%test_live_batch_support.py%' \ - AND tool_input NOT LIKE '%gh pr create%'" -0 --- the only match (without the exclusion) is the gh-pr-create block itself, i.e. the string --- only exists as prose inside the PR body draft, never as an executed command. - -=== 8. gh pr create body/URL evidence (claim 1) === -$ sqlite3 "file:/realm/db/polylogue/index.db?mode=ro" \ - "SELECT tr.text FROM blocks tu JOIN blocks tr \ - ON tr.tool_id=tu.tool_id AND tr.block_type='tool_result' AND tr.session_id=tu.session_id \ - WHERE tu.session_id='claude-code-session:28407f0f-4f30-4508-b29b-1fd0bc301a39' \ - AND tu.tool_input LIKE '%gh pr create%'" -https://github.com/Sinity/polylogue/pull/3282 diff --git a/.agent/demos/d4-behavioral-archaeology/NON-CLAIMS.md b/.agent/demos/d4-behavioral-archaeology/NON-CLAIMS.md deleted file mode 100644 index dc3327fbaa..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/NON-CLAIMS.md +++ /dev/null @@ -1,6 +0,0 @@ -# Non-claims - -- The fixture counts do not estimate production prevalence, archive scale, or provider-wide behavior. -- The packet does not prove that semantic search is complete when only 2 of 71 messages have synthetic embeddings. -- The packet does not prove that a chat UI could never implement equivalent features. -- The historical 2026-07-09 receipt is not evidence of the current 2026-08-09 run. diff --git a/.agent/demos/d4-behavioral-archaeology/PROMPT.md b/.agent/demos/d4-behavioral-archaeology/PROMPT.md deleted file mode 100644 index adc7575c63..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/PROMPT.md +++ /dev/null @@ -1,35 +0,0 @@ -# D4 "Behavioral Archaeology": Six DSL Queries, Rapid Fire - -Predeclaration receipt: `artifact:d4-behavioral-archaeology-predeclaration`. - -Run these six queries against a seeded demo archive (`polylogue demo seed`). -Each answers a question an engineering lead would ask about their team's AI -coding sessions — each impossible to answer from a chat UI transcript view. -Product primitives only (`polylogue` CLI query DSL); no bespoke scripts. - -1. **SEQ thrash-loop hunt** — repeated shell-tool calls in a row: - `polylogue find "sessions where seq(action:shell -> action:shell)" then select --format json` -2. **Tool call volume** — which tools are actually used: - `polylogue "actions where exit_code:>=0 | group by tool | count"` -3. **Which tools break** — failure count by tool: - `polylogue "actions where is_error:true | group by tool | count"` -4. **Semantic probe across providers**: - `polylogue find 'near:"flaky async test"'` -5. **Time-scoped session population**: - `polylogue find "since:2y"` -6. **Pipe a query straight into `read`**: - `polylogue find "origin:codex-session" then read --first --view messages` - -Then show `explain_query_expression` (CLI: `--explain`) once on query 1 to -prove the query means what it says — the parsed AST, not just prose. - -## Note on this run - -While authoring query 1 for this specific seeded fixture, running the exact -same predicate as a bare `find` (no `then` verb) vs `find ... then select` -produced DIFFERENT results — the bare form silently ignored the filter. This -is documented as a real finding in `report.md` (counterexamples) and filed -as its own bug (polylogue-70qb), not hidden. This -IS the point of the demo: a DSL query surfaces things a chat transcript -never could — including, in this case, a defect in the query surface -itself. diff --git a/.agent/demos/d4-behavioral-archaeology/checks.json b/.agent/demos/d4-behavioral-archaeology/checks.json deleted file mode 100644 index e6b95bcd4f..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/checks.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "pass": true, - "unsupported_claims": [], - "coverage_notes": "All 6 queries and the required explain route executed through polylogue at d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce against a private synthetic seeded demo corpus with 19 sessions and 71 messages. Query 4 returned one hit with only 2 of 71 messages embedded, so the packet does not claim semantic completeness. Query 5 is scoped as a time-filter demonstration, not abandonment classification. The bare-find predicate defect remains explicit and is retained as a historical finding." -} diff --git a/.agent/demos/d4-behavioral-archaeology/evidence.ndjson b/.agent/demos/d4-behavioral-archaeology/evidence.ndjson deleted file mode 100644 index a9509ad3d4..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/evidence.ndjson +++ /dev/null @@ -1,12 +0,0 @@ -{"ref": "artifact:d4-behavioral-archaeology-evidence", "cited_for": "Demo Packet v2 receipt root for the fresh run", "verified_via": "committed evidence.ndjson"} -{"ref": "artifact:d4-behavioral-archaeology-current-run", "cited_for": "Fresh private-data-free six-query and explain receipt at commit d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce", "verified_via": "run.log structured command sections and output digests"} -{"ref": "codex-session:demo-receipts", "cited_for": "Q1 SEQ thrash-loop match and Q6 piped-read specimen", "verified_via": "current run q1 then select --format json and q6 then read"} -{"ref": "claude-code-session:63705dcc-f3e5-4378-8118-8bc21e53bbb6", "cited_for": "Q1 SEQ thrash-loop match and Q4 semantic hit", "verified_via": "current run q1 and q4"} -{"ref": "codex-session:demo-00", "cited_for": "Q1 SEQ thrash-loop match", "verified_via": "current run q1 then select --format json"} -{"ref": "actions.group_by=tool", "cited_for": "Q2 tool call volume: Bash=9, Read=8, exec_command=3, and six singleton tools", "verified_via": "current run q2 actions where exit_code:>=0 | group by tool | count"} -{"ref": "actions.group_by=tool.is_error", "cited_for": "Q3 tool failure count: Bash=4, exec_command=2, Edit=1", "verified_via": "current run q3 actions where is_error:true | group by tool | count"} -{"ref": "embeddings.db:message_embeddings_meta", "cited_for": "Synthetic embedding coverage numerator: 2 messages embedded", "verified_via": "sqlite3 embeddings.db SELECT COUNT(*) FROM message_embeddings_meta"} -{"ref": "index.db:messages", "cited_for": "Fresh seeded corpus denominator: 71 total messages, making coverage 2/71", "verified_via": "sqlite3 index.db SELECT COUNT(*) FROM messages"} -{"ref": "since:2y", "cited_for": "Q5 time-scoped population: 13 of 19 sessions", "verified_via": "current run q5 find since:2y"} -{"ref": "codex-session:demo-receipts", "cited_for": "Q6 piped-read specimen: failed clock test, conflicting claim, edit, and successful recovery", "verified_via": "current run q6 find origin:codex-session then read --first --view messages"} -{"ref": "polylogue-70qb", "cited_for": "historical counterexample: bare find ignoring sessions-where predicates", "verified_via": "historical packet comparison; current receipt does not relabel it as a fresh command"} diff --git a/.agent/demos/d4-behavioral-archaeology/finding.yaml b/.agent/demos/d4-behavioral-archaeology/finding.yaml deleted file mode 100644 index e44f313c7f..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/finding.yaml +++ /dev/null @@ -1,12 +0,0 @@ -archive_cursor: polylogue-demo-seed-fixture-world -measure_version: demo-packet-v2 -commit_sha: d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce -sample_frame_predicate: "all 19 sessions in the private synthetic polylogue demo seed fixture" -run_date: "2026-08-09" -current_run_id: run:d4-behavioral-archaeology:20260809T061905Z -current_run_timestamp: "2026-08-09T06:19:05Z" -current_route: polylogue -historical_commit_sha: fdd5ea848 -historical_run_date: "2026-07-09" -historical_receipt_status: historical-only -claim: "six DSL queries answer engineering-lead questions about AI coding sessions that no chat UI transcript view can answer, using only existing product primitives" diff --git a/.agent/demos/d4-behavioral-archaeology/packet.json b/.agent/demos/d4-behavioral-archaeology/packet.json deleted file mode 100644 index 8c04e72339..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/packet.json +++ /dev/null @@ -1,284 +0,0 @@ -{ - "baseline": { - "method": "Use the recorded simpler comparison path rather than the structural product path.", - "name": "ordinary transcript browsing", - "receipts": [ - "artifact:d4-behavioral-archaeology-evidence" - ], - "result": "Inspect sessions one by one without structural predicates or grouped action rows." - }, - "claim": { - "declared_before_execution": true, - "receipts": [ - "artifact:d4-behavioral-archaeology-predeclaration" - ], - "scope": "the committed 19-session seed fixture and the fresh recorded commands", - "statement": "The recorded six-query run exercises the documented query primitives against the deterministic demo archive.", - "status": "supported" - }, - "controls": { - "missing_evidence": [ - { - "expected": "A missing ambient-activity source must remain unsupported rather than inferred from transcript prose.", - "id": "unsupported-cross-source-control", - "observed": { - "status": "not_supported" - }, - "passed": true, - "purpose": "Require missing evidence to remain explicit rather than converted into a positive claim.", - "receipts": [ - "artifact:d4-behavioral-archaeology-evidence" - ] - } - ], - "negative": [ - { - "expected": "Sparse embedding coverage must remain an explicit bounded result rather than a completeness claim.", - "id": "sparse-semantic-control", - "observed": { - "semantic_hits": 1, - "embedded_messages": 2, - "total_messages": 71 - }, - "passed": true, - "purpose": "Prevent an adjacent easier signal from being counted as the primary construct.", - "receipts": [ - "artifact:d4-behavioral-archaeology-evidence" - ] - } - ] - }, - "falsifier": { - "condition": "Any recorded query output cannot be reproduced from the deterministic archive or resolves to different cited rows.", - "evaluation_method": "Apply the stated condition to the structured current-run receipt, committed evidence, and run log.", - "receipts": [ - "artifact:d4-behavioral-archaeology-evidence" - ], - "result": "pass", - "triggered": false - }, - "mode": "public", - "non_claims": [ - "The fixture counts do not estimate production prevalence, archive scale, or provider-wide behavior.", - "The packet does not prove that semantic search is complete when embeddings are sparse.", - "The packet does not prove that a chat UI could never implement equivalent features." - ], - "oracle": { - "description": "The command log and cited deterministic rows are independent of the report prose.", - "expected": { - "commands_exit_zero": true, - "query_count": 6, - "explain_count": 1 - }, - "independent": true, - "method": "Re-run the exact commands and compare selected refs and grouped counts with evidence.ndjson.", - "receipts": [ - "artifact:d4-behavioral-archaeology-evidence" - ] - }, - "packet_id": "d4-behavioral-archaeology", - "primary_construct": { - "id": "query.dsl.composition", - "product_primitives": [ - "polylogue find", - "polylogue select", - "polylogue read", - "polylogue --explain" - ], - "statement": "The query DSL composes structural predicates, aggregations, reads, and explanations over the deterministic archive." - }, - "provenance": { - "archive_cursor": "polylogue-demo-seed-fixture-world", - "commit_sha": "d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce", - "measure_version": "demo-packet-v2", - "run_date": "2026-08-09", - "sample_frame_predicate": "all 19 sessions in the private synthetic deterministic demo archive", - "current_run": { - "run_id": "run:d4-behavioral-archaeology:20260809T061905Z", - "run_timestamp": "2026-08-09T06:19:05Z", - "code_sha": "d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce", - "route": "polylogue", - "archive_kind": "private synthetic seeded demo archive", - "private_data": false, - "run_log": "run.log", - "commands": [ - { - "id": "q1", - "section": "Q1: SEQ thrash-loop hunt (repeated shell calls)", - "command": "polylogue find \"sessions where seq(action:shell -> action:shell)\" then select --format json", - "exit_code": 0, - "output_sha256": "56a4b8a11a6be71ece8096b79be1bbc389d4235375a67fb841b30e5ee1df7ad4" - }, - { - "id": "q2", - "section": "Q2: tool call volume by tool", - "command": "polylogue \"actions where exit_code:>=0 | group by tool | count\"", - "exit_code": 0, - "output_sha256": "ec80035d0e7c46e434bb26f1efe857478a758dc6e37c3196c26064a1ed3f0c2f" - }, - { - "id": "q3", - "section": "Q3: which tools break (failure count by tool)", - "command": "polylogue \"actions where is_error:true | group by tool | count\"", - "exit_code": 0, - "output_sha256": "d74af1611854d6cd7f24f925b4de02b9d3388811b0c97390d8e6ab9fddd55e30" - }, - { - "id": "q4", - "section": "Q4: near: semantic probe across providers", - "command": "polylogue find 'near:\"flaky async test\"'", - "exit_code": 0, - "output_sha256": "86fbc77cdb0c9124a085a5db7339c63ed584d417c9b92d6e69c4d59dbf45b764" - }, - { - "id": "q5", - "section": "Q5: time-scoped session population (last 2 years)", - "command": "polylogue find \"since:2y\"", - "exit_code": 0, - "output_sha256": "c12d4efd26b669dee3ac46a41b9c2265a41dee4164b70c76449590d7eb85fc0a" - }, - { - "id": "q6", - "section": "Q6: query piped into read", - "command": "polylogue find \"origin:codex-session\" then read --first --view messages", - "exit_code": 0, - "output_sha256": "c6a11526af70310fbde28756bde722d635e4da2d7b61140c16ba8a0ac524f6d8" - } - ], - "explain": { - "section": "--explain demonstration (Q1)", - "command": "polylogue --explain find \"sessions where seq(action:shell -> action:shell)\"", - "exit_code": 0, - "output_sha256": "9f2a3e107ff5c3dd34731d8606ced970379915b6adbd89d8f93da02486e98dc0" - } - }, - "historical_receipt": { - "commit_sha": "fdd5ea848", - "run_date": "2026-07-09", - "status": "historical-only", - "description": "Original D4 receipt retained only as historical context; it is not the current reproduction." - } - }, - "receipts": [ - { - "artifact_path": "evidence.ndjson", - "description": "Committed evidence rows and references for this packet.", - "kind": "artifact", - "ref": "artifact:d4-behavioral-archaeology-evidence", - "resolved": true, - "sha256": "799d7ea9914e7c59cca908a65612b0479322e737fb26766b2296eee08790b3a3" - }, - { - "artifact_path": "PROMPT.md", - "description": "The committed prompt that states the packet claim before execution.", - "kind": "artifact", - "ref": "artifact:d4-behavioral-archaeology-predeclaration", - "resolved": true, - "sha256": "478e73d60135183895f680e0adbbd961527510d9f56b1a3795989776e5ca65d2" - }, - { - "artifact_path": "run.log", - "description": "Fresh private seeded six-query and explain receipt at the current code SHA.", - "kind": "artifact", - "ref": "artifact:d4-behavioral-archaeology-current-run", - "resolved": true, - "sha256": "4234c64ea92e637f803bee6a455e08b048ca7a9e01078768585115c0a8bd42d4" - } - ], - "reproduction": { - "commands": [ - "polylogue demo seed --root /private/synthetic/demo-archive --force", - "polylogue demo verify --root /private/synthetic/demo-archive --format json", - "polylogue find \"sessions where seq(action:shell -> action:shell)\" then select --format json", - "polylogue \"actions where exit_code:>=0 | group by tool | count\"", - "polylogue \"actions where is_error:true | group by tool | count\"", - "polylogue find 'near:\"flaky async test\"'", - "polylogue find \"since:2y\"", - "polylogue find \"origin:codex-session\" then read --first --view messages", - "polylogue --explain find \"sessions where seq(action:shell -> action:shell)\"" - ], - "deterministic": true, - "fixture": "polylogue demo seed private synthetic fixture", - "private_data": false - }, - "results": { - "measurements": [ - { - "name": "recorded_query_commands", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "commands", - "value": 6 - }, - { - "name": "seeded_sessions", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "sessions", - "value": 19 - }, - { - "name": "seeded_messages", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "messages", - "value": 71 - }, - { - "name": "q1_sequence_matches", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "sessions", - "value": 3 - }, - { - "name": "q2_successful_actions", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "actions", - "value": 26 - }, - { - "name": "q3_failed_actions", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "actions", - "value": 7 - }, - { - "name": "q4_semantic_hits", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "sessions", - "value": 1 - }, - { - "name": "q5_time_scoped_sessions", - "receipts": [ - "artifact:d4-behavioral-archaeology-current-run" - ], - "unit": "sessions", - "value": 13 - }, - { - "name": "synthetic_embedding_messages", - "receipts": [ - "artifact:d4-behavioral-archaeology-evidence" - ], - "unit": "messages", - "value": 2 - } - ], - "status": "pass", - "summary": "A fresh private seeded run at the exact current SHA exercised all six query surfaces and one explain route; observed counts and the bare-find defect remain explicit." - }, - "schema_version": "2.0.0", - "title": "Behavioral archaeology DSL packet" -} diff --git a/.agent/demos/d4-behavioral-archaeology/queries.ndjson b/.agent/demos/d4-behavioral-archaeology/queries.ndjson deleted file mode 100644 index 35ba2d9c6d..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/queries.ndjson +++ /dev/null @@ -1,6 +0,0 @@ -{"id": "q1", "text": "sessions where seq(action:shell -> action:shell)", "then": "select --format json", "historical_receipt": "then select --json (retired alias, historical only)", "current_run_id": "run:d4-behavioral-archaeology:20260809T061905Z", "lowered_spec": {"unit": "session", "entry": "boolean", "predicate_kind": "sequence"}} -{"id": "q2", "text": "actions where exit_code:>=0 | group by tool | count", "current_run_id": "run:d4-behavioral-archaeology:20260809T061905Z", "lowered_spec": {"unit": "action", "pipeline_stages": ["group:tool", "count"]}} -{"id": "q3", "text": "actions where is_error:true | group by tool | count", "current_run_id": "run:d4-behavioral-archaeology:20260809T061905Z", "lowered_spec": {"unit": "action", "pipeline_stages": ["group:tool", "count"]}} -{"id": "q4", "text": "near:\"flaky async test\"", "current_run_id": "run:d4-behavioral-archaeology:20260809T061905Z", "lowered_spec": {"unit": "session", "entry": "compact", "retrieval_lane": "dialogue"}} -{"id": "q5", "text": "since:2y", "current_run_id": "run:d4-behavioral-archaeology:20260809T061905Z", "lowered_spec": {"unit": "session", "entry": "compact"}} -{"id": "q6", "text": "origin:codex-session", "then": "read --first --view messages", "current_run_id": "run:d4-behavioral-archaeology:20260809T061905Z", "lowered_spec": {"unit": "session", "entry": "compact"}} diff --git a/.agent/demos/d4-behavioral-archaeology/report.md b/.agent/demos/d4-behavioral-archaeology/report.md deleted file mode 100644 index 0b6e54b390..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/report.md +++ /dev/null @@ -1,63 +0,0 @@ -# D4 "Behavioral Archaeology": Six DSL Queries, Rapid Fire - -## Claim - -Six DSL queries answer engineering-lead questions about AI coding sessions that no chat UI transcript view answers from a saved transcript alone, using only existing product primitives (`polylogue` CLI query DSL), with no bespoke scripts. - -## Corpus - -The fresh private-data-free run used the deterministic `polylogue demo seed` fixture at `run:d4-behavioral-archaeology:20260809T061905Z`. It contained 19 sessions, 71 messages, and 2 synthetic message embeddings. The command receipt was captured at commit `d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce` on 2026-08-09. - -## Method - -The seed and verification steps ran against a throwaway private synthetic archive. Each of the six PROMPT.md commands then ran once through the `polylogue` CLI, followed by the required `polylogue --explain` route for Q1. `run.log` records each exact command, output, exit code, and output digest. The packet validator cross-checks those structured fields. The older 2026-07-09 receipt at `fdd5ea848` is retained only as historical context. - -## Findings - -1. **SEQ thrash-loop hunt**: `seq(action:shell -> action:shell)` finds 3 sessions with consecutive shell-tool calls: `codex-session:demo-receipts`, `claude-code-session:63705dcc-...`, and `codex-session:demo-00`. The current command uses `then select --format json`. The older `then select --json` spelling appears only in clearly labeled historical-receipt fields. -2. **Tool call volume**: the current run reports 26 successful actions: Bash 9, Read 8, exec_command 3, and one each for Edit, Task, Write, apply_patch, read_file, and run_check. -3. **Which tools break**: the current run reports 7 failed actions: Bash 4, exec_command 2, and Edit 1. -4. **Semantic probe across providers**: `near:"flaky async test"` returns one result, a Claude Code session with the synthetic fixture message "I will inspect the generated fixture and adjust the next command." Only 2 of 71 messages have synthetic embeddings, so this result does not establish complete semantic coverage. -5. **Time-scoped session population**: `since:2y` lists 13 of 19 sessions. This demonstrates time filtering only. It does not reproduce the severity or resumability scoring of `find_abandoned_sessions`. -6. **Query piped into `read`**: `find 'origin:codex-session' then read --first --view messages` renders `codex-session:demo-receipts`, including a failed clock test command with exit code 1, a conflicting success claim, an `apply_patch`, and a later successful test result. - -## Specimens - -See `evidence.ndjson` for the cited session, action, embedding, and current-run references. The current command and output mapping is in `packet.json.provenance.current_run` and is independently bound to `run.log` by output digests. - -## Counterexamples - -The original D4 work also recorded a real defect: the bare command `polylogue find "sessions where seq(action:shell -> action:shell)"` ignored the explicit predicate and returned the full session set. That historical comparison remains documented as `polylogue-70qb`. It is not presented as a newly executed command in this receipt, and the retired `then select --json` alias is retained only as historical receipt text. - -## Limits - -- This is a deterministic seeded demo corpus, not the live archive. The 19-session and 71-message counts are illustrative and do not estimate production prevalence, archive scale, or provider-wide behavior. -- The semantic result is bounded by 2 synthetic embeddings out of 71 messages. It does not prove semantic search completeness. -- The time query demonstrates filtering only. It does not replicate `find_abandoned_sessions` severity or resumability scoring. -- The packet records a current run and a historical receipt. The historical receipt is not evidence for the current code SHA. - -## Non-claims - -- The fixture counts do not estimate production prevalence, archive scale, or provider-wide behavior. -- The packet does not prove that semantic search is complete when embeddings are sparse. -- The packet does not prove that a chat UI could never implement equivalent features. - -## Reproduce - -```bash -polylogue demo seed --root /path/to/private-synthetic-demo-archive --force -export POLYLOGUE_ARCHIVE_ROOT=/path/to/private-synthetic-demo-archive -export POLYLOGUE_FORCE_PLAIN=1 - -polylogue find 'sessions where seq(action:shell -> action:shell)' then select --format json -polylogue 'actions where exit_code:>=0 | group by tool | count' -polylogue 'actions where is_error:true | group by tool | count' -polylogue find 'near:"flaky async test"' -polylogue find 'since:2y' -polylogue find 'origin:codex-session' then read --first --view messages - -# --explain demonstration -polylogue --explain find 'sessions where seq(action:shell -> action:shell)' -``` - -See `run.log` for the exact current output of every command above. The packet's current-run receipt identifies the exact code SHA, UTC timestamp, production route, private synthetic archive class, command mapping, and output hashes. diff --git a/.agent/demos/d4-behavioral-archaeology/run.log b/.agent/demos/d4-behavioral-archaeology/run.log deleted file mode 100644 index 2931d16826..0000000000 --- a/.agent/demos/d4-behavioral-archaeology/run.log +++ /dev/null @@ -1,236 +0,0 @@ -RUN_ID=run:d4-behavioral-archaeology:20260809T061905Z -RUN_COMMANDS_STARTED_UTC=2026-08-09T06:19:05Z -RUN_TIMESTAMP_UTC=2026-08-09T06:19:05Z -HEAD_SHA=d23f5dd27a0bda0a9c0a4306ef98c898d4d920ce -ROUTE=polylogue -ARCHIVE_KIND=private synthetic seeded demo archive -PRIVATE_DATA=false -RECEIPT_REF=artifact:d4-behavioral-archaeology-current-run -=== Q1: SEQ thrash-loop hunt (repeated shell calls) === -$ polylogue find "sessions where seq(action:shell -> action:shell)" then select --format json -[{"id":"codex-session:demo-receipts","origin":"codex-session","title":"Fix the clock-sensitive test and prove the suite passes.","date":"2026-07-04"},{"id":"claude-code-session:63705dcc-f3e5-4378-8118-8bc21e53bbb6","origin":"claude-code-session","title":"synthetic-46324 syntheti","date":"2024-11-02"},{"id":"codex-session:demo-00","origin":"codex-session","title":"Could you review this code for potential issues?","date":"2024-01-23"}] -Q1_EXIT_CODE=0 -=== Q2: tool call volume by tool === -$ polylogue "actions where exit_code:>=0 | group by tool | count" -tool=Bash count=9 -tool=Read count=8 -tool=exec_command count=3 -tool=Edit count=1 -tool=Task count=1 -tool=Write count=1 -tool=apply_patch count=1 -tool=read_file count=1 -tool=run_check count=1 -Q2_EXIT_CODE=0 -=== Q3: which tools break (failure count by tool) === -$ polylogue "actions where is_error:true | group by tool | count" -tool=Bash count=4 -tool=exec_command count=2 -tool=Edit count=1 -Q3_EXIT_CODE=0 -=== Q4: near: semantic probe across providers === -$ polylogue find 'near:"flaky async test"' -1. claude-code-session synthetic-46324 syntheti I will inspect the generated fixture and adjust the next command. -Q4_EXIT_CODE=0 -=== Q5: time-scoped session population (last 2 years) === -$ polylogue find "since:2y" -codex-session:demo-recei 2026-07-04 codex-session Fix the clock-sensitive test and prove the suit... (10 msgs) -codex-session:demo-anti- 2026-07-04 codex-session Explain error budgets without running a command. (2 msgs) -codex-session:demo-termi 2026-07-04 codex-session Run the command and stop if it fails. (4 msgs) -claude-code-session:demo 2026-07-04 claude-code-session Run a sidechain check for the deterministic dem... (2 msgs) -claude-code-session:demo 2026-07-04 claude-code-session Fix the flaky clock test before continuing the... (4 msgs) -hermes-session:demo-00 2026-07-04 hermes-session 5 msgs · 2026-07-04 (5 msgs) -codex-session:demo-linea 2026-07-04 codex-session Map the demo lineage base context. (3 msgs) -antigravity-session:demo 2026-07-04 antigravity-session Antigravity demo cascade (2 msgs) -gemini-cli-session:demo- 2026-07-04 gemini-cli-session 1 file · 2 msgs · 2026-07-04 (2 msgs) -chatgpt-export:cross-mat 2026-07-04 chatgpt-export Flaky clock test fix summary (2 msgs) -chatgpt-export:cross-mat 2026-07-04 chatgpt-export Flaky clock test fix summary (2 msgs) -claude-ai-export:demo-te 2026-07-04 claude-ai-export Temporary demo context check (2 msgs) -claude-code-session:6370 2024-11-02 claude-code-session synthetic-46324 syntheti (12 msgs) -Q5_EXIT_CODE=0 -=== Q6: query piped into read === -$ polylogue find "origin:codex-session" then read --first --view messages -### Lineage boundary · unknown - -- session: `session:codex-session:demo-receipts` - -- authority: `session_row` - -- availability: `unavailable` - -- relation: `unknown` - -- composed transcript: `complete` - -- evidence: `session:codex-session:demo-receipts`, `provider:codex`, `origin:codex-session` - -> Caveat: root identity is unavailable from this bounded lineage authority - -> Caveat: the structural lineage relation is unknown - ---- - -**user · message** · `text` · `message:codex-session:demo-receipts:n:receipts-u0`, `provider:codex`, `origin:codex-session`, `material:human_authored`, `block:codex-session:demo-receipts:n:receipts-u0:0`, `occurred-at:2026-07-04T14:32:01+00:00`, `duration-ms:0`, `variant:0`, `active-path:true`, `active-leaf:false` - -Fix the clock-sensitive test and prove the suite passes. - ---- - -### Shell command · FAILED - -- tool: `exec_command` - -- semantic family: `shell` - -- classification: `persisted_semantic_type` - -- duration: `0 ms` - -- command: `pytest tests/test_clock.py -q` - -- evidence: `session:codex-session:demo-receipts`, `provider:codex`, `origin:codex-session`, `message:codex-session:demo-receipts:n:fc-receipts-test-fail`, `block:codex-session:demo-receipts:n:fc-receipts-test-fail:0`, `tool:exec_command`, `tool-id:call-receipts-test-fail`, `material:assistant_authored`, `duration-ms:0`, `parent-message:codex-session:demo-receipts:n:receipts-u0`, `variant:0`, `active-path:true`, `active-leaf:false`, `result-message:codex-session:demo-receipts:n:call-receipts-test-fail`, `result-block:codex-session:demo-receipts:n:call-receipts-test-fail:0`, `result-duration-ms:0`, `result-material:tool_result` - -- structural outcome: `failed`, `is_error=true`, `exit_code=1` - -**output** - -```text -{"metadata": {"exit_code": 1}, "output": "F tests/test_clock.py::test_uses_monotonic_clock\n1 failed in 0.18s"} -``` - ---- - -**assistant · message** · `text` · `message:codex-session:demo-receipts:n:receipts-a-claim`, `provider:codex`, `origin:codex-session`, `material:assistant_authored`, `block:codex-session:demo-receipts:n:receipts-a-claim:0`, `occurred-at:2026-07-04T14:32:04+00:00`, `duration-ms:0`, `parent-message:codex-session:demo-receipts:n:call-receipts-test-fail`, `variant:0`, `active-path:true`, `active-leaf:false` - -All tests pass. The clock fix is complete. - ---- - -**user · message** · `text` · `message:codex-session:demo-receipts:n:receipts-u1`, `provider:codex`, `origin:codex-session`, `material:human_authored`, `block:codex-session:demo-receipts:n:receipts-u1:0`, `occurred-at:2026-07-04T14:32:05+00:00`, `duration-ms:0`, `parent-message:codex-session:demo-receipts:n:receipts-a-claim`, `variant:0`, `active-path:true`, `active-leaf:false` - -The receipt disagrees. Correct the fixture and verify again. - ---- - -### File edit · succeeded - -- tool: `apply_patch` - -- semantic family: `file_edit` - -- classification: `persisted_semantic_type` - -- duration: `0 ms` - -- evidence: `session:codex-session:demo-receipts`, `provider:codex`, `origin:codex-session`, `message:codex-session:demo-receipts:n:fc-receipts-edit`, `block:codex-session:demo-receipts:n:fc-receipts-edit:0`, `tool:apply_patch`, `tool-id:call-receipts-edit`, `material:assistant_authored`, `duration-ms:0`, `parent-message:codex-session:demo-receipts:n:receipts-u1`, `variant:0`, `active-path:true`, `active-leaf:false`, `result-message:codex-session:demo-receipts:n:call-receipts-edit`, `result-block:codex-session:demo-receipts:n:call-receipts-edit:0`, `result-duration-ms:0`, `result-material:tool_result` - -- structural outcome: `succeeded`, `is_error=false`, `exit_code=0` - -**diff** - -```diff -*** Begin Patch -*** Update File: tests/test_clock.py -@@ -- shared_clock -+ isolated_clock -*** End Patch -``` - -**result** - -```text -{"metadata": {"exit_code": 0}, "output": "Done!"} -``` - -> Caveat: no exact path field is present in tool input - ---- - -### Shell command · succeeded - -- tool: `exec_command` - -- semantic family: `shell` - -- classification: `persisted_semantic_type` - -- duration: `0 ms` - -- command: `pytest tests/test_clock.py -q` - -- evidence: `session:codex-session:demo-receipts`, `provider:codex`, `origin:codex-session`, `message:codex-session:demo-receipts:n:fc-receipts-test-pass`, `block:codex-session:demo-receipts:n:fc-receipts-test-pass:0`, `tool:exec_command`, `tool-id:call-receipts-test-pass`, `material:assistant_authored`, `duration-ms:0`, `parent-message:codex-session:demo-receipts:n:call-receipts-edit`, `variant:0`, `active-path:true`, `active-leaf:false`, `result-message:codex-session:demo-receipts:n:call-receipts-test-pass`, `result-block:codex-session:demo-receipts:n:call-receipts-test-pass:0`, `result-duration-ms:0`, `result-material:tool_result` - -- structural outcome: `succeeded`, `is_error=false`, `exit_code=0` - -**output** - -```text -{"metadata": {"exit_code": 0}, "output": ". 1 passed in 0.16s"} -``` - ---- - -**assistant · message** · `text` · `message:codex-session:demo-receipts:n:receipts-a-verified`, `provider:codex`, `origin:codex-session`, `material:assistant_authored`, `block:codex-session:demo-receipts:n:receipts-a-verified:0`, `occurred-at:2026-07-04T14:32:10+00:00`, `duration-ms:0`, `parent-message:codex-session:demo-receipts:n:call-receipts-test-pass`, `variant:0`, `active-path:true`, `active-leaf:true` - -Verified after the correction: 1 passed in 0.16s. -Q6_EXIT_CODE=0 -=== --explain demonstration (Q1) === -$ polylogue --explain find "sessions where seq(action:shell -> action:shell)" -query: sessions where seq(action:shell -> action:shell) -lowerer: lark-query-expression-to-session-query-spec -units: action, session -execution legs: sequence-action -predicate: -{ - "actions": [ - "shell", - "shell" - ], - "kind": "sequence", - "steps": [ - { - "field": "action", - "field_ref": { - "name": "action", - "scope": "unit", - "source_name": "action", - "unit": "action" - }, - "kind": "field", - "op": "=", - "values": [ - "shell" - ] - }, - { - "field": "action", - "field_ref": { - "name": "action", - "scope": "unit", - "source_name": "action", - "unit": "action" - }, - "kind": "field", - "op": "=", - "values": [ - "shell" - ] - } - ], - "unit": "action" -} -lowering plan: -{ - "execution_legs": [ - "sequence-action" - ], - "lowerer": "lark-query-expression-to-session-query-spec", - "plan_description": [], - "selected_units": [ - "action", - "session" - ] -} -EXPLAIN_EXIT_CODE=0 -RUN_COMMANDS_FINISHED_UTC=2026-08-09T06:20:16Z diff --git a/.agent/demos/registry.json b/.agent/demos/registry.json deleted file mode 100644 index 367c242533..0000000000 --- a/.agent/demos/registry.json +++ /dev/null @@ -1,45 +0,0 @@ -[ - { - "slug": "packet-contract-stub", - "prompt_path": ".agent/demos/_packet-contract-stub/PROMPT.md", - "packet_dir": ".agent/demos/_packet-contract-stub", - "mode": "fixture", - "required_primitives": [ - "polylogue demo seed", - "polylogue find" - ] - }, - { - "slug": "d4-behavioral-archaeology", - "prompt_path": ".agent/demos/d4-behavioral-archaeology/PROMPT.md", - "packet_dir": ".agent/demos/d4-behavioral-archaeology", - "mode": "public", - "required_primitives": [ - "polylogue demo seed", - "polylogue find", - "polylogue --explain", - "polylogue read", - "polylogue select" - ] - }, - { - "slug": "d1-receipts", - "prompt_path": ".agent/demos/d1-receipts/PROMPT.md", - "packet_dir": ".agent/demos/d1-receipts", - "mode": "private", - "required_primitives": [ - "session_refs", - "polylogue find", - "polylogue read --view correlation" - ] - }, - { - "slug": "anti-demo-multi-source-reconstruction", - "prompt_path": ".agent/demos/anti-demo-multi-source-reconstruction/PROMPT.md", - "packet_dir": ".agent/demos/anti-demo-multi-source-reconstruction", - "mode": "anti-demo", - "required_primitives": [ - "schema inspection (grep over archive_tiers DDL)" - ] - } -] diff --git a/.agent/demos/uplift-two-arm/README.md b/.agent/demos/uplift-two-arm/README.md deleted file mode 100644 index a6ce2f49a6..0000000000 --- a/.agent/demos/uplift-two-arm/README.md +++ /dev/null @@ -1,18 +0,0 @@ -# Handoff-Pack Uplift Experiment - -This shelf is for current two-arm experiments that test whether bounded -Polylogue handoff packets improve continuation reconstruction compared with a -raw archive reference. - -The shelf is current, not append-only. Regenerate or replace `current/` when a -better protocol or cleaner run exists. - -## Current run (2026-07-09, n=5 pilot) - -Successor to the n=1 `jxe`/`jxe.2` pilot (raw-ref 8/10 vs handoff-pack 5/10, -diagnostic-negative, attributed to packet staleness). See `current/report.md` -for the full write-up: n=5 pairs, handoff-pack wins 4/5, directional but -explicitly **not** a publishable result (protocol calls for n>=12-20). One -pair's blind was compromised and is documented rather than discarded; one -pair is a genuine counterexample where the pack arm asserted a false fact -with high confidence. Cold-reader gate: PASS. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair1-handoff-pack-output.md b/.agent/demos/uplift-two-arm/current/arms/pair1-handoff-pack-output.md deleted file mode 100644 index 672f02d18f..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair1-handoff-pack-output.md +++ /dev/null @@ -1,59 +0,0 @@ -# Devloop Reconstruction Report — polylogue @ 91eb09549 - -## (a) Current devloop state - -The devloop has been executing a long, disciplined merge train (11+ squash-merged PRs in the last few hours: #2588–#2604), alternating between real correctness fixes and investigation/bookkeeping. Verified via `git log --oneline -15`: - -- `91eb09549` close polylogue-4ts.6 (#2604) -- `c06ca601c` feat(lineage): surface a completeness signal on composed reads (#2603) -- `1d7d79ed2`/`6c12e9234` close polylogue-cpf.1 / timestamp-doctrine lint (#2601-2602) -- `9f0e77116`/`7b5a5aa05` actions-view fan-out fix + close xnkf (#2597-2598) -- `aa4a520b4` record 4ts.3 investigation findings, unclaim (#2596) -- `ccd5cfdd5`/`086171701` close 4ts.4 / lineage read-transaction fix (#2594-2595) -- `f7c997386` record 1vpm.1 investigation findings, unclaim (#2593) -- `43d7fdf9b`/`64c079d6e` close 212.8 / honesty anti-demo (#2591-2592) - -This matches the pack's shipped-work list exactly and extends it one item further (`c06ca601c`, a lineage completeness signal, landed after the pack was generated). - -**Delivery gate board** (`.agent/tools/delivery-gate-status.py --fresh`), the authoritative frontier view: - -``` -A-trust-floor 23% closed 14 | ready 43 | blocked 3 <- frontier -B-storage-rebuild-bytes 4% closed 1 | ready 19 | blocked 3 -F-lineage-compaction 17% closed 2 | ready 3 | blocked 7 -I-analytics-experiments 0% closed 0 | wip 1 (1vpm.1) | ready 9 | blocked 20 -L-external-legibility 13% closed 4 | ready 20 | blocked 6 -``` - -A-trust-floor is the named frontier lane and is the biggest one still open (43 ready items), mostly the `polylogue-9e5.*` read-only audit family (coverage economics, dead-code sweeps, hash-boundary census, etc.) — this is exactly the "trust-floor P1s" the operator's `/goal` names. - -## (b) Open threads found - -1. **polylogue-1vpm.1 — live/committed state mismatch.** The last committed jsonl record (`f7c997386`, "record findings, unclaim") sets `status:"open"`. But `bd show polylogue-1vpm.1 --json` right now reports `status:"in_progress"`, `assignee:"Sinity"`, `updated_at: 2026-07-09T03:24:19Z` — a timestamp with **no corresponding commit** touching that bead in `.beads/issues.jsonl` history. This means the live embedded-dolt DB has state ahead of the last export/commit (some claim happened that hasn't been exported+committed yet). Not something I should "fix" here, but worth flagging before anyone else claims it — the committed bookkeeping record and the live DB disagree. -2. **F-lineage-compaction ready set is exactly 3 items**: `polylogue-4ts` (epic), `polylogue-4ts.3` (investigated, fix location known), `polylogue-4ts.5` (fresh, "compaction boundary-range columns + effective-context derivation" — not yet investigated). -3. **B-storage-rebuild-bytes ready set** includes `83u.2` (re-investigated/reframed to the "bytes reachable but not fetched" subset, per commit `6796ac1ed`) and `83u.3` ("preserve uploaded attachment bytes in live browser capture" — fresh, untouched). -4. **212.x demo-unlock parallel set**: 212.1/212.3/212.5/212.6 remain open with no assignee; per the pack each has its own blocker (deep interpretive read / missing join primitive / live stagecraft / classifier bug); none looked trivially unblockable from what I can see in their notes (no blocker detail beyond the delivery-upgrade boilerplate), so I did not find a shortcut through any of them. -5. **pj8** depends on `37t.4` (SessionStart preamble rollout) which is itself gated on real MCP-hook-wiring code, not just deployment — it's `open`, unclaimed, priority 3, and the pack says it's now "authorized" for the hook wiring step, but actually landing it still implies a SessionStart hook change that the pack says needs redeploy to prove — consistent with "postpone" guidance. -6. **3tl** — confirmed still fully untouched (17 open children), and independently confirmed P4/docs-marketing in this pass (no evidence contradicting the pack). -7. **cfk** — this very context-pack handoff is explicitly part of the live paired-arm uplift experiment it depends on; I was told not to read `.agent/demos/uplift-two-arm/`, so I can't and shouldn't try to advance or grade that experiment from inside it. - -## (c) Recommended next action - -**Claim and implement `polylogue-4ts.3`** — "Distinguish subagent auto-compaction from main-session acompact." - -Reasoning: -- It is explicitly named as part of the "lineage epic" the operator's goal calls out, and is one of only 3 ready items left in F-lineage-compaction (the epic is 17% closed, actively being worked this session — `4ts.4`/`4ts.6` just shipped). -- Unlike `1vpm.1` (comparable in size to `svfj`, a multi-surface feature — new query unit, new column, new registered `target_kind`) or `4ts.5` (still needs fresh investigation), `4ts.3` already has a completed, verified investigation on the bead itself, with an exact fix location I independently confirmed still exists at HEAD: - - `polylogue/sources/dispatch.py:354` `_claude_code_grouped_record_specs` - - `polylogue/sources/dispatch.py:441` `_claude_code_stream_sessions` - - Both are called from the dispatch layer (lines 541, 831) and are the layer with visibility into sibling session-group content needed to distinguish subagent self-compaction from main-session `acompact`. -- It requires no redeploy/MCP-reboot (unlike pj8/37t.4) and no live paired-arm experiment (unlike cfk). -- It's a real correctness bug (misattributed parent session on ~39/187 affected files per the GH thread cited in the bead), not read-only analysis, so it continues today's demonstrated pattern of shipping concrete lineage-correctness fixes rather than accumulating more audit-lane findings on top of the already-large 43-item A-trust-floor backlog. - -Secondary/fallback candidates if `4ts.3` turns out blocked on something I couldn't see (e.g. a fixture gap): `polylogue-83u.3` (fresh, unclaimed, in the storage-rebuild-bytes lane) or one of the trust-floor `9e5.*` P1 audits (e.g. `polylogue-9e5.6`, hash-boundary census — the closest literal match to "storage identity" in the operator's phrasing). - -## (d) Confidence and evidence - -**Confidence: high** on devloop state and gate-board numbers (directly queried live `bd`/git, not inferred from the pack); **medium-high** on the `4ts.3` recommendation (its investigation note is detailed and the code anchors check out, but I have not read the full GH issue thread it references, nor written/run the fix itself); **medium** on the `1vpm.1` state-mismatch finding (clearly true from the data, but I don't know *why* — could be an uncommitted claim from a concurrent session, consistent with this repo's documented shared-checkout/multi-session risk). - -Evidence used: `git log --oneline` / `git log -p -- .beads/issues.jsonl` on the current worktree; `bd show`/`bd ready --json` for `polylogue-pj8`, `37t.4`, `3tl`, `cfk`, `212.9`, `1vpm.1`, `83u.2`, `4ts.3`, `4ts`, `9e5.6`, and the 212.1/212.2/212.3/212.5/212.6 family; `.agent/tools/delivery-gate-status.py --fresh` (whole-board and per-gate); direct `grep` confirmation of the `dispatch.py` fix anchors named in the `4ts.3` bead notes. No files under `.agent/demos/uplift-two-arm/` were read, per instruction. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair1-raw-ref-output.md b/.agent/demos/uplift-two-arm/current/arms/pair1-raw-ref-output.md deleted file mode 100644 index 45d0677061..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair1-raw-ref-output.md +++ /dev/null @@ -1,63 +0,0 @@ -# Devloop State Reconstruction — polylogue @ commit 91eb09549 - -## (a) Current devloop state - -The worktree HEAD (`91eb09549`) sits at the end of a dense, uninterrupted run of small merged PRs. Recent `git log --oneline -30` shows a tight cadence of `feat/fix` commits each immediately followed by a `chore(beads): close (#N)` commit — the devloop is running "ship → close bead → next" with no idle gaps: - -``` -91eb09549 chore(beads): close polylogue-4ts.6 (#2604) -c06ca601c feat(lineage): surface a completeness signal on composed reads (#2603) -1d7d79ed2 chore(beads): close polylogue-cpf.1 (#2602) -6c12e9234 feat(devtools): timestamp-doctrine lint for durable-tier DDL (#2601) -01592e5e9 chore(beads): close polylogue-jsy (#2600) -b6b9fef2a fix(security): harden blob hash validation, drop misleading symlink check (#2599) -... -``` - -The two most recent substantive shipped items: -1. **`polylogue-cpf.1`** (PR #2601) — added `devtools/verify_timestamp_doctrine.py`, a lint rejecting `TEXT` timestamp columns in new durable-tier DDL. Wired into `devtools verify --lab`. This is doctrine-chain work under the `polylogue-cpf` epic ("Land the six doctrines"). -2. **`polylogue-4ts.6`** (PR #2603, just closed) — added `lineage_complete`/`lineage_truncation_reason` to `ArchiveSessionEnvelope` (both sync and async composition paths), wired it into the two MCP-facing payloads. CodeRabbit flagged on review that two more read surfaces (CLI reader payload, Python API `Session` model) and three async batch/paginated wrappers still silently drop the signal — this was **not silently ignored**: it was spun into a fresh follow-up bead, `polylogue-vv2b`, filed at close time with the exact same additive pattern already validated in #2603. - -`bd ready --json` currently returns **299 open, unblocked** beads — this is a long-horizon backlog, not near exhaustion. Beads are grouped into big epics (`9e5` audit lane, `cpf` doctrines, `t46`/`jnj`/`fnm` surface-algebra work, `20d`/`1xc` performance/scale, `bby` web workbench, `a7xr` substrate consolidation, etc.), most tagged with `delivery:` labels running a lettered sequence `A-trust-floor → ... → N-horizon`. - -Crucially, `.agent/tools/delivery-gate-status.py --fresh` (an ordinary repo script, not the forbidden demo dir) shows the priority field was **just reconciled to track this gate order** (commit `a2ee55ec4`, PR #2584, "Ref #8e1b" — mechanical sweep, 288 beads repriced). Gate state right now: - -``` -> A-trust-floor 23% closed(14) wip(0) ready(43) blocked(3) <- the active frontier - B-storage-rebuild-bytes 4% ready(19) - C-read-evidence-contract 2% ready(51) - D-agent-context-coordination 0% ready(34) - E-variants-preferences 0% - F-lineage-compaction 17% closed(2) ready(3) blocked(7) - ...through N-horizon -``` - -`A-trust-floor`'s exit criterion: "Full verification classified; security negative tests pass; missing bytes classified; numbers/time/prose-mined fields carry honest provenance; agent writes land as candidates." Priority-1 in `bd ready` == this gate by construction. - -## (b) Open threads found - -- **`polylogue-cpf` epic** (doctrine landing, gate A-trust-floor): `cpf.1` just closed. Siblings `cpf.2` (writer-class docstring + layering check) and `cpf.3` (injected-context deny-lexicon tripwire fixture) are both **open, wave:1, same epic, same "cheap lint" shape** as the just-shipped `cpf.1`. `cpf.4` (the broader "sweep silent soft-failure paths" class bead) explicitly names three concrete instances to verify against: `1xc.11` (closed 2026-07-05), `4ts.6` (**just closed**), and `tf0e` (still open, but in a different, lower-tier gate `K-interop-origin-export`). So 2 of 3 named instances are now resolved — `cpf.4` is closer to being closeable than it was an hour ago. -- **`polylogue-vv2b`** (new, filed at `4ts.6` close time): wire the same `lineage_complete` signal into the CLI reader payload (`cli/archive_query.py:2198`), the Python API `Session` model (`api/archive.py:1162`), and the three async batch/paginated wrappers. Small, additive, pattern already proven in #2603. It carries no `delivery:*` gate label yet (falls into the 26-bead "unlabeled_open" bucket), so it sits outside the just-reconciled priority scheme — priority 3 is a leftover default, not a gate-derived value. -- **Housekeeping loose end:** `polylogue-8e1b` ("Reconcile bead priority field with delivery-gate order") shipped its work in merged PR #2584 but the bead itself is still `status: in_progress` (assignee Sinity, no `closed_at`) — the work is done and merged; the bead just wasn't formally closed out. -- Two investigation-only threads were recorded and unclaimed rather than completed: `4ts.3` ("record 4ts.3 investigation findings, unclaim") and `1vpm.1` (same pattern) — these remain open for someone to pick back up with the recorded findings as a starting point. -- The `9e5` "Audit lane" epic has a large cluster of ready, priority-1/gate-A items (`9e5.1`, `9e5.3`–`9e5.27` minus closed ones) — read-only analysis work producing evidence artifacts, generally larger/more open-ended than the `cpf` doctrine lints. - -## (c) Recommended next action - -**Claim `polylogue-cpf.2`** ("Doctrine: writer-class docstring convention + layering check"). - -Why this one specifically, over the many other ready candidates: -- It is in **gate A-trust-floor**, the tool-confirmed active frontier (lowest completion %, and the gate the operator's own priority-reconciliation sweep just re-anchored as top priority). -- It is the **direct sibling of the commit that just landed** (`cpf.1`, same epic, same `wave:1` label, same "small schema/layering lint wired into `devtools lab policy`" shape) — the agent that did `cpf.1` proved the exact pattern this bead needs (add a check, register it under `devtools lab policy`, wire into `devtools verify --lab`, unit-test with a fixture that should fail vs. one that should pass). -- Its acceptance criteria are tight and mechanical: "A file declaring two writer classes fails the check; single-class files pass" — low risk, quick to verify, no architectural ambiguity. -- No blocking dependencies (parent `cpf` is an open epic, but parent-child edges don't block readiness). - -Second-best alternative, if the goal is "clean up what's already in flight" rather than "advance the top-priority gate": **`polylogue-vv2b`** — it's the freshest, most concrete, most fully-specified bead in the tracker (filed minutes before HEAD, names exact file/line locations, reuses a pattern the same session just implemented twice), but it's gate `F-lineage-compaction` / unlabeled, i.e. formally lower priority under the current gate ordering than `cpf.2`/`cpf.3`/`cpf.4`. - -Also worth a two-minute detour regardless of which bead is claimed next: `bd close polylogue-8e1b` — its work already shipped in merged PR #2584 and it's just sitting `in_progress` unclosed. - -## (d) Confidence and evidence used - -High confidence on the state summary (git log + `bd show` on the exact beads referenced by the last four commits gave a coherent, cross-corroborating picture: commit messages, bead `close_reason` text, and the newly-filed `vv2b` bead's dependency edge to `4ts.6` all agree). Medium-high confidence on the "what's next" recommendation — it follows the repo's own explicitly documented policy (gate-ordered priority, `delivery-gate-status.py` as "the gate board") rather than my own guess, but there are 43 other A-trust-floor-ready beads I did not individually inspect, so a different reasonable pick from that same set is defensible. - -Evidence used: `git log --oneline -30` / `--all -10`; `bd ready --json --limit 500` (299 items); `bd show` on `polylogue-cpf`, `cpf.1`–`cpf.4`, `4ts.6`, `4ts`, `1xc.11`, `tf0e`, `vv2b`, `8e1b`; `bd list --status in_progress --json`; and one ordinary repo script read/run, `.agent/tools/delivery-gate-status.py --fresh` / `--gate A-trust-floor --json` (explicitly not under the forbidden `.agent/demos/uplift-two-arm/` path). No MCP/Polylogue archive tools were used (not required — the bd/git evidence was sufficient and directly conclusive). diff --git a/.agent/demos/uplift-two-arm/current/arms/pair2-handoff-pack-output.md b/.agent/demos/uplift-two-arm/current/arms/pair2-handoff-pack-output.md deleted file mode 100644 index f9112f29c2..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair2-handoff-pack-output.md +++ /dev/null @@ -1,31 +0,0 @@ -# Devloop Reconstruction Report — commit 64c079d6e - -## (a) Current devloop state - -HEAD sits at 64c079d6e, the tail of a run of small, tightly-scoped merged PRs (9e5.29 -> 83u.4 -> 83u.6 -> svfj -> 212.7 -> 212.4 -> 212.8). - -The demo-unlock chain (9e5.29 -> svfj -> 212.7 -> {212.1-212.6,212.8 parallel} -> 1vpm.1 (+rxdo.7) -> 212.9) has its foundation fully cleared: 9e5.29, svfj, and 212.7 are all closed. Of the parallel demo set, 212.4 and 212.8 are closed -- 2 of the planned first-wave set are done. - -Gate-board snapshot: A-trust-floor 23% (frontier gate, 43 ready); B-storage-rebuild-bytes 4%; C-read-evidence-contract 2%; F-lineage-compaction 17%; L-external-legibility 13%. - -polylogue-1vpm.1 shows status=in_progress, assignee=Sinity, with no shipped commits yet on the checked-out history -- consistent with "not yet built." - -## (b) Open threads - -1. Demo portfolio (polylogue-212) -- 212.2 ("D1 'The receipts': claim-vs-evidence on a real PR") is the missing piece of the originally-planned first wave and is unblocked (all dependencies closed). 212.1 also implementation-ready. 212.3 blocked on a missing join primitive. 212.5/212.6 explicitly deferred. 212.9 depends on 1vpm.1 and optionally rxdo.7. -2. Lineage epic (polylogue-4ts) -- untouched; only 4ts.3 and 4ts.5 ready, both priority 3. -3. External legibility (polylogue-3tl) -- untouched as an epic, 6 children ready. -4. polylogue-cfk (re-run two-arm uplift experiment) -- open, likely needing an operator-gated/infra-dependent step. -5. polylogue-pj8 -- open, plausibly needs an MCP server reboot. -6. polylogue-rxdo.7 -- needed for the demo chain's optional annotation-loop step. -7. Large untouched P1 backlog under A-trust-floor (the 9e5.* family plus cpf.4). - -## (c) Recommended next action - -polylogue-212.2 -- "D1 'The receipts': claim-vs-evidence on a real PR." It is the one remaining piece of the demo epic's own first wave, fully unblocked, rated A-implementation-ready ("nearly free: all reads exist"), and keeps the demo-unlock chain moving. Lower-risk than picking an undifferentiated P1 trust-floor item, and doesn't require the redeploy/MCP-reboot precondition blocking cfk/pj8. - -Secondary candidate: polylogue-4ts.3 or 4ts.5 to start cracking open the lineage epic. - -## (d) Confidence - -High confidence on state reconstruction. Medium-high on the recommendation -- the ultimate call between continuing the demo thread versus the P1 trust-floor backlog is a judgment call the standing goal itself leaves open. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair2-raw-ref-output.md b/.agent/demos/uplift-two-arm/current/arms/pair2-raw-ref-output.md deleted file mode 100644 index ebd3837340..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair2-raw-ref-output.md +++ /dev/null @@ -1,25 +0,0 @@ -## Current Devloop State - -The repository sits at commit 64c079d6e, the tip of a long, rapid-fire single-day devloop. The devloop is explicitly organized around a delivery-gate sequence (A-trust-floor -> B-storage-rebuild-bytes -> C-read-evidence-contract -> ... -> N-horizon), with bead priority fields recently reconciled (PR #2584, bead polylogue-8e1b, still technically in_progress even though its mechanical sweep already landed). - -The immediately preceding thread of work is the demo portfolio epic (polylogue-212): converting demos into a formal Demo Finding Packet contract (polylogue-212.7, closed) and producing a "first public mini-portfolio" of exactly three packets: D1 receipts (212.2, NOT implemented), D4 behavioral archaeology (212.4, closed), the honesty anti-demo (212.8, closed, HEAD commit). When 212.7 was closed, a follow-up bead polylogue-xyel was filed to implement the real D1-receipts packet. - -Also active: 9e5 audit-lane, f2qv (cost honesty), cpf (six doctrines), 83u (attachment/blob integrity, 83u.2 corrected-and-unclaimed), svfj's follow-up polylogue-xl25 (relocated_lineage/quarantined states). polylogue-1vpm.1 is genuinely in_progress right now. - -## Open Threads - -- D1 receipts demo incomplete (polylogue-xyel, ready, unblocked). -- A real, freshly-discovered correctness bug: polylogue-70qb -- bare find "sessions where " silently ignores the boolean predicate. Discovered mid-authorship of 212.4, documented honestly as a counterexample. No delivery:* gate label yet. -- polylogue-xl25 -- implement relocated_lineage/quarantined states for the block-content-hash citation resolver. -- polylogue-8e1b -- still in_progress, unclosed despite mechanical sweep landing. -- Large standing backlogs at every later delivery gate. - -## Recommended Next Action - -Claim polylogue-xyel (real D1-receipts demo). It is the single named, unblocked loose end from the work sequence that produced the last three commits; small, well-scoped. - -Close second: polylogue-70qb -- a genuine correctness defect in a core surface, numerically higher priority under the reconciled scheme, but not gate-labeled so it fell outside the reconciliation sweep. - -## Confidence - -High confidence on state reconstruction and xyel as the most contextually-continuous next claim. Moderate uncertainty on xyel-vs-70qb prioritization since they represent genuinely different, defensible notions of "next." diff --git a/.agent/demos/uplift-two-arm/current/arms/pair3-handoff-pack-output.md b/.agent/demos/uplift-two-arm/current/arms/pair3-handoff-pack-output.md deleted file mode 100644 index acee41dd45..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair3-handoff-pack-output.md +++ /dev/null @@ -1,30 +0,0 @@ -# Devloop State Reconstruction — polylogue @ 01592e5e9 - -## (a) Current devloop state - -The gate board (A-trust-floor -> N-horizon) shows: A-trust-floor 23% (frontier, 14 closed/43 ready/3 blocked); B-storage-rebuild-bytes 4%; C-read-evidence-contract 2%; D-agent-context-coordination 0%; F-lineage-compaction 17%; I-analytics-experiments 0% closed, 1 wip (1vpm.1); L-external-legibility 13% (demo portfolio + 3tl/cfk live here). - -Two intertwined recent threads: (1) demo-unlock chain 9e5.29 -> svfj -> 212.7 -> 212.4+212.8, all closed; (2) correctness-audit thread: 4ts.4, xnkf, jsy, all closed, jsy landing exactly at HEAD. - -polylogue-1vpm.1 is in_progress, assigned, with a detailed investigation note: the hard identity/extraction problem is already solved by build_run_projection/session_runs, so real remaining scope narrows to five gaps: (1) delegation_kind taxonomy, (2) link_status field, (3) delegations DSL query unit, (4) delegation-card read view, (5) target_kind=delegation for assertions. Left claimed-but-unimplemented "due to time." - -## (b) Open threads - -- Demo portfolio: 4 of 6 parallel demo beads still open -- 212.1, 212.2, 212.3, 212.5. 212.6 blocked by polylogue-tsk (a real classifier bug). 212.9 has a non-blocking related dependency on 1vpm.1 and rxdo.7. -- rxdo.7 itself blocked by rxdo.1. -- pj8 depends on parent s7ae and 37t.4 (itself blocked by 37t.12). Matches the "needs redeploy" postpone caveat. -- cfk structurally unblocked but needs a live paired-arm experiment. -- 3tl -- all 17 children open, uniformly P4 docs/marketing polish. -- 4ts (lineage epic) only has a relates-to link to 38x, no P1/P2 children in flight. -- Trust-floor "storage identity" P1 cluster: 9e5.4, 9e5.5, 9e5.6, 9e5.19 -- all P1, area:storage, all open/ready, none started. jsy and xnkf look like exactly what these audits are meant to produce systematically, discovered incidentally so far. -- 37 P1 beads ready in A-trust-floor overall, dominated by the 9e5.* audit family. - -## (c) Recommended next action - -polylogue-1vpm.1 -- resume and finish the delegation derived-unit implementation. It is the only bead genuinely mid-flight with investigation already paid for and a concrete five-item implementation list recorded on the bead. Sits on the critical path to 212.9. - -Alternative if following the demo-parallel-then-1vpm.1 order strictly: 212.2 or 212.3. Third option if prioritizing trust-floor over demos: polylogue-9e5.6, sharing footprint/evidence base with the just-shipped jsy fix. - -## (d) Confidence and evidence - -High confidence on mechanical facts (bead statuses, dependency graphs, gate percentages). Moderate-to-high on the prioritization judgment. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair3-raw-ref-output.md b/.agent/demos/uplift-two-arm/current/arms/pair3-raw-ref-output.md deleted file mode 100644 index 0536a016c2..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair3-raw-ref-output.md +++ /dev/null @@ -1,28 +0,0 @@ -## Devloop State Report — polylogue @ 01592e5e9 - -### (a) Current state - -The devloop is deep in an extended "trust floor" hardening pass across security, storage-lineage, and evidence-integrity surfaces, with a parallel demo-authoring track validating the query/insights layer against real corpus behavior. The last ~30 commits show a tight, disciplined rhythm: land a fix/feature PR, then immediately land a companion chore(beads) commit that closes the bead with verified findings and files any follow-up debt discovered along the way. - -Most recently shipped: polylogue-jsy (blob hash validation hardening + symlink-check removal), the actions view pairing fix, holding one read transaction across lineage composition, the D4 behavioral archaeology demo, and a new Demo Finding Packet contract + registry lint. - -Two beads are in_progress: polylogue-8e1b (priority reconciliation sweep, work appears already executed, just needs closing) and polylogue-1vpm.1 (delegation-derived-unit materializer, mid-investigation). - -### (b) Open threads - -1. Attachment/blob integrity program (polylogue-83u) — 83u.2 still open and unclaimed. -2. Security & privacy epic (polylogue-kwsb) — jsy just closed under it, epic remains open. -3. Six-doctrines program (polylogue-cpf) — cpf.2/cpf.3/cpf.4 all still open/ready. -4. Cost/usage honesty (polylogue-f2qv) — multiple children ready. -5. Large read-only audit lane (polylogue-9e5) — ~20+ still-open children. -6. A freshly-discovered, unfixed correctness bug filed at 2026-07-09T00:27:55Z. - -### (c) Recommended next bead: polylogue-70qb - -"Bare find sessions where ignores the boolean predicate, returns unfiltered list" — discovered during the just-merged 212.4 demo work. Reproduces identically for a plain field predicate and for a seq() predicate, so it isn't SEQ-specific — it's the bare-find/boolean-entry-form code path silently defaulting to an unfiltered listing. - -Reasons: real, exactly-reproduced, user-facing correctness bug; cheap to fix and verify (three one-line CLI repro commands already given); zero open blocking dependencies; directly protects trust in the CLI's primary query surface. - -### (d) Confidence - -High on state reconstruction. Moderate-high on the specific recommendation. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair4-handoff-pack-output.md b/.agent/demos/uplift-two-arm/current/arms/pair4-handoff-pack-output.md deleted file mode 100644 index 4108da828a..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair4-handoff-pack-output.md +++ /dev/null @@ -1,27 +0,0 @@ -# Devloop Reconstruction Report — polylogue @ a2ee55ec4 - -## (a) Current devloop state - -At a2ee55ec4, the backlog is organized into 15 lettered delivery gates that must clear roughly in order. Gate A-trust-floor is the active frontier: 9 closed, 46 ready, 1 in-progress, 3 blocked (~15% complete). No later gate has meaningfully started. - -Two beads are literally in_progress at this exact commit: polylogue-9e5.29 (claimed by Sinity, priority 1, gate A-trust-floor) and polylogue-8e1b (the priority-reconciliation bead itself, not yet formally closed even though its PR landed). - -## (b) Open threads - -1. Demo-unlock chain: 9e5.29 is already claimed and in progress -- it's the live head of this chain, not a cold start. Of 212.7's dependencies, four already closed, three remain open: 9e5.29 (wip), svfj (open), 212.9 (open). Note: 212.7 genuinely depends on 212.9, which conflicts with the informal ordering placing 212.9 last -- this should be reconciled. -2. Trust-floor P1 backlog large and only partly overlapping with the demo chain: 43 open/in-progress P1 beads gate-wide. -3. Lineage epic (4ts): 4ts.1/4ts.2 already closed historically; remaining scope (4ts.3-4ts.7) all still open. -4. 3tl epic: 5 children already closed historically; remaining ~10 children all priority 4. -5. pj8: layers 1+2 already merged; explicit remaining blocker is a redeploy for the live MCP server to pick up new prompts. -6. cfk and s7ae remain open with no children closed yet. -7. 8e1b still in_progress despite its PR having merged as this very commit. - -## (c) Recommended next action - -Continue and close polylogue-9e5.29. Already claimed and in-progress; priority 1 in the active frontier gate; literal head-blocker of the demo-unlock chain. Concrete, anchored design (insights/rigor.py, RigorFieldContract). Checkable AC. - -Immediately after: close 8e1b (PR already merged). Then svfj as the next chain link, while separately flagging the 212.7<->212.9 ordering conflict. - -## (d) Confidence and evidence - -High confidence in the state reconstruction. The 212.7<->212.9 dependency-vs-narrative-order conflict is flagged as genuinely uncertain, a real graph inconsistency worth resolving rather than assuming the "212.9 last" framing is authoritative. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair4-raw-ref-output.md b/.agent/demos/uplift-two-arm/current/arms/pair4-raw-ref-output.md deleted file mode 100644 index ba97a981e7..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair4-raw-ref-output.md +++ /dev/null @@ -1,26 +0,0 @@ -## Devloop reconstruction report — polylogue @ a2ee55ec4 - -### (a) Current devloop state - -The backlog runs through a staged delivery-gate sequence A-trust-floor -> ... -> N-horizon. The tip commit (a2ee55ec4, "reconcile priority field with delivery-gate order") just finished a mechanical sweep reassigning every bead's priority to match this gate order -- 288 of 387 open/gate-labeled beads repointed. Pure bookkeeping, closing out #8e1b. - -Preceding it: temporal-provenance doctrine work (cpf.5/cpf.6, both closed); a wave of timeless-session correctness fixes across search/usage-timeline/CLI query-unit engine/work-events; security hardening (XSS fixes, browser-capture host-admission gate, bearer token requirement); devtools/verify robustness; the devloop scaffold retired in favor of Beads. - -Gate-board snapshot: A-trust-floor is the active frontier -- 23% closed (14/43/3). Every later gate essentially untouched. - -### (b) Open threads - -- A-trust-floor exit bar far from met -- most ready items are large audit/epic-shaped tasks (9e5.* audit family, f2qv.* cost-honesty chain, cpf.* doctrine chain). -- Many ready A-trust-floor items are actually readiness=D-horizon-ready in their own notes -- only a subset are truly A-implementation-ready. -- cpf epic: 3/6 children closed; remaining three (cpf.2/cpf.3/cpf.4) need more local inspection. -- f2qv cost/usage-honesty chain fully open, each blocking the next. -- polylogue-cuxz filed as a follow-on from the timeless-session fix wave. -- Blocked items: polylogue-27m, polylogue-b0b/b0b.1, polylogue-9e5.16. - -### (c) Recommended next action - -Claim polylogue-jnj.5 -- "Route ops reset --session/--source through the mutation contract." A-implementation-ready, zero blocking dependencies, blocks nothing. A genuine, narrowly-scoped security bug: ops reset writes a tombstone before the dry-run/confirmation branch runs, so a typo could silently suppress data without --yes. Child of the kwsb security/privacy epic. Concrete testable AC. - -### (d) Confidence and evidence - -High confidence on state summary. Medium-high on the specific recommendation -- another equally valid choice could reasonably be argued. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair5-handoff-pack-output.md b/.agent/demos/uplift-two-arm/current/arms/pair5-handoff-pack-output.md deleted file mode 100644 index 94816740f8..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair5-handoff-pack-output.md +++ /dev/null @@ -1,26 +0,0 @@ -# Devloop State Reconstruction — polylogue (as of / since commit 697470661) - -## (a) Current devloop state - -The summary's endpoint was commit 697470661. The repository's actual tip has moved substantially further -- origin/master is now at 91eb09549, 19 commits ahead. Since the checkout, many closures have landed: svfj, 212.7, 212.4, 212.8, 4ts.4, 4ts.6, cpf.1, jsy, xnkf all closed; 4ts.3 investigated and unclaimed. - -The delivery-gate frontier is unchanged: A-trust-floor still leads at 23%. - -## (b) Open threads - -1. Demo-unlock chain well past where the summary left it. Only 212.4 and 212.8 done of 212.1-212.6/212.8; the rest remain open and parallelizable. -2. polylogue-1vpm.1 is the live critical-path item and appears actively claimed right now. Sole dependency gating 212.9. Bead notes show a pre-implementation investigation narrowing scope to five concrete items. It was investigated-and-unclaimed once already, then re-claimed and set back to in_progress locally, uncommitted -- a worktree tracking master tip is present and not detached, consistent with an agent actively working this bead right now. -3. polylogue-4ts epic has three open children left (4ts.3, 4ts.5, 4ts.7) plus the parent. -4. polylogue-3tl, polylogue-cfk, polylogue-pj8 -- all still open at the epic level, essentially untouched. -5. polylogue-83u.2 remains open, corrected findings but still unclaimed. -6. polylogue-8e1b left in_progress even though its mechanical sweep already landed -- stale claim bookkeeping. - -## (c) Recommended next action - -Continue/finish polylogue-212.9's blocker, polylogue-1vpm.1 -- the standing goal's next unfinished chain link, already has a substantially narrowed, concrete implementation scope from its own investigation notes. - -Caveat: evidence strongly suggests this bead is already claimed by a concurrently running session (in_progress, uncommitted, recent timestamp, matching live non-detached worktree). A fresh agent should verify claim/lock status before starting. If actively held, pick up an independent demo item instead (212.1 or 212.2). - -## (d) Confidence and evidence - -High confidence on mechanical facts. Moderate confidence on the "concurrent agent is right now working 1vpm.1" inference -- reasonable but not certain. diff --git a/.agent/demos/uplift-two-arm/current/arms/pair5-raw-ref-output.md b/.agent/demos/uplift-two-arm/current/arms/pair5-raw-ref-output.md deleted file mode 100644 index 88c8584726..0000000000 --- a/.agent/demos/uplift-two-arm/current/arms/pair5-raw-ref-output.md +++ /dev/null @@ -1,25 +0,0 @@ -## Devloop Reconstruction Report — polylogue @ 697470661 - -### (a) Current devloop state - -A long-running, multi-agent Beads-driven backlog burn against a 15-gate delivery sequence. Gate-board: A-trust-floor (active frontier) 23% complete -- 14 closed, 43 ready, 3 blocked. All later gates 0-17%. - -Total backlog: 533 beads. Priority mechanically re-derived from gate membership; 37 ready priority-1 beads, all labeled delivery:A-trust-floor. - -Immediately preceding commits were working polylogue-83u (attachment/blob evidence integrity, a B-gate epic): 83u.6 and 83u.4 shipped/closed; 83u.2 investigated in depth, found to require a real architecture decision with no confirmed live target, deliberately left open/unclaimed. HEAD is an unrelated shipped fix (content-hash citation anchors). - -### (b) Open threads - -1. 83u.2 stuck pending a design decision -- correctly left unclaimed. -2. A-trust-floor gate (37 ready P1 beads) under-invested relative to the 83u work that just happened. Sub-clusters: f2qv cost/usage-honesty (f2qv.1 just closed, .2-.5 open); cpf doctrine/spine (cpf.5/.6 closed, .2/.3/.4 open); 9e5 large audit epic (~20 ready sub-beads); 38x reconciliation hub connecting several threads. -3. Two beads in_progress: polylogue-8e1b (mechanical, already reflected in commit) and polylogue-1vpm.1 (delegation-derived-unit materializer) -- the latter shows a concurrent worktree session actively working it. - -### (c) Recommended next action - -Claim polylogue-f2qv.2 -- "Codex disjoint-lane normalizer." On-gate, picks up the exact cluster whose sibling f2qv.1 just closed. Fully specified, concrete AC with a named cross-check and a regression-guard test reproducing the historical 7.69x naive-sum bug. Closes out one leg of the still-open 38x reconciliation hub. No unresolved dependencies. - -Runner-up: polylogue-9e5.23. - -### (d) Confidence and evidence - -Moderate-high on "what happened". Moderate on "what to do next" -- a judgment call among ~37 equally-gated ready beads. diff --git a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair1.json b/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair1.json deleted file mode 100644 index 4c99be68d2..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair1.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "written_before_arms_ran": true, - "written_at": "2026-07-09T03:20:00Z", - "checkpoint": "immediately after PR #2601 (cpf.1 timestamp-doctrine lint) merged to master, before 4ts.6 work began", - "checkpoint_commit": "6c12e9234", - "ground_truth_next_action": "Claim and implement polylogue-4ts.6 (lineage composition silently truncates transcripts; surface a completeness signal) -- add lineage_complete/lineage_truncation_reason to ArchiveSessionEnvelope and a matching async LineageCompleteness return type, with regression tests for depth-limit and dangling-branch-point cases.", - "ground_truth_open_threads": [ - "polylogue-4ts.3 (subagent auto-compaction misclassification) -- investigated, correctly scoped as needing dispatch-layer surgery (polylogue/sources/dispatch.py), left unclaimed", - "polylogue-1vpm.1 (delegation derived unit) -- investigated, found existing session_runs/build_run_projection infrastructure covers most of the hard part, left unclaimed as still a real multi-file feature", - "polylogue-83u.2 (attachment re-acquisition) -- investigated, found the static prework packet had wrong anchors, left unclaimed", - "the standing /bead-loop goal explicitly lists: trust-floor/storage-identity P1s, lineage epic (4ts), pj8, 3tl, cfk, plus a demo-unlock chain (9e5.29 -> svfj -> 212.7 -> 212.1-212.6/212.8 -> 1vpm.1 -> 212.9)" - ], - "ground_truth_caveats": [ - "This is a live, still-running session -- the 'subject' and the ground-truth author are the same agent, which is a real construct-validity limit (matches jxe's own documented limit: 'hidden prior knowledge and tool availability are not fully controlled').", - "Both arms are dispatched as isolated Agent-tool subagents from within the same session, not fully separate Claude Code processes like the original jxe pilot's agent_id-tracked arms -- a weaker isolation guarantee than the original protocol.", - "n=1 pilot; cannot estimate general uplift, matching jxe.2's own caveat." - ] -} diff --git a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair2.json b/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair2.json deleted file mode 100644 index b719363443..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair2.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "written_before_arms_ran": true, - "checkpoint": "immediately after PR #2592 (212.8 honesty anti-demo bead close) merged, before 1vpm.1/4ts.4 work began", - "checkpoint_commit": "64c079d6e", - "ground_truth_next_action": "Investigate polylogue-1vpm.1 (delegation derived unit) -- claimed, investigated, found existing session_runs/build_run_projection infrastructure covers most of the hard part, left unclaimed pending a fresh session's fuller implementation.", - "ground_truth_open_threads": [ - "polylogue-4ts.4 (torn-transcript transaction race) -- not yet started at this checkpoint, becomes the very next shipped item", - "polylogue-xnkf (actions view duplicate-tool_id fan-out) -- not yet started, ships several items later", - "the standing /bead-loop goal's demo-unlock chain: 212.1-212.6 parallel set still has 4 unclaimed items after 212.4/212.8" - ] -} diff --git a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair3.json b/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair3.json deleted file mode 100644 index 6a8ae6c1c1..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair3.json +++ /dev/null @@ -1,11 +0,0 @@ -{ - "written_before_arms_ran": true, - "checkpoint": "immediately after PR #2600 (jsy blob hash validation bead close) merged, before cpf.1 work began", - "checkpoint_commit": "01592e5e9", - "ground_truth_next_action": "Claim and implement polylogue-cpf.1 (doctrine lint: reject TEXT timestamps in new durable DDL) -- a new devtools lab policy check.", - "ground_truth_open_threads": [ - "polylogue-jsy just shipped (blob hash hardening + symlink check removal)", - "the cpf epic (doctrine landing) has several sibling lints not yet started: cpf.2 (writer-class docstring), cpf.3 (deny-lexicon tripwire)", - "4ts.4/4ts.6/xnkf lineage and storage fixes have not yet happened at this checkpoint" - ] -} diff --git a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair4.json b/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair4.json deleted file mode 100644 index 36b2721930..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair4.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "checkpoint": "immediately after PR #2584 (bead priority/gate-order reconciliation) merged, before 9e5.29 work began", - "checkpoint_commit": "a2ee55ec4", - "ground_truth_next_action": "Claim and implement polylogue-9e5.29 (number-over-empty gates / RigorFieldContract mechanism).", - "ground_truth_open_threads": [ - "the just-merged priority reconciliation repriced 288 beads to track the delivery-gate order", - "83u.4/83u.6 attachment acquisition work not yet started at this checkpoint", - "svfj/212.7/212.4/212.8/4ts.4/xnkf/jsy/cpf.1/4ts.6 all still fully in the future from this checkpoint" - ] -} diff --git a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair5.json b/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair5.json deleted file mode 100644 index ac4cad9392..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/ground-truth-pair5.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "checkpoint": "immediately after PR #2588 (svfj block content-hash citation anchors) merged, before its bead-close/212.7 work began", - "checkpoint_commit": "697470661", - "ground_truth_next_action": "Close polylogue-svfj bead (file relocated_lineage/quarantined follow-up), then claim and implement polylogue-212.7 (Demo Finding Packet contract + registry lint).", - "ground_truth_open_threads": [ - "svfj just shipped: block content-hash citation anchors + typed resolver, index schema v24->v25", - "the demo-unlock chain (9e5.29 -> svfj -> 212.7 -> 212.1-212.6/212.8 parallel -> 1vpm.1 -> 212.9) has 9e5.29 and svfj done, 212.7 is next", - "83u.2 was just corrected/unclaimed with verified findings one commit prior" - ] -} diff --git a/.agent/demos/uplift-two-arm/current/metrics/rubric.json b/.agent/demos/uplift-two-arm/current/metrics/rubric.json deleted file mode 100644 index 2307ff67bf..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/rubric.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "scoring": "0-10 per arm, integer", - "criteria": [ - "answer_correctness_against_ground_truth: does the reconstructed state summary match what actually happened?", - "open_thread_coverage: how many of the real open threads did it find?", - "next_action_specificity: did it name a specific, defensible next bead, not a vague direction?", - "caveat_honesty: did it flag its own confidence limits and evidence basis rather than overclaiming?" - ], - "note_to_judge": "Score each arm independently against the ground truth file. Do not assume either arm is the 'better' one by construction -- judge only on the merits of what's written. If both arms recommend a different but individually well-justified next action, note that explicitly rather than penalizing divergence from the ground truth's own next_action pick, since the ground truth's pick is one reasonable answer, not the only correct one." -} diff --git a/.agent/demos/uplift-two-arm/current/metrics/score.json b/.agent/demos/uplift-two-arm/current/metrics/score.json deleted file mode 100644 index 2d9b6975bd..0000000000 --- a/.agent/demos/uplift-two-arm/current/metrics/score.json +++ /dev/null @@ -1,55 +0,0 @@ -{ - "pair_1": { - "checkpoint_commit": "6c12e9234", - "blinding_compromised": true, - "blinding_compromise_reason": "Arm B (pack) self-referentially wrote 'This matches the pack's shipped-work list exactly', revealing its own identity to the judge", - "judge_1": { - "raw_ref_total": 29, - "handoff_pack_total": 33, - "winner": "handoff_pack", - "primary_differentiator": "open_thread_coverage -- pack arm surfaced pj8/3tl/cfk/212.x chain explicitly named in ground truth; raw-ref arm did not" - }, - "judge_2": { - "raw_ref_total": 22, - "handoff_pack_total": 35, - "winner": "handoff_pack", - "note": "explicitly checked for blinding leak, found none" - } - }, - "pair_2": { - "checkpoint_commit": "64c079d6e", - "note": "both arms converged on the same real next action (D1 receipts) under different bead-id labels (xyel vs 212.2)", - "judge": { - "raw_ref_total": 25, - "handoff_pack_total": 31, - "winner": "handoff_pack" - } - }, - "pair_3": { - "checkpoint_commit": "01592e5e9", - "judge": { - "raw_ref_total": 31, - "handoff_pack_total": 19, - "winner": "raw_ref", - "note": "pack arm confidently misstated 4ts.4/xnkf as already closed at this checkpoint (they were not yet), and missed the cpf epic entirely" - } - }, - "pair_5": { - "checkpoint_commit": "697470661", - "judge": { - "raw_ref_total": 12, - "handoff_pack_total": 36, - "winner": "handoff_pack", - "note": "raw-ref arm mischaracterized the checkpoint commit itself (svfj) as unrelated, missing the entire demo-unlock chain thread; pack arm correctly honest about reasoning past its own checkpoint" - } - }, - "pair_4": { - "checkpoint_commit": "a2ee55ec4", - "judge": { - "raw_ref_total": 17, - "handoff_pack_total": 32, - "winner": "handoff_pack", - "note": "ground truth verified via git log: 9e5.29 was indeed claimed/in-progress at this exact checkpoint and is the literal next thing worked; pack arm caught this exactly, raw-ref arm missed it entirely and recommended an unrelated bead" - } - } -} \ No newline at end of file diff --git a/.agent/demos/uplift-two-arm/current/pairs.json b/.agent/demos/uplift-two-arm/current/pairs.json deleted file mode 100644 index e0e92adba1..0000000000 --- a/.agent/demos/uplift-two-arm/current/pairs.json +++ /dev/null @@ -1,18 +0,0 @@ -{ - "n": 5, - "protocol": "identical paired two-arm design per pair: raw-ref arm (session ref + live query access only, forbidden from any pre-built summary) vs handoff-pack arm (a bounded pre-built context summary + live query access), same task, same checkpoint commit, isolated worktree subagents. Ground truth for each pair written before either arm ran.", - "pairs": [ - {"pair": 1, "checkpoint_commit": "6c12e9234", "raw_ref_score": 29, "handoff_pack_score": 33, "winner": "handoff_pack", "note": "blinding compromised (pack arm self-referenced 'the pack'); scored by 2 independent judges, both favored pack (33/29 and 35/22)"}, - {"pair": 2, "checkpoint_commit": "64c079d6e", "raw_ref_score": 25, "handoff_pack_score": 31, "winner": "handoff_pack", "note": "both arms converged on the same real next action under different bead-id labels"}, - {"pair": 3, "checkpoint_commit": "01592e5e9", "raw_ref_score": 31, "handoff_pack_score": 19, "winner": "raw_ref", "note": "pack arm confidently misstated two beads as already-closed when they were not yet at this checkpoint"}, - {"pair": 4, "checkpoint_commit": "a2ee55ec4", "raw_ref_score": 17, "handoff_pack_score": 32, "winner": "handoff_pack", "note": "ground truth verified via git log; pack arm correctly identified the literal in-progress bead, raw-ref missed it"}, - {"pair": 5, "checkpoint_commit": "697470661", "raw_ref_score": 12, "handoff_pack_score": 36, "winner": "handoff_pack", "note": "raw-ref arm mischaracterized the checkpoint commit itself as unrelated"} - ], - "aggregate": { - "pairs_won_by_handoff_pack": 4, - "pairs_won_by_raw_ref": 1, - "sign_test": "4/5 -- directionally favors handoff_pack but NOT statistically significant at n=5 (binomial p~0.375 two-tailed under a null of no difference)", - "mean_raw_ref": 22.8, - "mean_handoff_pack": 30.2 - } -} diff --git a/.agent/demos/uplift-two-arm/current/report.md b/.agent/demos/uplift-two-arm/current/report.md deleted file mode 100644 index 880160e6b7..0000000000 --- a/.agent/demos/uplift-two-arm/current/report.md +++ /dev/null @@ -1,53 +0,0 @@ -# Handoff-Pack Uplift Re-Run (n=5 pilot) - -Successor to campaign `polylogue-jxe`, which closed diagnostic-negative (raw-ref 8/10 vs handoff-pack 5/10, n=1, attributed to packet staleness). This re-run tests whether a *fresh* pack (regenerated at continuation time) plus live query access beats raw-ref plus live query access, per the protocol decision recorded on `polylogue-cfk`. - -## Claim - -Directional evidence (4 of 5 pairs) that a fresh, bounded context summary given alongside live query access produces a better devloop-state reconstruction than live query access alone, at this sample size. **Not a publishable claim** — n=5 is a de-noising step per the original protocol's own staged design (n=1 pilot → n≥3 de-noise → n=12-20 publishable), not the final tier. - -**What this experiment actually measures.** Raw access is a strict superset of what's in the pack — nothing in the pack is exclusive information the raw-ref arm couldn't in principle derive itself, since the pack is a compressed derivative of the same underlying `bd`/`git` state. So this is not a test of whether information is *missing* from raw access; it's a test of whether front-loaded synthesis beats from-scratch synthesis *within the bounded effort of one task attempt*, and at what correctness cost. Read the 4 pack wins below as "raw-ref, given one shot, produced a shallower synthesis than an arm that started pre-digested" rather than "raw-ref was blind to something" — and read pair 3 (the one raw-ref win) as the sharpest evidence of the corresponding cost: a pack can encode a synthesis that has drifted from live truth, and an arm that trusts it uncritically can assert something false that an arm forced to re-derive from live state every time structurally cannot. - -## Method - -5 pairs, each at a distinct real checkpoint commit in this session's own devloop history. Each pair: two isolated Agent-tool subagents (worktree-isolated), one given only `bd`/`git`/repo-source access ("raw-ref"), one given the same access plus a hand-written bounded summary of recent shipped work and standing goals ("handoff-pack"). Both attempt the same task: reconstruct devloop state and recommend a specific next bead. A ground-truth file was written *before* dispatching each pair's arms, recording the checkpoint and (where knowable) what actually happened next. A separate, blinded judge subagent scored each pair against its ground truth without being told which arm was which. - -## Corpus - -This session's own live devloop (2026-07-09), five checkpoints: commits `6c12e9234`, `64c079d6e`, `01592e5e9`, `a2ee55ec4`, `697470661`. - -## Findings - -| Pair | Checkpoint | raw-ref | handoff-pack | Winner | -|---|---|---|---|---| -| 1 | 6c12e9234 | 29 (judge_1); 22 (judge_2) | 33 (judge_1); 35 (judge_2) | handoff-pack (both judges) | -| 2 | 64c079d6e | 25 | 31 | handoff-pack | -| 3 | 01592e5e9 | 31 | 19 | raw-ref | -| 4 | a2ee55ec4 | 17 | 32 | handoff-pack | -| 5 | 697470661 | 12 | 36 | handoff-pack | - -Handoff-pack wins 4 of 5 pairs. Mean scores: raw-ref 22.8/40, handoff-pack 30.2/40. - -**Where handoff-pack won**, the differentiator was consistently *open-thread coverage* and *avoiding a total miss on the checkpoint's own significance* — e.g. in pair 5, the raw-ref arm called the checkpoint commit itself "an unrelated shipped fix" when it was the load-bearing event; in pair 4, raw-ref never identified the literally-in-progress bead as in-progress. The pack arms consistently surfaced named epics/threads (pj8, 3tl, cfk, the 212.x chain, the lineage epic) that the raw-ref arms, working from `bd`/`git` alone, more often missed or under-weighted. - -**Where raw-ref won (pair 3)**, the pack arm made a confident, uncaveated factual error — asserting two beads (`4ts.4`, `xnkf`) were "already closed" at a checkpoint where they had not yet shipped. This is a genuine, real failure mode: a pack can encode information that's subtly ahead of (or stale relative to) the actual checkpoint, and an arm that trusts it uncritically can assert something false with high confidence. The raw-ref arm, forced to derive everything from the live state at that exact commit, could not make this specific class of error. - -## Specimens - -All 10 arm transcripts and 6 judge verdicts are in `arms/*.md` and recorded inline above; raw scores in `pairs.json`. `pairs.json`'s pair-1 row uses judge_1's scores (29/33) as the aggregate table's value with judge_2's (22/35) not separately averaged in, since both judges agreed on the winning direction and picking either alone doesn't change the sign-test outcome; a future revision could instead average the two. - -## Counterexamples - -Pair 3 is the explicit counterexample: a well-resourced arm can still fail when its input encodes inaccurate-for-this-checkpoint information and it doesn't verify against live state before asserting. This is exactly the "packet staleness" failure mode the original `jxe` campaign identified — not fully eliminated by giving the pack arm live query access, since it can simply not use that access to catch a factual error the pack primed it to assume. - -## Limits - -- **n=5 is not a publishable sample.** The protocol's own staged design calls for n≥12-20 for a publishable uplift claim. This result should be read as "worth continuing to n=12-20," not as a conclusion. -- **Correlated subject and rater.** All 5 checkpoints come from the same session/devloop, authored, ground-truthed, and (for the initial pilot pair) partially scored within the same overall session that generated the subject matter. This is a real construct-validity limit shared with the original `jxe` pilot. -- **Weaker isolation than the original protocol.** The original `jxe`/`jxe.2` pilot used fully separate Claude Code CLI invocations with distinct tracked `agent_id`s per arm. This re-run uses `Agent`-tool subagents in isolated git worktrees within the same orchestrating session — real process/context isolation, but not as strong a guarantee as fully independent CLI sessions. -- **Pair 1's blind was compromised** (the pack arm's own text self-referenced "the pack"), discovered after dispatch. A second independent judge, explicitly told to check for and account for such a leak, still scored the same direction (35 vs 22), which is reassuring but doesn't retroactively fix the methodology for that pair — later pairs' arm prompts were corrected to forbid self-referential process commentary. -- **This pilot did not test the "packet staleness" root cause directly** — i.e., it didn't measure whether `qt3`'s fast regeneration + `yps` freshness metadata actually keep a pack fresh under load; it assumed the packs handed to the pack arms were fresh (hand-written summaries, not regenerated via the production pack-generation pipeline). A full re-run per the original protocol would generate the pack through the actual pipeline at continuation time and verify freshness metadata, which this pilot did not do. - -## Reproduce - -The five ground-truth files (`metrics/ground-truth*.json`), the `pairs.json` aggregate, and every arm/judge transcript are committed under this directory. Re-running with the production pack-generation pipeline (rather than hand-written summaries) and extending to n≥12 pairs, drawn from genuinely independent subjects rather than one session's own consecutive checkpoints, is the next step toward a publishable result — tracked as a follow-up on `polylogue-cfk` rather than assumed complete by this pilot. diff --git a/.circleci/config.yml b/.circleci/config.yml index b824682990..b7f98e3694 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -3,12 +3,12 @@ version: 2.1 # CircleCI free-tier CI while GitHub Actions is billing-locked (and a cheap # second opinion afterwards). Credit budget: ~30k credits/month on the free # plan. Design keeps the per-push cost low and the heavy work nightly: -# - quick-gate (PRs + master pushes): public claims + `devtools verify -# --quick` itself (not a hand-picked subset of its steps -- see +# - quick-gate (PRs + master pushes): `devtools verify --quick` itself (not a +# hand-picked subset of its steps -- see # polylogue-ze5i: this job used to reimplement ruff/mypy/render-all as # separate `run` steps and silently diverged from what # devtools/verify.py's `if not commit:` block actually gates, so -# schema-versioning/classifier-fingerprints/demo-tour-freshness and +# schema-versioning and # every other check added to that block over time never ran in real CI # even though `.github/workflows/ci.yml` referenced them -- that GHA # workflow does not execute, only CircleCI does). No pytest; the test @@ -17,10 +17,9 @@ version: 2.1 # only. # - lab-policies (scheduled pipeline named "nightly"): the `lab policy *` # checks that stay --lab-only because they scan the whole Beads/backlog -# corpus rather than the current diff (backlog-hygiene, bead-graph) or +# corpus rather than the current diff (bead-graph) or # are otherwise not cheap/deterministic enough for every push -# (timestamp-doctrine, insight-honesty, demo-packet-registry, docs-drift, -# campaign-archive-boundaries). Runs nightly so a policy that fails +# (timestamp-doctrine, insight-honesty). Runs nightly so a policy that fails # continuously is at least visible somewhere, per polylogue-ze5i AC2, # instead of only reachable via a human remembering `devtools verify # --lab`. @@ -68,9 +67,6 @@ jobs: - mypy-v1-{{ .Branch }} - mypy-v1-master - mypy-v1- - - run: - name: Public claims gate - command: ~/.local/bin/uv run devtools verify public-claims --json | tee /tmp/polylogue-public-claims.json - run: name: Structured PR scope carrier command: | @@ -95,10 +91,6 @@ jobs: key: mypy-v1-{{ .Branch }}-{{ epoch }} paths: - .mypy_cache - - store_artifacts: - path: /tmp/polylogue-public-claims.json - destination: diagnostics - when: always - store_artifacts: path: /tmp/polylogue-pr-scope.log destination: diagnostics @@ -113,11 +105,9 @@ jobs: - bootstrap # Each currently-red check runs with `when: always` so one failure # doesn't hide the pass/fail signal of the others in the same job -- - # backlog-hygiene and bead-graph are known-red until the pre-existing - # backlog debt they report is triaged (tracked separately; see - # polylogue-ze5i notes), and that must not mask a genuine regression - # in timestamp-doctrine/insight-honesty/demo-packet-registry/ - # docs-drift/campaign-archive-boundaries going unnoticed. + # bead-graph reports repository-wide graph state rather than the current + # diff, and that must not mask a genuine regression + # in timestamp-doctrine/insight-honesty going unnoticed. - run: name: lab policy timestamp-doctrine command: ~/.local/bin/uv run devtools lab policy timestamp-doctrine @@ -127,24 +117,8 @@ jobs: command: ~/.local/bin/uv run devtools lab policy insight-honesty when: always - run: - name: lab policy demo-packet-registry - command: ~/.local/bin/uv run devtools lab policy demo-packet-registry - when: always - - run: - name: lab policy docs-drift - command: ~/.local/bin/uv run devtools lab policy docs-drift - when: always - - run: - name: lab policy campaign-archive-boundaries - command: ~/.local/bin/uv run devtools lab policy campaign-archive-boundaries - when: always - - run: - name: lab policy backlog-hygiene (known backlog debt, see polylogue-ze5i) - command: ~/.local/bin/uv run devtools lab policy backlog-hygiene - when: always - - run: - name: lab policy bead-graph (known backlog debt, see polylogue-ze5i) - command: ~/.local/bin/uv run devtools lab policy bead-graph + name: lab policy bead-graph + command: ~/.local/bin/uv run devtools lab policy bead-graph --export .beads/issues.jsonl when: always full-suite: diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index ba3b626d42..5cbb22a5a4 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -20,6 +20,8 @@ _Exact commands run and any manual validation performed._ ## Bead disposition matrix +_Required for Bead-scoped PRs. Delete this section when the carrier uses `scope_kind=self_contained`._ + | Assigned Bead | Whole-Bead disposition | Evidence refs | Named successor for residual work | | --- | --- | --- | --- | | `polylogue-...` | satisfied / partial / deferred / superseded | `test:...`, `command:...` | `polylogue-...` or n/a | diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4f53387317..1d80d4e141 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -30,7 +30,6 @@ jobs: python-version: "3.14" - run: uv sync --extra dev --frozen - run: uv run devtools render all --check - - run: uv run devtools verify public-claims --json # An index schema bump merged without its lifecycle.py delta declaration # downgrades every live generation to a full raw replay (polylogue-9rw0). - run: uv run devtools lab policy schema-versioning @@ -57,12 +56,8 @@ jobs: POLYLOGUE_PYTEST_BASETEMP_ROOT: ${{ runner.temp }}/polylogue-pytest test: - # The full non-integration suite runs single-process under coverage - # instrumentation across three Python versions; 45 min was on the edge and - # legs were being cut mid-run (reported as "cancelled"), so the test gate - # could never go green. 90 min gives headroom; passing runs cost their - # actual ~45-50 min regardless of the ceiling. - # + # The full suite runs under coverage instrumentation. Keep a generous job + # ceiling so pytest's per-test timeout can report the actual stalled node. # This heavy gate is intentionally OFF the per-PR path (operator decision): # the ~18 min wait on every PR was not worth it. It still runs post-merge on # master (regression net) and on demand via workflow_dispatch, but no longer diff --git a/.tokeignore b/.tokeignore index 847d7a3f99..005105a5ea 100644 --- a/.tokeignore +++ b/.tokeignore @@ -28,9 +28,8 @@ webui/package-lock.json webui/src/api/generated.ts webui/src/generated/ -# Generated/rendered doc surfaces and demo artifacts. +# Generated/rendered doc surfaces. docs/cli-reference.md -docs/examples/demo-tour/report.json # Rendered byte-identical copy of polylogue/agent_integration/data/deep-reference.md; # the source copy stays counted, this duplicate does not. diff --git a/CLAUDE.md b/CLAUDE.md index 7321576791..406af0f9ef 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -197,7 +197,7 @@ Two evolution regimes, enforced by `devtools lab policy schema-versioning`: non-semantic delta upgrades an existing generation **in place** through `index_fast_forward_plan()` on connect. Only a `SEMANTIC_REPARSE` delta — one whose result depends on parser semantics — routes to - `polylogue ops reset --index && polylogued run`. A bump without a declaration + `polylogue ops maintenance rebuild-index`. A bump without a declaration is a policy violation, not a free rebuild: the lint fails and the archive silently falls back to full raw replay. @@ -293,7 +293,7 @@ scripts under any name. (Its evidence may still sit in a gitignored, untracked `.agent/archive/devloop-2026-07/` in some working checkouts — polylogue-ocby — it is not part of the repo and a fresh clone will not have it.) Repo agent conventions: `.agent/CONVENTIONS.md`; run -`devtools lab policy acceptance-contracts` and `devtools lab policy bead-graph` before shipping bead-state deltas. Run `bd prime` when task +`devtools lab policy bead-graph --export .beads/issues.jsonl` before shipping bead-state deltas. Run `bd prime` when task context, ready work, blockers, or project memory matter. Use `bd ready --json`, `bd show --json`, `bd update --claim --json`, `bd close --reason "…" --json`. Create linked Beads issues for discovered @@ -354,8 +354,12 @@ Don't treat CI as the first verification pass — anticipate failures locally. See [Schema regimes](#schema-regimes-durability-keyed). Durable tiers → numbered additive migration + backup manifest; derived tiers → edit canonical DDL + -rebuild plan (`polylogue ops reset --index && polylogued run`), never an upgrade -helper (`devtools lab policy schema-versioning` rejects them). +an explicitly declared lifecycle delta. Non-semantic deltas may use the +clone-validated `index_fast_forward_plan()` route; semantic deltas require +`polylogue ops maintenance rebuild-index`. Ad hoc open-path upgrade code is not +an accepted third route. `devtools lab policy schema-versioning` validates the +declarations, durable migration slots, and same-version benign-DDL shapes; it +does not infer architecture from helper names. ### Multi-lane / merge-train tooling — use these, don't reinvent the discipline by hand @@ -377,8 +381,10 @@ workflow, not optional conveniences — use them at the point named, every time: - **Before claiming a batch of ready beads**: `devtools workspace bead-cluster` — footprint/overlap/contention clustering so overlapping-file beads land on one branch instead of colliding across parallel lanes. -- **When dispatching a worktree-isolated lane**: `devtools workspace lane-brief - --out ` for its dispatch prompt (footprint, prior art, hazards). +- **When dispatching a worktree-isolated lane**: give the worker the current + Bead acceptance criteria, verified file ownership, relevant prior commits, + concrete non-goals, and exact verification commands directly. Do not insert + a generated Markdown packet between the current evidence and the worker. - **Before opening a non-draft PR for a Bead lane**: render the versioned carrier with `devtools workspace pr-scope render --input `, put it in the PR body beside the human whole-Bead disposition matrix, then run @@ -407,7 +413,7 @@ workflow, not optional conveniences — use them at the point named, every time: then auto-records a receipt if none is fresh for the current head sha (running `--command`, default `devtools verify`), BLOCKs the merge on any `merge-gate check` failure (no fresh receipt, stale receipt, nonzero - exit, a changed head-bound scope attestation, or an unacked review comment newer than the head commit), strips a + exit, a changed head-bound scope attestation, an unresolved GitHub review thread, or a changes-requested review), strips a doubled `(#N) (#N)` squash-subject suffix, then runs the actual `gh pr merge --squash`. `--dry-run` runs every check without merging; `--with-verify` immediately runs and records the merge-train's terminal @@ -420,9 +426,6 @@ workflow, not optional conveniences — use them at the point named, every time: `devtools workspace merge-gate record/check` commands still exist for ad hoc receipt inspection, but the merge action itself should go through `workspace merge`. -- Sizing/triage input: `devtools workspace backlog-calibration` for - lead-time/discovery/staleness distributions before deciding batch size. - If you build a new tool in this family, add it here in the same sentence — a tool without a line in this file is a tool the next session won't use. @@ -561,7 +564,7 @@ Core loop: [Verification](#verification--testmon-inner-loop-never-blanket-run). - `devtools test ` — focused pytest through the managed harness. - `devtools lab …` — executable schema/provider/pipeline/lane checks. -- `devtools workspace …` — task history, frontier, worktree-gc, evidence. +- `devtools workspace …` — task history, worktree-gc, evidence. Adding a devtools command: add a `CommandSpec` to `devtools/command_catalog.py`, implement in `devtools/.py`, run `devtools render devtools-reference`. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index cfb48e2ca3..87c535c270 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -117,8 +117,10 @@ changes require a copy-forward design and explicit operator consent; do not hide them behind a routine migration. Derived tiers (`index.db`, `embeddings.db`) are rebuildable products. They do -not get in-place migration chains. A PR that bumps their schema edits the -canonical DDL and provides a **rebuild/blue-green plan**: +do not get ad hoc in-place migration chains. A PR that bumps their schema edits +the canonical DDL, declares the delta in `storage/sqlite/lifecycle.py`, and +provides either a clone-validated non-semantic fast-forward plan or a +**rebuild/blue-green plan** for semantic changes: - which user-visible archive operation triggers rebuild/re-acquisition from source (e.g. `polylogue ops reset --index && polylogued run` for index-tier @@ -136,8 +138,11 @@ index additions that can be grouped into one schema bump, and do not call a full reingest necessary unless the changed semantics actually require replaying source rows. -The policy lint (`devtools lab policy schema-versioning`) rejects derived-tier -upgrade helpers while allowing numbered durable-tier SQL migrations. +The policy lint (`devtools lab policy schema-versioning`) validates derived-tier +lifecycle declarations and clone-safe benign-DDL shapes while allowing numbered +durable-tier SQL migrations. It does not attempt to classify Python helpers by +their names; ad hoc open-path upgrades remain unsupported because the runtime +routes derived changes only through declared fast-forward plans or rebuilds. ## Versioning and Releases diff --git a/README.md b/README.md index f072b63cc9..8182ffc453 100644 --- a/README.md +++ b/README.md @@ -9,15 +9,13 @@ Live documentation

- Polylogue archives AI conversations and coding-agent runs from multiple tools in one searchable local archive. It imports supported histories from ChatGPT, Claude and Claude Code, Codex, Gemini, Hermes, and other sources, then exposes sessions, messages, tool calls and results, branches, subagents, usage, and costs through a CLI, Python API, local HTTP reader, and MCP server. -By default, the archive stays on your machine. The author's archive contains -more than **18,000 sessions and 4.7 million messages**. +By default, the archive stays on your machine. [Getting started](docs/getting-started.md) | [Live documentation](https://sinity.github.io/polylogue/) | [Demo](docs/demos.md) | [Architecture](docs/architecture.md) | [CLI reference](docs/cli-reference.md) @@ -222,7 +220,6 @@ Start with the task-oriented guides below. The complete documentation map is in | [Getting Started](docs/getting-started.md) | Install Polylogue, create an archive, and run a first query. | | [Installation](docs/installation.md) | Package, source-checkout, Nix, and managed deployment options. | | [Demos and Proofs](docs/demos.md) | Run the private-data-free tour and see what each demo establishes. | -| [Proof Artifacts](docs/proof-artifacts.md) | Links between public claims and reproducible checks. | | [Architecture](docs/architecture.md) | Storage, data flow, and component responsibilities. | | [Code Navigation](docs/code-navigation.md) | Find the owning package, runtime path, and verification for a code change. | | [Search & Query](docs/search.md) | Search syntax, filters, action queries, ranking, and output formats. | diff --git a/TESTING.md b/TESTING.md index 77eeee105a..3973786889 100644 --- a/TESTING.md +++ b/TESTING.md @@ -222,10 +222,9 @@ and `.cache/verify/current-pytest-output.log` to see the active/latest test node selected/deselected node IDs, collection duration, slowest setup/call/teardown phases, captured output, and termination reason if a focused run stalls. -For optional lane, mutation-campaign, and benchmark inventories, see -[docs/test-quality-workflows.md](docs/test-quality-workflows.md). Those registries are -secondary navigation over executable checks; the source of truth for behavior is -pytest plus the concrete `polylogue`/`devtools` commands they invoke. +Optional lane, mutation-campaign, and benchmark commands remain discoverable +through `devtools --help`; pytest and the concrete commands are the behavioral +authority. ### Known limitation: collection-time-only imports are invisible to testmon @@ -245,37 +244,12 @@ inherent to how testmon (and coverage-context-based selective testing in general) works — it is **not** dependency-graph staleness, and running `devtools verify --seed-testmon` does not fix it. -Confirmed reproducible (2026-07-12, polylogue-csg7) with an isolated, -freshly-seeded testmon run scoped to exactly one test file: after -`TESTMON_DATAFILE= pytest --testmon --testmon-noselect -tests/unit/devtools/test_verify_manifests.py`, `devtools/manifest_models.py` -still has 0 `file_fp` rows even though a `--cov` run over the same test file -reports 80% statement coverage on that module — all of it from Pydantic -model/field declarations executed when `devtools.verify_manifests` is -imported at module-collection time; every uncovered line is inside a -`@field_validator`/`@model_validator` method body, which only runs when -`validate_manifest()` is actually called (no test in that file calls it). - -Cross-referencing a full-suite `coverage.json` (`--cov=polylogue`) against -`file_fp` filenames finds **95 files** under `polylogue/` with nonzero -covered statements but zero testmon dependency rows — largely `*_models.py`, -`types.py`, `protocols.py`, `enums.py`, and `api/contracts/*.py`, i.e. modules -whose test-suite touch points are import-only. Query: - -```python -import json, sqlite3 - -cov = json.load(open(".cache/coverage/coverage.json"))["files"] -tm = {r[0] for r in sqlite3.connect(".cache/testmon/testmondata").execute("SELECT DISTINCT filename FROM file_fp")} -gaps = [(f, d["summary"]["covered_lines"]) for f, d in cov.items() if d["summary"]["covered_lines"] > 0 and f not in tm] -``` - **Blast radius:** the default `devtools verify` gate (`--testmon --testmon-forceselect`) is the only local pre-merge signal for a change scoped to one of these files — `devtools test ` forwards a literal pytest selection and is not testmon-aware, so it does not share this gap (point it at the file's *owning test module*, not the changed source file). -A change confined to one of these 95 files can select zero tests locally and +A change confined to one of these files can select zero tests locally and still report a clean `devtools verify`. The heavy full-suite `devtools verify coverage` CI job (`.github/workflows/ci.yml`) does not use testmon selection and still catches such a regression, but only **post-merge** (it is @@ -288,9 +262,7 @@ upstream tool behavior. When changing a file that is purely declarative do not trust "0 tests selected" from the default `devtools verify` gate as proof of safety; run the file's owning test module directly with `devtools test `, and rely on `mypy --strict` (already in the default gate) -to catch structural regressions in `TypedDict`/protocol shapes. See -polylogue-csg7 for the investigation and a follow-up tracking item for making -this gap machine-checkable. +to catch structural regressions in `TypedDict`/protocol shapes. ## Test Suite Layout @@ -450,15 +422,13 @@ checks HTTP/DOM/API contracts rather than screenshots. ```bash devtools bench mutation list devtools bench mutation run -devtools bench mutation index ``` Policy: - keep the committed mutmut configuration broad; narrow work happens through focused campaigns -- write local artifacts under `.local/mutation-campaigns/` -- rebuild the mutation index after a campaign run +- write per-run JSON artifacts under `.local/mutation-campaigns/` ## Protected Files diff --git a/browser-extension/README.md b/browser-extension/README.md index 226957d1ce..ae8504e98f 100644 --- a/browser-extension/README.md +++ b/browser-extension/README.md @@ -232,7 +232,7 @@ pages the badge shows grey and no data is sent. |---------|-------| | Badge is grey | Navigate to a supported page (chatgpt.com, claude.ai, or grok.com) | | Badge is red | Receiver is not running — start `polylogued browser-capture serve` | -| Captures not appearing in archive | Run `polylogue check` to verify the daemon is ingesting | +| Captures not appearing in archive | Run `polylogue ops doctor --runtime --daemon` to verify the daemon is ingesting | | Popup says `stale` | The receiver has a newer spool artifact than the indexed archive. Leave the daemon running and inspect the debug log request id if it does not converge. | | Popup says `dom` / `dom_degraded` | Reload the provider page, wait for the conversation to load fully, then capture again so the native app payload can be observed. | | A button click seems ineffective | The button status line should show Working/Done/Failed. Open **Debug log** and export JSON if the state does not change. | diff --git a/devtools/acceptance_route_registry.py b/devtools/acceptance_route_registry.py deleted file mode 100644 index 22f8bb18de..0000000000 --- a/devtools/acceptance_route_registry.py +++ /dev/null @@ -1,64 +0,0 @@ -"""Authoritative route registry for acceptance-contract dispatch.""" - -from __future__ import annotations - -from collections.abc import Mapping -from pathlib import Path -from typing import Any - -from polylogue.core.json import JSONDecodeError -from polylogue.core.json import loads as json_loads - -_REGISTRY_PATH = Path(__file__).parents[1] / "docs" / "plans" / "beads-acceptance-route-registry.json" - - -class AcceptanceRouteRegistryError(ValueError): - """Raised when the committed route registry is missing or malformed.""" - - -def load_registry(path: Path = _REGISTRY_PATH) -> dict[str, dict[str, Any]]: - if not path.is_file(): - raise AcceptanceRouteRegistryError(f"{path}: acceptance route registry is missing") - try: - document = json_loads(path.read_text(encoding="utf-8")) - except JSONDecodeError as exc: - raise AcceptanceRouteRegistryError(f"{path}: invalid JSON: {exc}") from exc - if not isinstance(document, Mapping) or document.get("schema_version") != 1: - raise AcceptanceRouteRegistryError(f"{path}: unsupported acceptance route registry schema") - entries = document.get("routes") - if not isinstance(entries, list): - raise AcceptanceRouteRegistryError(f"{path}: routes must be a list") - registry: dict[str, dict[str, Any]] = {} - for entry in entries: - if not isinstance(entry, Mapping): - raise AcceptanceRouteRegistryError(f"{path}: route entries must be objects") - identifier = entry.get("identifier") - if not isinstance(identifier, str) or not identifier: - raise AcceptanceRouteRegistryError(f"{path}: route entry has no identifier") - if identifier in registry: - raise AcceptanceRouteRegistryError(f"{path}: duplicate route identifier {identifier!r}") - registry[identifier] = dict(entry) - return registry - - -def resolve_route( - identifier: object, *, registry: Mapping[str, Mapping[str, Any]] | None = None -) -> Mapping[str, Any] | None: - if not isinstance(identifier, str): - return None - routes = load_registry() if registry is None else registry - return routes.get(identifier) - - -def registry_digest(registry: Mapping[str, Mapping[str, Any]] | None = None) -> str: - """Return the digest of the sorted route-entry payload.""" - import hashlib - - from polylogue.core.json import dumps as json_dumps - - routes = load_registry() if registry is None else registry - payload = json_dumps( - [routes[identifier] for identifier in sorted(routes)], - sort_keys=True, - ) - return hashlib.sha256(payload.encode("utf-8")).hexdigest() diff --git a/devtools/action_contract_report.py b/devtools/action_contract_report.py deleted file mode 100644 index 85b2a7d210..0000000000 --- a/devtools/action_contract_report.py +++ /dev/null @@ -1,96 +0,0 @@ -"""Render the public CLI action-contract report for generated docs.""" - -from __future__ import annotations - -from collections.abc import Iterable - -from devtools.render_cli_output_schemas import SCHEMAS, CliOutputSchema -from polylogue.operations.action_contracts import ACTION_CONTRACTS - - -def _path_cell(path: tuple[str, ...]) -> str: - return f"`polylogue {' '.join(path)}`" - - -def _code_cell(value: str) -> str: - return f"`{value}`" - - -def _list_cell(values: Iterable[str]) -> str: - rendered = tuple(_code_cell(value) for value in values) - return ", ".join(rendered) if rendered else "-" - - -def _format_set(values: frozenset[str]) -> tuple[str, ...]: - order = {"human": 0, "json": 1, "ndjson": 2} - return tuple(sorted(values, key=order.__getitem__)) - - -def _contract_rows() -> list[str]: - rows = [ - "| Action | Effect | Target | Input | Cardinality | Safety | Formats | Destinations | Confirm | Select | Machine envelope | Guards | Next actions | Completion |", - "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", - ] - for contract in ACTION_CONTRACTS: - rows.append( - " | ".join( - ( - _path_cell(contract.path), - _code_cell(contract.effect), - _code_cell(contract.target), - _code_cell(contract.input_unit), - _code_cell(contract.cardinality), - _code_cell(contract.safety_level), - _list_cell(_format_set(contract.formats)), - _list_cell(contract.destination_support), - _code_cell(contract.confirmation_command) if contract.confirmation_command else "-", - _code_cell(contract.selection_command) if contract.selection_command else "-", - _code_cell(contract.machine_envelope), - _list_cell(contract.guards), - _list_cell(contract.next_actions), - _code_cell(contract.completion_context) if contract.completion_context else "-", - ) - ).join(("| ", " |")) - ) - return rows - - -def _schema_rows(schemas: tuple[CliOutputSchema, ...] = SCHEMAS) -> list[str]: - rows = [ - "| Schema | Model | Surfaces |", - "| --- | --- | --- |", - ] - for schema in schemas: - rows.append( - " | ".join( - ( - f"`{schema.name}`", - _code_cell(schema.model.__name__), - "
".join(_code_cell(surface) for surface in schema.surfaces), - ) - ).join(("| ", " |")) - ) - return rows - - -def render_action_contract_report() -> str: - """Return the generated CLI action/output contract report.""" - lines = [ - "## Public Action Contracts", - "", - "This section is generated from `polylogue.operations.action_contracts.ACTION_CONTRACTS`.", - "It records the public action floor, not every utility command in the Click tree.", - "", - *_contract_rows(), - "", - "## Published Machine Output Schemas", - "", - "This section is generated from `devtools.render_cli_output_schemas.SCHEMAS`.", - "The schema files live under `docs/schemas/cli-output/`.", - "", - *_schema_rows(), - ] - return "\n".join(lines).rstrip() - - -__all__ = ["render_action_contract_report"] diff --git a/devtools/affordance_usage.py b/devtools/affordance_usage.py index 97bd131128..28b21c0c7f 100644 --- a/devtools/affordance_usage.py +++ b/devtools/affordance_usage.py @@ -210,32 +210,15 @@ def _demo_summary(report: dict[str, Any]) -> dict[str, Any]: "index_db": report["index_db"], "snapshot_identity": report["snapshot_identity"], "index_schema_version": report["index_schema_version"], - "claim": ( - "Polylogue can compare agent affordance usage across normalized action evidence " - "without summing unlike tool-name spellings or provider-specific call shapes." - ), - "non_claim": ( - "This is not a human-quality utility evaluation of any particular tool family. " - "It measures captured usage evidence, failure signals, and coverage gaps; " - "usefulness still requires qualitative review of session context and outcomes." - ), - "proof_report": { - "report_version": report["report_version"], - "action_scope": report["action_scope"], - "recent_window_days": report["recent_window_days"], - "patterns": report["patterns"], - "detail_patterns": report["detail_patterns"], - "top_families": summary["top_families"], - "recent_top_families": summary["recent_top_families"], - "surface_inventory_summary": surface_summary, - }, - "caveats": [ - "Counts describe captured action evidence, not independent proof of user benefit.", - "Failure rates are provider-reported tool-result signals where available; missing outcome structure is not success.", - "Recent windows are adoption-sensitive and can legitimately differ from all-time counts.", - "Zero captured agent use is not enough to remove operator-only surfaces; those rows carry an operator-only caveat.", - ], - "source_files": [ + "report_version": report["report_version"], + "action_scope": report["action_scope"], + "recent_window_days": report["recent_window_days"], + "patterns": report["patterns"], + "detail_patterns": report["detail_patterns"], + "top_families": summary["top_families"], + "recent_top_families": summary["recent_top_families"], + "surface_inventory_summary": surface_summary, + "files": [ "affordance-usage.report.json", "family-counts.csv", "evidence-kind-counts.csv", diff --git a/devtools/artifact_graph.py b/devtools/artifact_graph.py index 250b56bc86..ded75732b1 100644 --- a/devtools/artifact_graph.py +++ b/devtools/artifact_graph.py @@ -6,17 +6,13 @@ import json import sys -from devtools.scenario_coverage import build_runtime_scenario_coverage from polylogue.artifacts.graph import build_artifact_graph def render_artifact_graph(*, as_json: bool) -> str: graph = build_artifact_graph() - coverage = build_runtime_scenario_coverage() if as_json: - payload = graph.to_dict() - payload["scenario_coverage"] = coverage.to_dict() - return json.dumps(payload, indent=2) + return json.dumps(graph.to_dict(), indent=2) lines: list[str] = ["Artifact Paths:"] for path in graph.paths: @@ -48,60 +44,15 @@ def render_artifact_graph(*, as_json: bool) -> str: lines.append(f"- {target.name} [{target.mode.value}/{target.category.value}]: {target.description}") lines.append(f" - artifacts: {artifacts}") lines.append(f" - operations: {operations}") - lines.append("") - lines.append("Runtime Path Coverage:") - for path_name, path_coverage in sorted(coverage.paths.items()): - status = "complete" if path_coverage.complete else "partial" - rendered_refs = ", ".join(f"{ref.source}:{ref.name}" for ref in path_coverage.refs) or "—" - lines.append(f"- {path_name} [{status}]: {rendered_refs}") - if path_coverage.uncovered_artifacts: - lines.append(f" - uncovered artifacts: {', '.join(path_coverage.uncovered_artifacts)}") - if path_coverage.uncovered_operations: - lines.append(f" - uncovered operations: {', '.join(path_coverage.uncovered_operations)}") - lines.append("") - lines.append("Runtime Scenario Coverage:") - if not coverage.artifacts and not coverage.operations: - lines.append("- none") - return "\n".join(lines) - for artifact_name, refs in sorted(coverage.artifacts.items()): - rendered_refs = ", ".join(f"{ref.source}:{ref.name}" for ref in refs) - lines.append(f"- artifact {artifact_name}: {rendered_refs}") - for operation_name, refs in sorted(coverage.operations.items()): - rendered_refs = ", ".join(f"{ref.source}:{ref.name}" for ref in refs) - lines.append(f"- operation {operation_name}: {rendered_refs}") - for target_name, refs in sorted(coverage.maintenance_targets.items()): - rendered_refs = ", ".join(f"{ref.source}:{ref.name}" for ref in refs) - lines.append(f"- maintenance {target_name}: {rendered_refs}") - if coverage.uncovered_artifacts: - lines.append("- uncovered artifacts: " + ", ".join(coverage.uncovered_artifacts)) - if coverage.uncovered_operations: - lines.append("- uncovered operations: " + ", ".join(coverage.uncovered_operations)) - if coverage.uncovered_maintenance_targets: - lines.append("- uncovered maintenance targets: " + ", ".join(coverage.uncovered_maintenance_targets)) return "\n".join(lines) def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--json", action="store_true", help="Emit the artifact graph as JSON.") - parser.add_argument( - "--strict", - action="store_true", - help="Fail with exit 1 if any runtime artifact, declared operation, maintenance target, or operation path is uncovered.", - ) args = parser.parse_args(argv) sys.stdout.write(render_artifact_graph(as_json=args.json)) sys.stdout.write("\n") - if args.strict: - coverage = build_runtime_scenario_coverage() - if ( - coverage.uncovered_artifacts - or coverage.uncovered_operations - or coverage.uncovered_declared_operations - or coverage.uncovered_maintenance_targets - or any(not path.complete for path in coverage.paths.values()) - ): - return 1 return 0 diff --git a/devtools/authored_scenario_catalog.py b/devtools/authored_scenario_catalog.py deleted file mode 100644 index 6474976966..0000000000 --- a/devtools/authored_scenario_catalog.py +++ /dev/null @@ -1,92 +0,0 @@ -"""Central authored catalog for verification lanes and campaigns.""" - -from __future__ import annotations - -from dataclasses import dataclass -from functools import lru_cache - -from devtools.benchmark_catalog import ( - BenchmarkCampaignEntry, - build_benchmark_entries, - build_synthetic_benchmark_entries, -) -from devtools.lane_models import LaneEntry -from devtools.mutation_catalog import MutationCampaignEntry, build_mutation_entries -from devtools.validation_catalog import build_validation_lane_entries -from polylogue.scenarios import ( - CorpusScenario, - ScenarioProjectionEntry, - ScenarioProjectionSource, - compile_projection_entries, -) -from polylogue.schemas.operator.inference import list_inferred_corpus_scenarios - - -@dataclass(frozen=True) -class AuthoredScenarioCatalog: - validation_lanes: tuple[LaneEntry, ...] - mutation_campaigns: tuple[MutationCampaignEntry, ...] - benchmark_campaigns: tuple[BenchmarkCampaignEntry, ...] - synthetic_benchmark_campaigns: tuple[BenchmarkCampaignEntry, ...] - inferred_corpus_scenarios: tuple[CorpusScenario, ...] - - @property - def contract_lanes(self) -> tuple[LaneEntry, ...]: - return tuple(entry for entry in self.validation_lanes if entry.category == "contract") - - @property - def live_lanes(self) -> tuple[LaneEntry, ...]: - return tuple(entry for entry in self.validation_lanes if entry.category == "live") - - @property - def composite_lanes(self) -> tuple[LaneEntry, ...]: - return tuple(entry for entry in self.validation_lanes if entry.category == "composite") - - def projection_sources(self) -> tuple[ScenarioProjectionSource, ...]: - result: list[ScenarioProjectionSource] = [] - result.extend(self.validation_lanes) # type: ignore[arg-type] - result.extend(self.mutation_campaigns) - result.extend(self.benchmark_campaigns) - result.extend(self.synthetic_benchmark_campaigns) - result.extend(self.inferred_corpus_scenarios) - return tuple(result) - - def validation_lane_index(self) -> dict[str, LaneEntry]: - return {entry.name: entry for entry in self.validation_lanes} - - def mutation_campaign_index(self) -> dict[str, MutationCampaignEntry]: - return {entry.name: entry for entry in self.mutation_campaigns} - - def benchmark_campaign_index(self) -> dict[str, BenchmarkCampaignEntry]: - return {entry.name: entry for entry in self.benchmark_campaigns} - - def synthetic_benchmark_campaign_index(self) -> dict[str, BenchmarkCampaignEntry]: - return {entry.name: entry for entry in self.synthetic_benchmark_campaigns} - - def compile_projection_entries(self) -> tuple[ScenarioProjectionEntry, ...]: - return tuple( - sorted( - compile_projection_entries(self.projection_sources()), - key=lambda item: (item.source_kind.value, item.name), - ) - ) - - -@lru_cache(maxsize=1) -def get_authored_scenario_catalog() -> AuthoredScenarioCatalog: - # Repo-authored surfaces (rendered docs, verify gates) must be hermetic: - # pin the schema registry to the in-repo SCHEMA_DIR so operator-local - # inferred schemas under data_home()/schemas cannot leak into generated - # docs or flip `render --check` per machine. - from polylogue.schemas.registry import SCHEMA_DIR, SchemaRegistry - - return AuthoredScenarioCatalog( - validation_lanes=build_validation_lane_entries(), - mutation_campaigns=build_mutation_entries(), - benchmark_campaigns=build_benchmark_entries(), - synthetic_benchmark_campaigns=build_synthetic_benchmark_entries(), - inferred_corpus_scenarios=list_inferred_corpus_scenarios(registry=SchemaRegistry(storage_root=SCHEMA_DIR)), - ) - - -__all__ = ["AuthoredScenarioCatalog", "get_authored_scenario_catalog"] diff --git a/devtools/backlog_calibration.py b/devtools/backlog_calibration.py deleted file mode 100644 index 47b32f171d..0000000000 --- a/devtools/backlog_calibration.py +++ /dev/null @@ -1,444 +0,0 @@ -"""backlog-calibration: measured duration/discovery models over the bead corpus. - -Fits the numbers a backlog-execution plan needs from the actual bead history -(and optionally the merged-PR history) instead of guessing them: - - - closed-bead lead-time percentiles, overall and split by priority, type, - epic-child vs standalone, and dependency degree; - - a right-censoring-honest survival view (fraction of a >=14d-old cohort - closed within 1/3/7/14 days) -- closed-only medians are survivorship- - biased and this section is the corrective; - - close-reason classification (worked vs already-satisfied / obsolete / - duplicate / misframed) with median age-at-closure per class -- the - "verify before dispatching" economy; - - discovery-vs-drain dynamics: created and closed per day, the - created-per-close ratio, and net backlog growth; - - optionally, PR open->merge latency by changed-file bucket from a - `gh pr list --json` dump. - -Calibrated against 2026-07 history for the backlog-execution design -(/realm/inbox/polylogue-audits-2026-07-31/backlog-execution-design.html); -registered as a devtools command so the model can be re-fitted as the corpus -grows instead of going stale like the guesses it replaced. - -Usage: - # Fresh export (bd export -o under the hood): - devtools workspace backlog-calibration - - # From an existing export / in tests: - devtools workspace backlog-calibration --input beads.jsonl - - # Include PR merge-latency calibration: - # gh pr list --state merged --limit 4000 \ - # --json number,createdAt,mergedAt,additions,deletions,changedFiles > prs.json - devtools workspace backlog-calibration --prs prs.json - - # Machine-readable: - devtools workspace backlog-calibration --json -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -import tempfile -from collections import Counter, defaultdict -from collections.abc import Sequence -from datetime import UTC, datetime -from pathlib import Path -from re import IGNORECASE -from re import compile as re_compile -from typing import Any - -BeadDict = dict[str, Any] - -_DAY_SECONDS = 86400.0 -_PERCENTILES = (10, 25, 50, 75, 90, 95) -_SURVIVAL_WINDOWS_DAYS = (1, 3, 7, 14) -_COHORT_MIN_AGE_DAYS = 14.0 - -# Close-reason classes, checked in order; first match wins. "worked" is the -# fall-through. Regexes over free text are advisory classification, not truth. -_CLOSE_REASON_CLASSES: tuple[tuple[str, Any], ...] = ( - ("duplicate", re_compile(r"\bdup(?:e|licate)?\b", IGNORECASE)), - ( - "already-satisfied", - re_compile(r"\balready\b|\bsatisfied by\b|\bfixed by\b|\bfixed in\b", IGNORECASE), - ), - ( - "obsolete", - re_compile(r"\bobsolete\b|\bsuperseded\b|\bno longer\b|\bstale\b|\bmoot\b", IGNORECASE), - ), - ( - "misframed", - re_compile(r"\bmisframed\b|\binvalid\b|\bnot a bug\b|\bworking as\b", IGNORECASE), - ), -) -NO_IMPLEMENTATION_CLASSES = frozenset(c for c, _ in _CLOSE_REASON_CLASSES) - - -def _parse_ts(value: Any) -> datetime | None: - if not isinstance(value, str) or not value: - return None - try: - parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) - except ValueError: - return None - if parsed.tzinfo is None: - parsed = parsed.replace(tzinfo=UTC) - return parsed - - -def _percentile(sorted_values: list[float], pct: float) -> float: - if not sorted_values: - raise ValueError("empty distribution") - if len(sorted_values) == 1: - return sorted_values[0] - rank = (len(sorted_values) - 1) * pct / 100.0 - lower = int(rank) - if lower >= len(sorted_values) - 1: - return sorted_values[-1] - frac = rank - lower - return sorted_values[lower] + (sorted_values[lower + 1] - sorted_values[lower]) * frac - - -def summarize_days(values: list[float]) -> dict[str, Any]: - """Percentile summary of a list of durations expressed in days.""" - if not values: - return {"n": 0} - ordered = sorted(values) - summary: dict[str, Any] = {"n": len(ordered)} - for pct in _PERCENTILES: - summary[f"p{pct}_days"] = round(_percentile(ordered, pct), 3) - summary["max_days"] = round(ordered[-1], 3) - return summary - - -def classify_close_reason(reason: str | None) -> str: - """Classify a free-text close reason; 'worked' is the fall-through.""" - text = reason or "" - for name, pattern in _CLOSE_REASON_CLASSES: - if pattern.search(text): - return name - return "worked" - - -def _lead_days(bead: BeadDict) -> float | None: - created = _parse_ts(bead.get("created_at")) - closed = _parse_ts(bead.get("closed_at")) - if created is None or closed is None or bead.get("status") != "closed": - return None - return (closed - created).total_seconds() / _DAY_SECONDS - - -def _epic_child_ids(beads: list[BeadDict]) -> frozenset[str]: - children: set[str] = set() - for bead in beads: - for dep in bead.get("dependencies") or []: - if isinstance(dep, dict) and dep.get("type") == "parent-child": - children.add(str(bead.get("id"))) - return frozenset(children) - - -def _split_summaries(closed: list[tuple[BeadDict, float]], key: Any) -> dict[str, dict[str, Any]]: - groups: dict[str, list[float]] = defaultdict(list) - for bead, lead in closed: - groups[str(key(bead))].append(lead) - return {name: summarize_days(values) for name, values in sorted(groups.items())} - - -def _survival(beads: list[BeadDict], as_of: datetime) -> dict[str, Any]: - cohort = [ - bead - for bead in beads - if (created := _parse_ts(bead.get("created_at"))) is not None - and (as_of - created).total_seconds() / _DAY_SECONDS >= _COHORT_MIN_AGE_DAYS - ] - - def _fractions(members: list[BeadDict]) -> dict[str, Any]: - row: dict[str, Any] = {"n": len(members)} - for window in _SURVIVAL_WINDOWS_DAYS: - closed_in = sum(1 for bead in members if (lead := _lead_days(bead)) is not None and lead <= window) - row[f"closed_within_{window}d_pct"] = round(100.0 * closed_in / len(members), 1) if members else None - return row - - by_priority = { - f"P{priority}": _fractions([b for b in cohort if b.get("priority") == priority]) for priority in range(5) - } - return { - "cohort_min_age_days": _COHORT_MIN_AGE_DAYS, - "overall": _fractions(cohort), - "by_priority": by_priority, - } - - -def _discovery(beads: list[BeadDict]) -> dict[str, Any]: - created_per_day: Counter[str] = Counter() - closed_per_day: Counter[str] = Counter() - for bead in beads: - created = _parse_ts(bead.get("created_at")) - closed = _parse_ts(bead.get("closed_at")) - if created is not None: - created_per_day[created.date().isoformat()] += 1 - if closed is not None: - closed_per_day[closed.date().isoformat()] += 1 - days = sorted(set(created_per_day) | set(closed_per_day)) - if not days: - return {"days": [], "note": "no dated beads"} - # The first day of a corpus is typically a bulk import, not discovery. - import_day = days[0] - series: list[dict[str, Any]] = [ - { - "day": day, - "created": created_per_day.get(day, 0), - "closed": closed_per_day.get(day, 0), - "net": created_per_day.get(day, 0) - closed_per_day.get(day, 0), - } - for day in days - ] - post = [row for row in series if row["day"] != import_day] - created_total = sum(int(row["created"]) for row in post) - closed_total = sum(int(row["closed"]) for row in post) - nets: list[float] = sorted(float(row["net"]) for row in post) - return { - "import_day_excluded": import_day, - "days": series, - "post_import_created": created_total, - "post_import_closed": closed_total, - "created_per_close": (round(created_total / closed_total, 2) if closed_total else None), - "median_net_per_day": (round(_percentile(nets, 50), 1) if nets else None), - "net_negative_days": sum(1 for net in nets if net < 0), - "post_import_day_count": len(post), - } - - -def _close_reasons(closed: list[tuple[BeadDict, float]]) -> dict[str, Any]: - with_reason = [(bead, lead) for bead, lead in closed if bead.get("close_reason")] - classes: dict[str, list[float]] = defaultdict(list) - for bead, lead in with_reason: - classes[classify_close_reason(bead.get("close_reason"))].append(lead) - no_impl = sum(len(v) for name, v in classes.items() if name in NO_IMPLEMENTATION_CLASSES) - return { - "closed_with_reason": len(with_reason), - "no_implementation_pct": (round(100.0 * no_impl / len(with_reason), 1) if with_reason else None), - "classes": {name: summarize_days(values) for name, values in sorted(classes.items())}, - } - - -_PR_FILE_BUCKETS: tuple[tuple[int, int, str], ...] = ( - (1, 3, "1-2"), - (3, 6, "3-5"), - (6, 11, "6-10"), - (11, 31, "11-30"), - (31, 10**9, "31+"), -) - - -def _pr_latency(prs: list[dict[str, Any]]) -> dict[str, Any]: - rows: list[tuple[int, float]] = [] - for pr in prs: - created = _parse_ts(pr.get("createdAt")) - merged = _parse_ts(pr.get("mergedAt")) - files = pr.get("changedFiles") - if created is None or merged is None or not isinstance(files, int): - continue - rows.append((files, (merged - created).total_seconds() / 3600.0)) - buckets = { - label: summarize_days([hours / 24.0 for files, hours in rows if lo <= files < hi]) - for lo, hi, label in _PR_FILE_BUCKETS - } - return { - "n": len(rows), - "overall_latency_hours": { - k: (round(v * 24.0, 3) if isinstance(v, float) else v) - for k, v in summarize_days([hours / 24.0 for _, hours in rows]).items() - }, - "by_changed_files_days": buckets, - "note": ( - "open->merge latency measures the merge train, not implementation: " - "PRs in this repo open after the work is done" - ), - } - - -def build_report( - beads: list[BeadDict], - *, - as_of: datetime, - prs: list[dict[str, Any]] | None = None, -) -> dict[str, Any]: - closed = [(bead, lead) for bead in beads if (lead := _lead_days(bead)) is not None] - epic_children = _epic_child_ids(beads) - corpus_age_days = None - created_times = [t for bead in beads if (t := _parse_ts(bead.get("created_at")))] - if created_times: - corpus_age_days = round((as_of - min(created_times)).total_seconds() / _DAY_SECONDS, 1) - report: dict[str, Any] = { - "as_of": as_of.isoformat(), - "population": { - "total": len(beads), - "by_status": dict(Counter(str(b.get("status")) for b in beads)), - "corpus_age_days": corpus_age_days, - "censoring_note": ( - "closed-only lead times are right-censored by corpus age and " - "survivorship-biased by still-open beads; read the survival " - "section for population-honest fractions" - ), - }, - "closed_lead_days": { - "overall": summarize_days([lead for _, lead in closed]), - "by_priority": _split_summaries(closed, lambda b: f"P{b.get('priority')}"), - "by_type": _split_summaries(closed, lambda b: b.get("issue_type")), - "by_epic_membership": _split_summaries( - closed, - lambda b: "epic-child" if str(b.get("id")) in epic_children else "standalone", - ), - "by_dependency_degree": _split_summaries( - closed, - lambda b: min(int(b.get("dependency_count") or 0), 2), - ), - }, - "survival": _survival(beads, as_of), - "close_reasons": _close_reasons(closed), - "discovery": _discovery(beads), - } - if prs is not None: - report["pr_merge_latency"] = _pr_latency(prs) - return report - - -def _load_beads_jsonl(path: Path) -> list[BeadDict]: - beads: list[BeadDict] = [] - for line_number, line in enumerate(path.read_text().splitlines(), start=1): - if not line.strip(): - continue - try: - record = json.loads(line) - except json.JSONDecodeError as exc: - raise SystemExit(f"{path}:{line_number}: not valid JSON ({exc})") from exc - if isinstance(record, dict): - beads.append(record) - return beads - - -def _export_beads() -> list[BeadDict]: - with tempfile.NamedTemporaryFile(suffix=".jsonl", prefix="backlog-calib-") as handle: - result = subprocess.run( - ["bd", "export", "-o", handle.name], - capture_output=True, - text=True, - check=False, - ) - if result.returncode != 0: - raise SystemExit(f"bd export failed: {result.stderr.strip() or result.stdout.strip()}") - return _load_beads_jsonl(Path(handle.name)) - - -def _fmt_days(value: Any) -> str: - if not isinstance(value, (int, float)): - return "-" - return f"{value * 24:.1f}h" if value < 1 else f"{value:.1f}d" - - -def _render_summary_line(name: str, summary: dict[str, Any]) -> str: - if summary.get("n", 0) == 0: - return f" {name:<16} n=0" - return ( - f" {name:<16} n={summary['n']:<5} p50={_fmt_days(summary.get('p50_days')):<7} " - f"p90={_fmt_days(summary.get('p90_days')):<7} max={_fmt_days(summary.get('max_days'))}" - ) - - -def _render_human(report: dict[str, Any]) -> str: - lines: list[str] = [] - population = report["population"] - lines.append( - f"backlog-calibration as of {report['as_of']} -- " - f"{population['total']} beads ({population['by_status']}), " - f"corpus age {population['corpus_age_days']}d" - ) - lines.append(f"NOTE: {population['censoring_note']}") - lead = report["closed_lead_days"] - lines.append("\nclosed lead time (days):") - lines.append(_render_summary_line("overall", lead["overall"])) - for section in ("by_priority", "by_type", "by_epic_membership", "by_dependency_degree"): - lines.append(f" {section}:") - for name, summary in lead[section].items(): - lines.append(_render_summary_line(name, summary)) - survival = report["survival"] - lines.append( - f"\nsurvival (cohort created >={survival['cohort_min_age_days']:.0f}d ago, " - f"n={survival['overall']['n']}): fraction closed within window" - ) - for name, row in [("overall", survival["overall"]), *survival["by_priority"].items()]: - if row["n"] == 0: - continue - windows = " ".join(f"{window}d={row[f'closed_within_{window}d_pct']}%" for window in _SURVIVAL_WINDOWS_DAYS) - lines.append(f" {name:<8} n={row['n']:<5} {windows}") - reasons = report["close_reasons"] - lines.append( - f"\nclose reasons (n={reasons['closed_with_reason']} with reason; " - f"{reasons['no_implementation_pct']}% needed no implementation):" - ) - for name, summary in reasons["classes"].items(): - lines.append(_render_summary_line(name, summary)) - discovery = report["discovery"] - lines.append( - f"\ndiscovery vs drain (excluding import day {discovery.get('import_day_excluded')}): " - f"created={discovery.get('post_import_created')} " - f"closed={discovery.get('post_import_closed')} " - f"created-per-close={discovery.get('created_per_close')} " - f"median-net/day={discovery.get('median_net_per_day')} " - f"net-negative-days={discovery.get('net_negative_days')}" - f"/{discovery.get('post_import_day_count')}" - ) - if "pr_merge_latency" in report: - pr = report["pr_merge_latency"] - lines.append(f"\nPR open->merge latency (n={pr['n']}): {pr['note']}") - for label, summary in pr["by_changed_files_days"].items(): - lines.append(_render_summary_line(f"{label} files", summary)) - return "\n".join(lines) - - -def main(argv: Sequence[str] | None = None) -> int: - parser = argparse.ArgumentParser( - prog="devtools workspace backlog-calibration", - description="Measured duration/discovery models over the bead corpus.", - ) - parser.add_argument( - "--input", - "-i", - metavar="FILE", - help="Bead JSONL export (bd export -o FILE); default runs bd export itself", - ) - parser.add_argument( - "--prs", - metavar="FILE", - help=( - "Optional gh dump: gh pr list --state merged --limit 4000 " - "--json number,createdAt,mergedAt,additions,deletions,changedFiles" - ), - ) - parser.add_argument("--json", action="store_true", dest="json_out", help="JSON output") - args = parser.parse_args(argv) - - beads = _load_beads_jsonl(Path(args.input)) if args.input else _export_beads() - prs: list[dict[str, Any]] | None = None - if args.prs: - loaded = json.loads(Path(args.prs).read_text()) - if not isinstance(loaded, list): - raise SystemExit(f"{args.prs}: expected a JSON array of PRs") - prs = loaded - - report = build_report(beads, as_of=datetime.now(UTC), prs=prs) - if args.json_out: - print(json.dumps(report, indent=2)) - else: - print(_render_human(report)) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/devtools/bead_batch_show.py b/devtools/bead_batch_show.py deleted file mode 100644 index 8ecd2ebf8c..0000000000 --- a/devtools/bead_batch_show.py +++ /dev/null @@ -1,49 +0,0 @@ -"""Batch-show beads: id, status, prio, title, desc head, deps, notes tail. - -Usage: devtools workspace bead-batch-show [ ...] -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -from typing import Any - - -def _dep_str(dep: dict[str, Any]) -> str: - target = dep.get("depends_on_id") or dep.get("to_id") or dep.get("id") or "?" - return f"{target}({dep.get('type') or dep.get('dep_type') or '?'})" - - -def _show_one(bead_id: str) -> None: - result = subprocess.run(["bd", "show", bead_id, "--json"], capture_output=True, text=True) - try: - record = json.loads(result.stdout)[0] - except Exception: - print(f"== {bead_id} MISSING/ERROR: {result.stderr.strip()[:120]}") - return - print(f"== {record['id']} [{record['status']}] P{record['priority']} {record['issue_type']} | {record['title']}") - desc = (record.get("description") or "").replace("\n", " ")[:280] - print(f" DESC: {desc}") - deps = record.get("dependencies") or [] - if deps: - print(" DEPS:", ", ".join(_dep_str(d) if isinstance(d, dict) else str(d) for d in deps)) - notes = (record.get("notes") or "").replace("\n", " ") - if notes: - print(f" NOTES-tail: ...{notes[-240:]}") - print() - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("bead_ids", nargs="+", help="Bead ids to show (e.g. polylogue-kapb)") - args = parser.parse_args(argv) - for bead_id in args.bead_ids: - _show_one(bead_id) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/devtools/bead_cluster.py b/devtools/bead_cluster.py index 6103192e8f..4ce76a130c 100644 --- a/devtools/bead_cluster.py +++ b/devtools/bead_cluster.py @@ -52,7 +52,7 @@ ``frontier_clusters`` entries with ``len(beads) > 1`` rather than trusting a frozen number. The plain any-overlap graph (``_build_overlap_graph`` / ``_connected_components``) is kept as a general-purpose utility (also used -by ``devtools workspace lane-brief`` and other footprint tooling) but is no +by footprint and overlap tooling) but is no longer used to shape the FRONTIER-READY cluster output. Usage: diff --git a/devtools/beads_acceptance_applier.py b/devtools/beads_acceptance_applier.py deleted file mode 100644 index 24c1406ca1..0000000000 --- a/devtools/beads_acceptance_applier.py +++ /dev/null @@ -1,75 +0,0 @@ -"""Apply an exact, already-reconciled acceptance wave to a file copy. - -This is a local file actuator for testing and coordinator dry runs. It never -invokes ``bd`` and never writes a Beads database. -""" - -from __future__ import annotations - -import argparse -from pathlib import Path -from typing import Any - -from devtools import beads_acceptance_reconciliation as reconciliation -from polylogue.core.json import dumps as json_dumps - - -def apply_guarded_wave(*, repository: Path, before: Path, wave: Path, report: Path, output: Path) -> dict[str, Any]: - """Write the exact guarded result, or accept an identical prior result.""" - _, before_rows, wave_rows, report_value = reconciliation._validate_report_and_wave( - repository=repository, - before=before, - wave=wave, - report_path=report, - ) - expected_rows = dict(before_rows) - expected_rows.update(wave_rows) - expected_order = list(before_rows) - expected_digest = reconciliation.equality_digest(expected_rows) - if output.exists(): - existing = reconciliation.load_jsonl(output) - if list(existing) == expected_order and reconciliation.equality_digest(existing) == expected_digest: - return { - "ok": True, - "idempotent": True, - "output_population_digest": expected_digest, - "report_digest": reconciliation.report_digest(report_value), - "targeted_ids": report_value["targeted_ids"], - } - raise reconciliation.ReconciliationError("output already exists with a different guarded population") - reconciliation._write_jsonl(output, (expected_rows[bead_id] for bead_id in expected_order)) - return { - "ok": True, - "idempotent": False, - "output_population_digest": expected_digest, - "report_digest": reconciliation.report_digest(report_value), - "targeted_ids": report_value["targeted_ids"], - } - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--repository", type=Path, required=True) - parser.add_argument("--before", type=Path, required=True) - parser.add_argument("--wave", type=Path, required=True) - parser.add_argument("--report", type=Path, required=True) - parser.add_argument("--output", type=Path, required=True) - parser.add_argument("--json", action="store_true") - args = parser.parse_args(argv) - try: - result = apply_guarded_wave( - repository=args.repository, - before=args.before, - wave=args.wave, - report=args.report, - output=args.output, - ) - except reconciliation.ReconciliationError as exc: - print(str(exc)) - return 1 - print(json_dumps(result, indent=2, sort_keys=True)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/beads_acceptance_contracts.py b/devtools/beads_acceptance_contracts.py deleted file mode 100644 index fc8e72aa3a..0000000000 --- a/devtools/beads_acceptance_contracts.py +++ /dev/null @@ -1,655 +0,0 @@ -from __future__ import annotations - -import argparse -import hashlib -import re -from collections.abc import Iterable, Mapping -from pathlib import Path -from typing import Any - -from devtools.acceptance_route_registry import ( - AcceptanceRouteRegistryError, - load_registry, - registry_digest, - resolve_route, -) -from polylogue.core.json import JSONDecodeError -from polylogue.core.json import dumps as json_dumps -from polylogue.core.json import loads as json_loads - -_ALLOWED_TYPES = { - "implementation", - "live_operation", - "audit", - "decision", - "epic", - "test_harness", - "process", - "documentation", -} -_ALLOWED_RISKS = {"ordinary", "read-only", "durable-mutation", "semantic-integrity", "resource-concurrency"} -_ALLOWED_CONFIDENCE = {"high", "medium", "planner-review"} -_ALLOWED_CLOSURE_DISPOSITIONS = {"whole-or-explicit-partial"} -_ALLOWED_ROUTE_MODES = {"named"} -_ALLOWED_ROUTE_DISPATCH = {"production", "read-only", "decision", "documentation"} -_ROUTE_IDENTIFIER = re.compile(r"^[a-z][a-z0-9]*(?:[._:/-][a-z0-9]+)*$") -_ROUTE_DISPATCH_BY_TYPE = { - "implementation": frozenset({"production"}), - "live_operation": frozenset({"production"}), - "audit": frozenset({"read-only"}), - "decision": frozenset({"decision"}), - "epic": frozenset({"production"}), - "test_harness": frozenset({"production"}), - "process": frozenset({"production"}), - "documentation": frozenset({"documentation"}), -} -_ROUTE_CLASS_BY_TYPE = { - "implementation": "ImplementationRoute", - "live_operation": "LiveOperationRoute", - "audit": "AuditRoute", - "decision": "DecisionRoute", - "epic": "EpicRoute", - "test_harness": "TestHarnessRoute", - "process": "ProcessRoute", - "documentation": "DocumentationRoute", -} -_EVIDENCE_SPAN_FIELDS = frozenset({"source_field", "snapshot", "snapshot_digest", "range", "text_digest"}) -_EVIDENCE_RANGE_FIELDS = frozenset({"start", "end"}) -_ALLOWED_VERIFICATION_MANAGERS = {"devtools"} -_ALLOWED_VERIFICATION_FOCUSED = {"devtools test"} -_ALLOWED_VERIFICATION_DEFAULT = {"devtools verify"} -_ALLOWED_RECEIPT_KINDS = {"live-operation"} -_ALLOWED_RECEIPT_REQUIREMENTS = {"required"} -_REQUIRED_RECEIPT_BINDINGS = frozenset( - {"archive_identity", "operation", "target", "before_state", "after_state", "result_status"} -) -_PLACEHOLDER = re.compile( - r"(?:<[^>]+>|\.{3}|\b(?:TBD|TODO|FIXME|as appropriate|figure out|choose an approach|add suitable tests)\b)", - re.I, -) -_ROUTE_PLACEHOLDER = re.compile(r"\b(?:where applicable|as appropriate)\b", re.I) -_SOURCE_FIELDS = ("id", "title", "description", "design", "notes", "priority", "issue_type") -_SHA256 = re.compile(r"^[0-9a-f]{64}$") -_MANIFEST_ID = re.compile(r"^polylogue-[a-z0-9]+(?:\.[a-z0-9]+)*$") -_DEFAULT_MANIFEST = Path(__file__).parents[1] / "docs" / "plans" / "beads-acceptance-contracts-2026-08-07.txt" -_EXPECTED_MANIFEST_COUNT = 218 -_EXPECTED_MANIFEST_DIGEST = "703df11c81dae8af6d7106bc4737502ca8baddc9013916bbb68922696d8206b5" - - -class DependencyProjectionError(ValueError): - """Raised when a structured dependency projection contains an invalid scalar.""" - - -def _dependency_projection(issue: Mapping[str, Any]) -> list[dict[str, str | None]]: - """Return a stable, scope-bearing projection of Bead dependencies.""" - raw_dependencies = issue.get("dependencies") - if raw_dependencies is None: - return [] - if not isinstance(raw_dependencies, list): - return [{"invalid_type": type(raw_dependencies).__name__}] - dependencies: list[dict[str, str | None]] = [] - for index, dependency in enumerate(raw_dependencies): - if isinstance(dependency, dict): - depends_on_id = _dependency_scalar(dependency, index, ("depends_on_id", "to_id", "id"), "depends_on_id") - dependency_type = _dependency_scalar(dependency, index, ("type", "dep_type"), "type") - dependencies.append( - { - "depends_on_id": depends_on_id, - "type": dependency_type, - } - ) - elif isinstance(dependency, str): - dependencies.append({"depends_on_id": dependency, "type": None}) - else: - dependencies.append({"invalid_type": type(dependency).__name__}) - return sorted( - dependencies, - key=lambda dependency: ( - dependency.get("depends_on_id") or "", - dependency.get("type") or "", - dependency.get("invalid_type") or "", - ), - ) - - -def _dependency_scalar(dependency: Mapping[str, Any], index: int, keys: tuple[str, ...], label: str) -> str | None: - for key in keys: - value = dependency.get(key) - if value is None: - continue - if not isinstance(value, str): - raise DependencyProjectionError( - f"dependencies[{index}].{label} must be a string or null (got {type(value).__name__})" - ) - return value - return None - - -def dependency_digest(issue: Mapping[str, Any]) -> str: - """Return the digest for the canonical dependency projection.""" - return hashlib.sha256(json_dumps(_dependency_projection(issue), sort_keys=True).encode("utf-8")).hexdigest() - - -def source_digest(issue: dict[str, Any]) -> str: - """Return the digest bound to scope fields and the stable dependency projection.""" - payload = {key: issue.get(key) for key in _SOURCE_FIELDS} - payload["dependencies"] = _dependency_projection(issue) - return hashlib.sha256(json_dumps(payload, sort_keys=True).encode("utf-8")).hexdigest() - - -def _decode_document(value: object) -> dict[str, Any] | None: - if isinstance(value, dict): - return value - if not isinstance(value, str): - return None - try: - parsed = json_loads(value) - except JSONDecodeError: - return None - return parsed if isinstance(parsed, dict) else None - - -def _require_string(errors: list[str], value: object, key: str) -> None: - if not isinstance(value, str) or not value.strip(): - errors.append(f"{key} must be a non-empty string") - - -def _require_string_list(errors: list[str], contract: dict[str, Any], key: str, *, optional: bool = False) -> bool: - value = contract.get(key) - if value == [] and optional: - return True - if not isinstance(value, list) or not value: - if optional: - errors.append(f"{key} must be a list of strings; use [] when empty") - else: - errors.append(f"{key} must be a non-empty list of strings") - return False - if any(not isinstance(item, str) or not item.strip() for item in value): - errors.append(f"{key} must contain only non-empty strings") - return False - return True - - -def _require_mapping(errors: list[str], value: object, key: str) -> Mapping[str, Any] | None: - if not isinstance(value, Mapping): - errors.append(f"{key} must be an object") - return None - return value - - -def _validate_route_spec(errors: list[str], contract: dict[str, Any]) -> bool: - route_spec = _require_mapping(errors, contract.get("route_spec"), "route_spec") - if route_spec is None: - return False - valid = True - if set(route_spec) != {"mode", "identifier", "class", "dispatch"}: - errors.append("route_spec fields must be exactly mode, identifier, class, and dispatch") - valid = False - mode = route_spec.get("mode") - if not isinstance(mode, str) or mode not in _ALLOWED_ROUTE_MODES: - errors.append("route_spec.mode must be named") - valid = False - identifier = route_spec.get("identifier") - if not isinstance(identifier, str) or not identifier.strip(): - errors.append("route_spec.identifier must be a non-empty named identifier") - valid = False - elif not _ROUTE_IDENTIFIER.fullmatch(identifier): - errors.append("route_spec.identifier must be a structured named identifier") - valid = False - dispatch = route_spec.get("dispatch") - if not isinstance(dispatch, str) or dispatch not in _ALLOWED_ROUTE_DISPATCH: - errors.append("route_spec.dispatch is invalid") - valid = False - else: - contract_type = contract.get("contract_type") - allowed_dispatch = _ROUTE_DISPATCH_BY_TYPE.get( - contract_type if isinstance(contract_type, str) else "", frozenset() - ) - if dispatch not in allowed_dispatch: - errors.append( - f"route_spec.dispatch {dispatch!r} is incompatible with contract_type {contract.get('contract_type')!r}" - ) - valid = False - identifier = route_spec.get("identifier") - try: - registered = resolve_route(identifier) - except AcceptanceRouteRegistryError as exc: - errors.append(str(exc)) - return False - if registered is None: - errors.append(f"route_spec.identifier {identifier!r} is not registered") - return False - registered_bead_id = registered.get("bead_id") - if not isinstance(registered_bead_id, str) or not registered_bead_id: - errors.append("registered route authority must bind a non-empty Bead id") - valid = False - if not isinstance(contract.get("bead_id"), str) or registered_bead_id != contract.get("bead_id"): - errors.append("route_spec.identifier is registered for a different Bead") - valid = False - registered_contract_type = registered.get("contract_type") - if not isinstance(registered_contract_type, str) or registered_contract_type != contract.get("contract_type"): - errors.append("route_spec.identifier is registered for a different contract_type") - valid = False - registered_dispatch = registered.get("dispatch") - if not isinstance(registered_dispatch, str) or registered_dispatch != dispatch: - errors.append("route_spec.dispatch does not match the registered route class") - valid = False - contract_type = contract.get("contract_type") - expected_class = _ROUTE_CLASS_BY_TYPE.get(contract_type if isinstance(contract_type, str) else "") - route_class = route_spec.get("class") - if not isinstance(route_class, str) or route_class != expected_class: - errors.append("route_spec.class does not match the contract_type") - valid = False - registered_class = registered.get("class") - if not isinstance(registered_class, str) or registered_class != route_class: - errors.append("route_spec.class does not match the registered route class") - valid = False - targets = registered.get("targets") - if not isinstance(targets, list) or any(not isinstance(target, str) for target in targets): - errors.append("registered route authority has no target list") - valid = False - elif targets != contract.get("routes"): - errors.append("route_spec targets do not match the registered route authority") - valid = False - return valid - - -def _validate_verification_route(errors: list[str], contract: dict[str, Any]) -> bool: - route = _require_mapping(errors, contract.get("verification_route"), "verification_route") - if route is None: - return False - valid = True - manager = route.get("manager") - if not isinstance(manager, str) or manager not in _ALLOWED_VERIFICATION_MANAGERS: - errors.append("verification_route.manager must be devtools") - valid = False - focused = route.get("focused") - if not isinstance(focused, str) or focused not in _ALLOWED_VERIFICATION_FOCUSED: - errors.append("verification_route.focused must be devtools test") - valid = False - default = route.get("default") - if not isinstance(default, str) or default not in _ALLOWED_VERIFICATION_DEFAULT: - errors.append("verification_route.default must be devtools verify") - valid = False - return valid - - -def _validate_receipt(errors: list[str], contract: dict[str, Any]) -> bool: - receipt = _require_mapping(errors, contract.get("receipt"), "receipt") - if receipt is None: - return False - valid = True - kind = receipt.get("kind") - if not isinstance(kind, str) or kind not in _ALLOWED_RECEIPT_KINDS: - errors.append("receipt.kind must be live-operation") - valid = False - requirement = receipt.get("requirement") - if not isinstance(requirement, str) or requirement not in _ALLOWED_RECEIPT_REQUIREMENTS: - errors.append("receipt.requirement must be required") - valid = False - bindings = receipt.get("bindings") - if not isinstance(bindings, list) or any(not isinstance(item, str) for item in bindings): - errors.append("receipt.bindings must be a list of strings") - return False - if set(bindings) != _REQUIRED_RECEIPT_BINDINGS or len(bindings) != len(_REQUIRED_RECEIPT_BINDINGS): - errors.append("receipt.bindings must include each required live-operation dimension exactly once") - valid = False - return valid - - -def _validate_evidence_spans(errors: list[str], issue: dict[str, Any], contract: dict[str, Any]) -> None: - """Validate evidence as byte ranges over the Bead's source fields.""" - evidence = contract.get("evidence") - spans = contract.get("evidence_spans") - if not isinstance(evidence, list) or not isinstance(spans, list): - errors.append("evidence_spans must provide one typed span for every evidence item") - return - if len(spans) != len(evidence): - errors.append("evidence_spans must contain exactly one span for every evidence item") - return - for index, (value, span) in enumerate(zip(evidence, spans, strict=True)): - if not isinstance(span, Mapping): - errors.append(f"evidence_spans[{index}] must be an object") - continue - if set(span) != _EVIDENCE_SPAN_FIELDS: - errors.append( - f"evidence_spans[{index}] fields must be exactly source_field, snapshot, snapshot_digest, range, and " - "text_digest" - ) - source_field = span.get("source_field") - if not isinstance(source_field, str) or source_field not in {"title", "description", "design", "notes"}: - errors.append(f"evidence_spans[{index}].source_field must name title, description, design, or notes") - source_snapshot = issue.get(source_field) if isinstance(source_field, str) else None - if not isinstance(source_snapshot, str) or not source_snapshot: - errors.append(f"evidence_spans[{index}].source_field must reference a non-empty Bead source field") - snapshot = span.get("snapshot") - snapshot_bytes: bytes | None = None - if not isinstance(snapshot, str): - errors.append(f"evidence_spans[{index}].snapshot must be a UTF-8 snapshot string") - else: - snapshot_bytes = snapshot.encode("utf-8") - if isinstance(source_snapshot, str) and snapshot != source_snapshot: - errors.append(f"evidence_spans[{index}].snapshot does not match the Bead source field") - snapshot_digest = span.get("snapshot_digest") - if not isinstance(snapshot_digest, str) or not _SHA256.fullmatch(snapshot_digest): - errors.append(f"evidence_spans[{index}].snapshot_digest must be a lowercase SHA-256 digest") - elif snapshot_bytes is not None and snapshot_digest != hashlib.sha256(snapshot_bytes).hexdigest(): - errors.append(f"evidence_spans[{index}].snapshot_digest does not match the snapshot") - evidence_range = span.get("range") - if not isinstance(evidence_range, Mapping) or set(evidence_range) != _EVIDENCE_RANGE_FIELDS: - errors.append(f"evidence_spans[{index}].range must contain exactly start and end") - else: - start = evidence_range.get("start") - end = evidence_range.get("end") - if ( - not isinstance(start, int) - or isinstance(start, bool) - or not isinstance(end, int) - or isinstance(end, bool) - or start < 0 - or end <= start - ): - errors.append(f"evidence_spans[{index}].range must be a non-empty half-open integer range") - elif snapshot_bytes is not None and end > len(snapshot_bytes): - errors.append(f"evidence_spans[{index}].range exceeds the snapshot byte length") - text_digest = span.get("text_digest") - if not isinstance(text_digest, str) or not _SHA256.fullmatch(text_digest): - errors.append(f"evidence_spans[{index}].text_digest must be a lowercase SHA-256 digest") - elif snapshot_bytes is not None and isinstance(evidence_range, Mapping): - start = evidence_range.get("start") - end = evidence_range.get("end") - if ( - isinstance(start, int) - and not isinstance(start, bool) - and isinstance(end, int) - and not isinstance(end, bool) - and 0 <= start < end <= len(snapshot_bytes) - ): - span_bytes = snapshot_bytes[start:end] - try: - span_text = span_bytes.decode("utf-8") - except UnicodeDecodeError: - errors.append(f"evidence_spans[{index}].range must align to UTF-8 boundaries") - else: - expected_text_digest = hashlib.sha256(span_bytes).hexdigest() - if text_digest != expected_text_digest: - errors.append(f"evidence_spans[{index}].text_digest does not match the snapshot range") - if value != span_text: - errors.append(f"evidence_spans[{index}].range text does not match the evidence item") - - -def load_manifest(path: Path) -> tuple[str, ...]: - """Load and ratchet the committed required-contract inventory.""" - if not path.is_file(): - raise SystemExit(f"{path}: acceptance-contract manifest is missing") - required_values = path.read_text(encoding="utf-8").split() - if ( - len(required_values) != _EXPECTED_MANIFEST_COUNT - or len(set(required_values)) != _EXPECTED_MANIFEST_COUNT - or any(not _MANIFEST_ID.fullmatch(value) for value in required_values) - ): - raise SystemExit(f"{path}: acceptance-contract manifest inventory is invalid") - manifest_digest = hashlib.sha256(("\n".join(sorted(required_values)) + "\n").encode("utf-8")).hexdigest() - if manifest_digest != _EXPECTED_MANIFEST_DIGEST: - raise SystemExit(f"{path}: acceptance-contract manifest inventory changed") - return tuple(sorted(required_values)) - - -def _strings(value: Any) -> Iterable[str]: - if isinstance(value, str): - yield value - elif isinstance(value, list): - for item in value: - yield from _strings(item) - elif isinstance(value, dict): - for item in value.values(): - yield from _strings(item) - - -def render(contract: dict[str, Any]) -> str: - rows: list[str] = [] - n = 1 - - def add(label: str, value: str) -> None: - nonlocal n - rows.append(f"{n}. {label}: {value.strip()}") - n += 1 - - add("Outcome", contract["outcome"]) - if contract.get("confidence") == "planner-review": - add("Dispatch gate", "Planner review is required before implementation dispatch.") - route_spec = contract["route_spec"] - add( - "Route authority", - f"{route_spec['mode']} {route_spec['identifier']} {route_spec['dispatch']} route coverage is required.", - ) - for value in contract.get("retained_scope", []): - add("Existing scope retained", value) - for value in contract.get("routes", []): - add("Production route", value) - for value in contract.get("evidence", []): - add("Evidence", value) - for value in contract.get("verification", []): - add("Verification", value) - for value in contract.get("anti_vacuity", []): - add("Anti-vacuity", value) - for value in contract.get("safety", []): - add("Safety", value) - contract_type = contract.get("contract_type") - if isinstance(contract_type, str) and contract_type in {"implementation", "test_harness"}: - verification_route = contract["verification_route"] - add( - "Managed verification route", - f"focused={verification_route['focused']}; default={verification_route['default']}", - ) - if contract_type == "live_operation": - receipt = contract["receipt"] - add( - "Receipt requirement", - f"{receipt['kind']} result={receipt['requirement']} bindings={','.join(receipt['bindings'])}", - ) - add("Closure disposition", contract["closure"]["disposition"]) - add( - "Partial closure successor", - "required when the closure disposition is whole-or-explicit-partial.", - ) - add("Closure", contract["closure"]["rule"]) - return "\n".join(rows) - - -def validate(issue: dict[str, Any]) -> list[str]: - errors: list[str] = [] - metadata = _decode_document(issue.get("metadata")) - if metadata is None: - return ["metadata is not a JSON object"] - contract = metadata.get("acceptance_contract_v1") - if not isinstance(contract, dict): - return ["missing metadata.acceptance_contract_v1"] - if contract.get("schema_version") != 1: - errors.append("schema_version must be 1") - if contract.get("bead_id") != issue.get("id"): - errors.append("bead_id does not match issue id") - contract_type = contract.get("contract_type") - if not isinstance(contract_type, str) or contract_type not in _ALLOWED_TYPES: - errors.append("invalid contract_type") - risk = contract.get("risk") - if not isinstance(risk, str) or risk not in _ALLOWED_RISKS: - errors.append("invalid risk") - confidence = contract.get("confidence") - if not isinstance(confidence, str) or confidence not in _ALLOWED_CONFIDENCE: - errors.append("confidence must be high, medium, or planner-review") - _require_string(errors, contract.get("bead_id"), "bead_id") - _require_string(errors, contract.get("outcome"), "outcome") - for key in ("routes", "evidence", "verification", "anti_vacuity"): - _require_string_list(errors, contract, key) - _require_string_list(errors, contract, "retained_scope", optional=True) - _require_string_list(errors, contract, "safety", optional=True) - _validate_route_spec(errors, contract) - if isinstance(contract_type, str) and contract_type in {"implementation", "test_harness"}: - _validate_verification_route(errors, contract) - closure = contract.get("closure") - if not isinstance(closure, Mapping): - errors.append("closure must be an object") - else: - _require_string(errors, closure.get("rule"), "closure.rule") - disposition = closure.get("disposition") - if not isinstance(disposition, str) or disposition not in _ALLOWED_CLOSURE_DISPOSITIONS: - errors.append("closure.disposition must be whole-or-explicit-partial") - if not isinstance(closure.get("successor_required_for_partial"), bool): - errors.append("closure.successor_required_for_partial must be boolean") - elif disposition == "whole-or-explicit-partial" and not closure.get("successor_required_for_partial"): - errors.append("whole-or-explicit-partial requires successor_required_for_partial=true") - try: - computed_source_digest = source_digest(issue) - computed_dependency_digest = dependency_digest(issue) - except DependencyProjectionError as exc: - errors.append(str(exc)) - computed_source_digest = None - computed_dependency_digest = None - digest = contract.get("source_digest") - if not isinstance(digest, str) or not _SHA256.fullmatch(digest): - errors.append("source_digest must be a lowercase SHA-256 digest") - elif computed_source_digest is not None and digest != computed_source_digest: - errors.append("source_digest does not match the Bead source snapshot") - dependency_digest_value = contract.get("dependency_digest") - if not isinstance(dependency_digest_value, str) or not _SHA256.fullmatch(dependency_digest_value): - errors.append("dependency_digest must be a lowercase SHA-256 digest") - elif computed_dependency_digest is not None and dependency_digest_value != computed_dependency_digest: - errors.append("dependency_digest does not match the Bead dependency projection") - if issue.get("dependencies") is not None and not isinstance(issue.get("dependencies"), list): - errors.append("dependencies must be a list") - if contract_type == "live_operation" and not contract.get("safety"): - errors.append("live_operation requires safety clauses") - if contract_type == "live_operation": - _validate_receipt(errors, contract) - if risk == "durable-mutation" and not contract.get("safety"): - errors.append("durable-mutation requires safety clauses") - for value in contract.get("routes", []) if isinstance(contract.get("routes"), list) else []: - if isinstance(value, str) and _ROUTE_PLACEHOLDER.search(value): - errors.append("routes contains a generic placeholder; use named route fields") - for key, value in contract.items(): - if key in {"evidence", "evidence_spans"}: - continue - for text in _strings(value): - if _PLACEHOLDER.search(text): - errors.append(f"placeholder in contract: {text[:80]}") - _validate_evidence_spans(errors, issue, contract) - if not errors: - expected = render(contract) - if issue.get("acceptance_criteria") != expected: - errors.append("acceptance_criteria drifted from structured contract") - return sorted(set(errors)) - - -def load(path: Path) -> list[dict[str, Any]]: - rows: list[dict[str, Any]] = [] - for n, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): - if not line.strip(): - continue - try: - row = json_loads(line) - except JSONDecodeError as exc: - raise SystemExit(f"{path}:{n}: {exc}") from exc - if not isinstance(row, dict): - raise SystemExit(f"{path}:{n}: expected a JSON object, got {type(row).__name__}") - rows.append(row) - return rows - - -def validate_route_registry(required_ids: Iterable[str]) -> list[str]: - """Check that the committed route authority covers exactly the manifest IDs.""" - try: - registry = load_registry() - except AcceptanceRouteRegistryError as exc: - return [str(exc)] - required = set(required_ids) - errors: list[str] = [] - entries = list(registry.items()) - if len(entries) != len(required): - errors.append(f"route registry entry count mismatch: expected {len(required)}, found {len(entries)}") - bound_ids: list[str] = [] - for identifier, entry in entries: - bead_id = entry.get("bead_id") - if not isinstance(bead_id, str) or not bead_id: - errors.append(f"route registry entry {identifier!r} must bind one non-empty manifest Bead id") - else: - bound_ids.append(bead_id) - if bead_id not in required: - errors.append(f"route registry entry {identifier!r} binds unlisted Bead {bead_id!r}") - for field in ("class", "contract_type", "dispatch"): - value = entry.get(field) - if not isinstance(value, str) or not value or value == "*": - errors.append(f"route registry entry {identifier!r} has invalid {field} authority") - targets = entry.get("targets") - if ( - not isinstance(targets, list) - or not targets - or any(not isinstance(target, str) or not target for target in targets) - ): - errors.append(f"route registry entry {identifier!r} must have a non-empty string target list") - if len(bound_ids) != len(set(bound_ids)): - errors.append("route registry contains duplicate Bead bindings") - if set(bound_ids) != required: - errors.append(f"route registry Bead population mismatch: expected {len(required)}, found {len(set(bound_ids))}") - return errors - - -def route_registry_digest() -> str: - return registry_digest() - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser( - description="Validate structured Beads acceptance contracts without guessing from prose." - ) - parser.add_argument("issues", type=Path, nargs="?", default=Path(".beads/issues.jsonl")) - parser.add_argument( - "--manifest", - type=Path, - default=_DEFAULT_MANIFEST, - help="newline-separated Bead ids that must carry a valid contract", - ) - parser.add_argument("--json", action="store_true") - args = parser.parse_args(argv) - required_values = load_manifest(args.manifest) - registry_errors = validate_route_registry(required_values) - failures: dict[str, list[str]] = {} - required = set(required_values) - seen = set() - for issue in load(args.issues): - bid = issue.get("id") - if not isinstance(bid, str) or bid not in required: - continue - seen.add(bid) - errors = validate(issue) - if errors: - failures[bid] = errors - for missing in sorted(required - seen): - failures[missing] = ["manifest id missing from issues or contract"] - regeneration_required = [{"id": bead_id, "reasons": failures[bead_id]} for bead_id in sorted(failures)] - report = { - "ok": not failures and not registry_errors, - "dispatch_blocked": bool(failures or registry_errors), - "manifest": { - "expected_count": _EXPECTED_MANIFEST_COUNT, - "digest": _EXPECTED_MANIFEST_DIGEST, - }, - "validated": len(seen), - "route_registry_errors": registry_errors, - "failures": failures, - "regeneration_required": regeneration_required, - } - if args.json: - print(json_dumps(report, indent=2, sort_keys=True)) - else: - for error in registry_errors: - print(f"route_registry: {error}") - for bid, errors in sorted(failures.items()): - for error in errors: - print(f"{bid}: {error}") - print(f"validated={len(seen)} failures={len(failures)} regeneration_required={len(regeneration_required)}") - return 1 if failures or registry_errors else 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/beads_acceptance_reconciliation.py b/devtools/beads_acceptance_reconciliation.py deleted file mode 100644 index d18f42a790..0000000000 --- a/devtools/beads_acceptance_reconciliation.py +++ /dev/null @@ -1,533 +0,0 @@ -"""Reconcile guarded acceptance contracts against a read-only Beads export. - -This module deliberately operates on JSONL files. It never invokes ``bd`` and -never selects an authority for a changed record. The generated import wave is -made from the live record, with only the two contract fields replaced, so -``bd import --allow-stale`` cannot overwrite unrelated live work. -""" - -from __future__ import annotations - -import argparse -import copy -import datetime as dt -import hashlib -import re -import sys -from collections.abc import Iterable, Mapping -from decimal import Decimal -from pathlib import Path -from typing import Any - -from devtools import beads_acceptance_contracts as _contracts -from polylogue.core.json import JSONDecodeError -from polylogue.core.json import dumps as json_dumps -from polylogue.core.json import loads as json_loads - -source_digest = _contracts.source_digest -validate = _contracts.validate - - -def render(contract: dict[str, Any]) -> str: - """Expose the merged validator renderer for synthetic reconciliation fixtures.""" - return _contracts.render(contract) - - -_CONTRACT_KEY = "acceptance_contract_v1" -_CONTRACT_FIELDS = frozenset({"acceptance_criteria", "metadata"}) -_REPORT_CATEGORIES = ( - "master_only", - "live_only", - "master_newer", - "live_newer", - "same_timestamp_different", - "contract_refused", -) -_BEADS_TIMESTAMP = re.compile( - r"^(?P\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})" - r"(?:\.(?P\d+))?(?PZ|[+-]\d{2}:\d{2})$" -) -_REPORT_VERSION = 2 - - -class ReconciliationError(ValueError): - """Raised when a file-level reconciliation input is unsafe to consume.""" - - -def _metadata_object(value: object) -> dict[str, Any] | None: - if value is None: - return {} - if isinstance(value, dict): - return copy.deepcopy(value) - if not isinstance(value, str): - return None - try: - decoded = json_loads(value) - except JSONDecodeError: - return None - return copy.deepcopy(decoded) if isinstance(decoded, dict) else None - - -def _contract(issue: Mapping[str, Any]) -> dict[str, Any] | None: - metadata = _metadata_object(issue.get("metadata")) - if metadata is None: - return None - value = metadata.get(_CONTRACT_KEY) - return copy.deepcopy(value) if isinstance(value, dict) else None - - -def _source_digest_or_refuse(issue: Mapping[str, Any]) -> str: - try: - return source_digest(dict(issue)) - except _contracts.DependencyProjectionError as exc: - raise ReconciliationError(f"{issue.get('id')}: {exc}") from exc - - -def load_jsonl(path: Path) -> dict[str, dict[str, Any]]: - """Load a Beads JSONL export, rejecting malformed or duplicate IDs.""" - rows: dict[str, dict[str, Any]] = {} - for line_number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): - if not line.strip(): - continue - try: - value = json_loads(line) - except JSONDecodeError as exc: - raise ReconciliationError(f"{path}:{line_number}: invalid JSON: {exc}") from exc - if not isinstance(value, dict): - raise ReconciliationError(f"{path}:{line_number}: expected a JSON object") - bead_id = value.get("id") - if not isinstance(bead_id, str) or not bead_id: - raise ReconciliationError(f"{path}:{line_number}: missing string Bead id") - if bead_id in rows: - raise ReconciliationError(f"{path}:{line_number}: duplicate Bead id {bead_id}") - rows[bead_id] = value - return rows - - -def _canonical_rows(rows: Mapping[str, Mapping[str, Any]], ids: Iterable[str]) -> str: - return "\n".join(json_dumps(rows[bead_id], sort_keys=True) for bead_id in sorted(ids)) + "\n" - - -def equality_digest(rows: Mapping[str, Mapping[str, Any]], ids: Iterable[str] | None = None) -> str: - """Hash complete canonical rows, ordered by Bead ID.""" - selected = rows.keys() if ids is None else ids - return hashlib.sha256(_canonical_rows(rows, selected).encode("utf-8")).hexdigest() - - -def non_contract_equality_digest(rows: Mapping[str, Mapping[str, Any]], ids: Iterable[str]) -> str: - """Hash rows after removing only the two contract fields.""" - scrubbed: dict[str, dict[str, Any]] = {} - for bead_id in ids: - row = dict(copy.deepcopy(rows[bead_id])) - row.pop("acceptance_criteria", None) - raw_metadata = row.get("metadata") - metadata = _metadata_object(raw_metadata) - if metadata is not None: - metadata.pop(_CONTRACT_KEY, None) - if not metadata: - # The contract is the only allowed metadata mutation. Remove - # the now-empty container from both sides of the projection; - # all non-contract metadata keys remain visible and checked. - row.pop("metadata", None) - elif isinstance(raw_metadata, str): - row["metadata"] = json_dumps(metadata) - else: - row["metadata"] = metadata - scrubbed[bead_id] = row - return equality_digest(scrubbed) - - -def _parse_beads_timestamp(value: object) -> tuple[dt.datetime, Decimal]: - if not isinstance(value, str): - raise ValueError("updated_at must be a string on both repository and live records") - match = _BEADS_TIMESTAMP.fullmatch(value) - if match is None: - raise ValueError("updated_at must be a valid canonical Beads timestamp") - try: - parsed = dt.datetime.fromisoformat(f"{match['second']}{match['timezone'].replace('Z', '+00:00')}") - except ValueError as exc: - raise ValueError("updated_at must be a valid canonical Beads timestamp") from exc - if parsed.tzinfo is None: - raise ValueError("updated_at must include a timezone") - fraction = Decimal(f"0.{match['fraction'] or '0'}") - return parsed.astimezone(dt.UTC).replace(microsecond=0), fraction - - -def _classify_timestamp(master: Mapping[str, Any], live: Mapping[str, Any]) -> str: - master_timestamp = _parse_beads_timestamp(master.get("updated_at")) - live_timestamp = _parse_beads_timestamp(live.get("updated_at")) - if master_timestamp == live_timestamp: - return ( - "same_timestamp_same" - if source_digest(dict(master)) == source_digest(dict(live)) - else "same_timestamp_different" - ) - if master_timestamp > live_timestamp: - return "master_newer" - return "live_newer" - - -def _guarded_row( - *, - master: Mapping[str, Any], - live: Mapping[str, Any], - contract: Mapping[str, Any], -) -> dict[str, Any]: - metadata = _metadata_object(live.get("metadata")) - if metadata is None: - raise ReconciliationError(f"{live.get('id')}: live metadata is not a JSON object") - row = copy.deepcopy(dict(live)) - metadata[_CONTRACT_KEY] = copy.deepcopy(dict(contract)) - row["metadata"] = json_dumps(metadata) if isinstance(live.get("metadata"), str) else metadata - row["acceptance_criteria"] = master.get("acceptance_criteria") - changed = {key for key in set(row) | set(live) if key not in _CONTRACT_FIELDS and row.get(key) != live.get(key)} - if changed: - raise ReconciliationError(f"{live.get('id')}: guarded wave changed non-contract fields {sorted(changed)}") - return row - - -def _write_jsonl(path: Path, rows: Iterable[Mapping[str, Any]]) -> None: - path.write_text( - "".join(json_dumps(row, sort_keys=True) + "\n" for row in rows), - encoding="utf-8", - ) - - -def report_digest(report: Mapping[str, Any]) -> str: - """Hash the exact reconciliation report payload.""" - return hashlib.sha256(json_dumps(dict(report), sort_keys=True).encode("utf-8")).hexdigest() - - -def _load_report(path: Path) -> dict[str, Any]: - try: - value = json_loads(path.read_text(encoding="utf-8")) - except (OSError, JSONDecodeError) as exc: - raise ReconciliationError(f"{path}: invalid reconciliation report: {exc}") from exc - if not isinstance(value, dict): - raise ReconciliationError(f"{path}: reconciliation report must be an object") - return value - - -def _require_complete_report(actual: Mapping[str, Any], expected: Mapping[str, Any]) -> None: - """Require every report field to equal a fresh recomputation from bound inputs.""" - if set(actual) != set(expected): - raise ReconciliationError("reconciliation report fields do not match the canonical report shape") - for field in sorted(expected): - if actual[field] != expected[field]: - raise ReconciliationError(f"reconciliation report field {field!r} does not match canonical recomputation") - - -def _validate_report_and_wave( - *, - repository: Path, - before: Path, - wave: Path, - report_path: Path, -) -> tuple[dict[str, dict[str, Any]], dict[str, dict[str, Any]], dict[str, dict[str, Any]], dict[str, Any]]: - master = load_jsonl(repository) - before_rows = load_jsonl(before) - wave_rows = load_jsonl(wave) - report = _load_report(report_path) - required_ids = _contracts.load_manifest(_contracts._DEFAULT_MANIFEST) - registry_errors = _contracts.validate_route_registry(required_ids) - if registry_errors: - raise ReconciliationError("canonical route registry validation failed: " + "; ".join(registry_errors)) - if report.get("report_version") != _REPORT_VERSION: - raise ReconciliationError("reconciliation report version is stale or unsupported") - if report.get("manifest_digest") != _contracts._EXPECTED_MANIFEST_DIGEST: - raise ReconciliationError("reconciliation report manifest digest does not match this exact head") - if report.get("route_registry_digest") != _contracts.route_registry_digest(): - raise ReconciliationError("reconciliation report route registry digest is stale") - if report.get("contract_denominator") != len(required_ids): - raise ReconciliationError("reconciliation report contract denominator is stale") - if report.get("repository_population_digest") != equality_digest(master): - raise ReconciliationError("canonical repository population digest does not match the reconciliation report") - if report.get("live_population_digest") != equality_digest(before_rows): - raise ReconciliationError("before population digest does not match the reconciliation report") - target_ids = report.get("targeted_ids") - if ( - not isinstance(target_ids, list) - or target_ids != sorted(set(target_ids)) - or any(not isinstance(i, str) for i in target_ids) - ): - raise ReconciliationError("reconciliation report targeted_ids must be sorted and unique") - wave_ids = list(wave_rows) - if wave_ids != target_ids: - raise ReconciliationError("targeted wave order or population does not match the reconciliation report") - if set(target_ids) - set(before_rows) or set(target_ids) - set(master): - raise ReconciliationError("targeted wave contains records absent from the bound populations") - expected_wave_digest = equality_digest(wave_rows, target_ids) - if report.get("targeted_wave_equality_digest") != expected_wave_digest: - raise ReconciliationError("targeted wave digest does not match the reconciliation report") - row_digests = report.get("targeted_wave_row_digests") - if not isinstance(row_digests, dict) or set(row_digests) != set(target_ids): - raise ReconciliationError("reconciliation report does not carry every targeted wave row digest") - for bead_id in target_ids: - if row_digests.get(bead_id) != equality_digest({bead_id: wave_rows[bead_id]}, [bead_id]): - raise ReconciliationError(f"targeted wave row digest mismatch for {bead_id}") - canonical = master[bead_id] - contract = _contract(canonical) - if contract is None or validate(canonical): - raise ReconciliationError(f"canonical contract revalidation failed for {bead_id}") - if _source_digest_or_refuse(before_rows[bead_id]) != contract.get("source_digest"): - raise ReconciliationError(f"stale source digest refuses targeted row {bead_id}") - expected = _guarded_row(master=canonical, live=before_rows[bead_id], contract=contract) - if wave_rows[bead_id] != expected: - raise ReconciliationError(f"targeted wave row is not the canonical guarded row for {bead_id}") - expected_report, expected_wave = reconcile(repository, before) - expected_wave_rows = {row["id"]: row for row in expected_wave} - if wave_rows != expected_wave_rows: - raise ReconciliationError("targeted wave does not match the canonical reconciliation recomputation") - _require_complete_report(report, expected_report) - return master, before_rows, wave_rows, report - - -def reconcile( - repository: Path, - live_export: Path, - *, - manifest: Path | None = None, -) -> tuple[dict[str, Any], list[dict[str, Any]]]: - """Return a report and a minimal guarded import wave. - - A contract is eligible only when the live source digest equals the digest - carried by the canonical contract. Timestamp ordering is reported but is - never used as an authority choice. A malformed live metadata document is - an explicit contract refusal. - """ - try: - required_ids = _contracts.load_manifest(manifest or _contracts._DEFAULT_MANIFEST) - except SystemExit as exc: - raise ReconciliationError(str(exc)) from exc - master = load_jsonl(repository) - live = load_jsonl(live_export) - missing_manifest_ids = sorted(set(required_ids) - set(master)) - if missing_manifest_ids: - raise ReconciliationError( - "canonical acceptance-contract manifest is incomplete: " - f"missing {len(missing_manifest_ids)} IDs {missing_manifest_ids}" - ) - master_contracts: dict[str, dict[str, Any]] = {} - invalid_contracts: dict[str, list[str]] = {} - for bead_id in required_ids: - contract = _contract(master[bead_id]) - if contract is None: - invalid_contracts[bead_id] = ["missing metadata.acceptance_contract_v1"] - continue - master_contracts[bead_id] = contract - errors = validate(master[bead_id]) - if errors: - invalid_contracts[bead_id] = errors - if invalid_contracts: - details = "; ".join(f"{bead_id}: {', '.join(errors)}" for bead_id, errors in sorted(invalid_contracts.items())) - raise ReconciliationError(f"canonical contract validation failed: {details}") - - master_ids = set(master) - live_ids = set(live) - report: dict[str, Any] = { - "report_version": _REPORT_VERSION, - "repository": str(repository), - "live_export": str(live_export), - "counts": dict.fromkeys(_REPORT_CATEGORIES, 0), - "ids": {category: [] for category in _REPORT_CATEGORIES}, - "contract_denominator": len(required_ids), - "contract_present_denominator": len(set(required_ids) & live_ids), - "contract_guarded_count": 0, - "contract_refused_denominator": 0, - "contract_refused_reasons": {}, - "contract_deferred_denominator": 0, - "contract_deferred_reasons": {}, - "already_guarded_ids": [], - "targeted_ids": [], - } - for report_category, ids in ( - ("master_only", master_ids - live_ids), - ("live_only", live_ids - master_ids), - ): - report["ids"][report_category] = sorted(ids) - report["counts"][report_category] = len(ids) - - wave: list[dict[str, Any]] = [] - refused_reasons: dict[str, list[str]] = {} - for bead_id in sorted(master_ids & live_ids): - master_row = master[bead_id] - live_row = live[bead_id] - timestamp_error: str | None = None - try: - timestamp_category = _classify_timestamp(master_row, live_row) - except ValueError as exc: - timestamp_category = None - timestamp_error = str(exc) - if timestamp_category in {"master_newer", "live_newer", "same_timestamp_different"}: - report["ids"][timestamp_category].append(bead_id) - report["counts"][timestamp_category] += 1 - - contract = master_contracts.get(bead_id) - if contract is None: - continue - report["contract_refused_denominator"] += 1 - reasons: list[str] = [] - expected_digest = contract["source_digest"] - try: - actual_digest = source_digest(live_row) - except _contracts.DependencyProjectionError as exc: - reasons.append(str(exc)) - else: - if actual_digest != expected_digest: - reasons.append(f"source digest mismatch: expected contract {expected_digest}, live {actual_digest}") - if _metadata_object(live_row.get("metadata")) is None: - reasons.append("live metadata is not a JSON object") - if timestamp_error is not None and timestamp_error not in reasons: - reasons.append(timestamp_error) - if reasons: - report["ids"]["contract_refused"].append(bead_id) - report["counts"]["contract_refused"] += 1 - refused_reasons[bead_id] = reasons - continue - if timestamp_category == "live_newer": - report["contract_deferred_denominator"] += 1 - report["contract_deferred_reasons"][bead_id] = ( - "live-newer record is excluded from the targeted wave; coordinator adjudication is required" - ) - continue - candidate = _guarded_row(master=master_row, live=live_row, contract=contract) - current_contract = _contract(live_row) - if ( - live_row.get("acceptance_criteria") == master_row.get("acceptance_criteria") - and current_contract == contract - ): - report["already_guarded_ids"].append(bead_id) - continue - wave.append(candidate) - report["targeted_ids"].append(bead_id) - - report["ids"]["contract_refused"] = sorted(report["ids"]["contract_refused"]) - report["contract_refused_reasons"] = {bead_id: refused_reasons[bead_id] for bead_id in sorted(refused_reasons)} - report["contract_deferred_reasons"] = { - bead_id: report["contract_deferred_reasons"][bead_id] for bead_id in sorted(report["contract_deferred_reasons"]) - } - report["contract_guarded_count"] = len(report["targeted_ids"]) + len(report["already_guarded_ids"]) - report["manifest_digest"] = _contracts._EXPECTED_MANIFEST_DIGEST - report["route_registry_digest"] = _contracts.route_registry_digest() - report["repository_population_digest"] = equality_digest(master) - report["live_equality_digest"] = equality_digest(live) - report["live_population_digest"] = report["live_equality_digest"] - report["targeted_non_contract_equality_digest"] = non_contract_equality_digest(live, report["targeted_ids"]) - report["targeted_wave_equality_digest"] = equality_digest({row["id"]: row for row in wave}, report["targeted_ids"]) - report["targeted_wave_non_contract_equality_digest"] = non_contract_equality_digest( - {row["id"]: row for row in wave}, report["targeted_ids"] - ) - report["targeted_wave_row_digests"] = {row["id"]: equality_digest({row["id"]: row}, [row["id"]]) for row in wave} - return report, wave - - -def verify_post_import(*, repository: Path, before: Path, after: Path, wave: Path, report: Path) -> dict[str, Any]: - """Verify that a targeted import changed only the guarded contract fields.""" - _, before_rows, wave_rows, reconciliation = _validate_report_and_wave( - repository=repository, - before=before, - wave=wave, - report_path=report, - ) - after_rows = load_jsonl(after) - target_ids = set(wave_rows) - if not target_ids <= set(before_rows): - raise ReconciliationError("targeted wave contains records absent from the before export") - if not target_ids <= set(after_rows): - raise ReconciliationError("targeted wave contains records absent from the after export") - unchanged_ids = set(before_rows) - target_ids - added_ids = sorted(set(after_rows) - set(before_rows)) - removed_ids = sorted(set(before_rows) - set(after_rows)) - if added_ids or removed_ids: - raise ReconciliationError( - f"post-import export changed the record universe: added={added_ids}, removed={removed_ids}" - ) - changed_outside_wave = sorted( - bead_id for bead_id in unchanged_ids if before_rows[bead_id] != after_rows.get(bead_id) - ) - if changed_outside_wave: - raise ReconciliationError(f"post-import export changed records outside targeted wave: {changed_outside_wave}") - before_non_contract = non_contract_equality_digest(before_rows, target_ids) - after_non_contract = non_contract_equality_digest(after_rows, target_ids) - expected_wave = equality_digest(wave_rows, reconciliation["targeted_ids"]) - actual_target = equality_digest(after_rows, target_ids) - if before_non_contract != after_non_contract: - raise ReconciliationError("post-import targeted records changed non-contract fields") - if expected_wave != actual_target: - raise ReconciliationError("post-import targeted records differ from the guarded wave") - before_population_digest = equality_digest(before_rows) - after_population_digest = equality_digest(after_rows) - if reconciliation.get("live_population_digest") != before_population_digest: - raise ReconciliationError("full before population digest changed after reconciliation") - return { - "ok": True, - "report_digest": report_digest(reconciliation), - "reconciliation_wave_digest": reconciliation["targeted_wave_equality_digest"], - "before_population_digest": before_population_digest, - "after_population_digest": after_population_digest, - "targeted_ids": sorted(target_ids), - "unchanged_outside_wave": len(unchanged_ids), - "targeted_non_contract_equality_digest": after_non_contract, - "targeted_wave_equality_digest": actual_target, - } - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser( - description="Reconcile repository acceptance contracts with a read-only live Beads export." - ) - parser.add_argument("--repository", type=Path, help="canonical repository JSONL") - parser.add_argument("--live", type=Path, help="read-only live Beads export") - parser.add_argument( - "--manifest", type=Path, default=_contracts._DEFAULT_MANIFEST, help="ratcheted contract ID manifest" - ) - parser.add_argument("--wave", type=Path, help="write the guarded targeted import JSONL") - parser.add_argument("--report", type=Path, help="write the reconciliation report JSON") - parser.add_argument("--verify-before", type=Path, help="before export for post-import verification") - parser.add_argument("--verify-after", type=Path, help="after export for post-import verification") - parser.add_argument("--verify-wave", type=Path, help="guarded wave for post-import verification") - parser.add_argument("--verify-repository", type=Path, help="canonical repository bound to the report") - parser.add_argument("--verify-report", type=Path, help="exact reconciliation report bound to the wave") - parser.add_argument("--json", action="store_true", help="emit the report or verification result as JSON") - args = parser.parse_args(argv) - try: - if any( - ( - args.verify_before, - args.verify_after, - args.verify_wave, - args.verify_repository, - args.verify_report, - ) - ): - if not all( - (args.verify_before, args.verify_after, args.verify_wave, args.verify_repository, args.verify_report) - ): - raise ReconciliationError( - "post-import verification requires --verify-repository, --verify-report, --verify-before, " - "--verify-after, and --verify-wave" - ) - result = verify_post_import( - repository=args.verify_repository, - before=args.verify_before, - after=args.verify_after, - wave=args.verify_wave, - report=args.verify_report, - ) - else: - if not args.repository or not args.live: - raise ReconciliationError("reconciliation requires --repository and --live") - result, wave = reconcile(args.repository, args.live, manifest=args.manifest) - if args.wave: - _write_jsonl(args.wave, wave) - if args.report: - args.report.write_text(json_dumps(result, indent=2, sort_keys=True) + "\n", encoding="utf-8") - except ReconciliationError as exc: - print(str(exc), file=sys.stderr) - return 1 - print(json_dumps(result, indent=2, sort_keys=True)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/beads_state_report.py b/devtools/beads_state_report.py deleted file mode 100644 index f2d5704268..0000000000 --- a/devtools/beads_state_report.py +++ /dev/null @@ -1,2542 +0,0 @@ -"""beads-state-report: a self-contained HTML state-of-the-backlog report. - -Reads ``.beads/issues.jsonl`` and emits one standalone HTML file covering the -whole bead population -- open AND closed -- across eight views: pulse (now vs -7/14 days ago, reconstructed from timestamps), shape (status x priority x -type), structure (epic/program trees with per-epic trend sparklines), topology -(the ``blocks`` dependency graph: ready / blocked / top blockers / cycles / -densest cluster / the parallel frontier), time (creation, closure, burnup, age -x priority heatmap), health (a review queue of graph defects), themes (where -open work concentrates by subsystem), and the dated review passes (the -hand-verified VERIFICATION(...) verdict subset and the 2026-07-31 P0 -reconciliation markers, rendered only when their markers exist in the data). - -Design rule: **interpretation is computed, never fossilized.** Every claim in -the findings list is emitted by a conditional generator in ``compute_insights`` -that checks the condition against the live data and renders the variant that -is actually true (or nothing). There is no hand-authored paragraph that can -silently go false on regeneration. The only authored content is timeless -framing (what a section shows, how to read the notation) and two named -constants from dated review passes, kept as constants precisely so their -dated provenance is explicit. - -Temporal reconstruction: every bead and every dependency edge in the export -carries ``created_at`` (and beads a ``closed_at``), so the open/ready/blocked -state of the graph at any past instant is recomputable. Caveats are inherent -and stated in the report: deleted beads/edges are invisible, and a reopened -bead's earlier closure is lost (``closed_at`` reflects only the latest close). -That makes trend figures *derived*, not measured -- they are tagged as such. - -Usage: devtools workspace beads-state-report [--out PATH] [--fresh] [--json] -``--fresh`` runs ``bd export -o .beads/issues.jsonl`` first, because ``bd`` -mutations do not immediately re-export and a stale file yields stale counts. -""" - -from __future__ import annotations - -import argparse -import datetime as dt -import html -import json -import re -import subprocess -import sys -from collections import Counter, defaultdict -from collections.abc import Iterable, Sequence -from dataclasses import dataclass, field -from pathlib import Path -from typing import Any - -from devtools import repo_root as _get_root - -IssueDict = dict[str, Any] -# (src issue, depends-on issue, relation type, edge created_at ISO string) -Edge = tuple[str, str, str, str] - -# -------------------------------------------------------------------------- -# Constants from dated review passes. These are the ONLY numbers in this -# module that were not computed from the input file; each is an -# operator-supplied figure from a specific, dated pass and is rendered with -# that provenance attached. -# -------------------------------------------------------------------------- -# 2026-07-31 landing-check lane: the operator's own count of the hand-verified -# subset (strict regex extraction below finds fewer; the gap is reported). -OPERATOR_VERIFIED_COUNT = 190 -# 2026-07-31 P0 reconciliation pass: the pass's own summary table covered this -# many beads (marker extraction below finds fewer; the gap is reported). -RECONCILIATION_EXPECTED = 24 - -VERDICT_STRICT = re.compile(r"VERIFICATION\s*\(([^)]*)\)\s*:\s*(STALE|PARTIAL|LIVE)", re.IGNORECASE) -VERDICT_LOOSE = re.compile(r"\b(STALE|PARTIAL|LIVE)\b") - -# The 2026-07-31 P0 reconciliation pass wrote one -# `RECONCILIATION 2026-07-31[...]: ` marker per bead it touched. -# Scoping the token search to a window *after* that literal anchor matters: -# bare tokens like MISFRAMED are ordinary English words that also appear, -# unrelated, in older notes on other beads. -RECONCILIATION_ANCHOR = re.compile(r"RECONCILIATION 2026-07-31") -RECONCILIATION_TOKENS: tuple[str, ...] = ( - "FIXED-AND-EFFECTIVE", - "FIXED-PENDING-REBUILD", - "FIXED-PENDING-DEPLOY", - "MISFRAMED", - "GENUINELY OPEN", -) -RECONCILIATION_TOKEN_RE = re.compile("|".join(re.escape(t) for t in RECONCILIATION_TOKENS)) -RECONCILIATION_WINDOW = 1200 - -# Thresholds for computed insights. Named so the conditions are auditable. -STALE_CLAIM_DAYS = 7 # in_progress untouched this long = zombie claim -AGED_URGENT_DAYS = 14 # open P0/P1 older than this = urgency not being consumed -VELOCITY_WINDOW_DAYS = 14 # trailing window vs the window before it - -SUBSYSTEMS: dict[str, tuple[str, ...]] = { - # canonical subsystem -> area:* label suffixes folded into it - "storage": ("storage", "substrate", "blob", "blobs", "schema", "schemas", "archive", "durability"), - "sources": ( - "sources", - "source", - "ingest", - "parsers", - "parsing", - "capture", - "lineage", - "attachments", - "protocol", - "identity", - ), - "daemon": ("daemon", "ops", "pipeline", "events", "perf", "performance", "reliability"), - "insights": ("insights", "analytics", "cost", "usage", "temporal", "evidence", "analysis", "embeddings"), - "cli": ("cli", "query", "query-dsl", "search", "surface", "api", "config"), - "mcp": ("mcp", "context", "coordination", "orchestration", "agents", "delegation", "delegations"), - "web/extension": ("web", "browser", "browser-capture", "extension", "rendering", "ux"), - "tests": ("test", "testing", "tests", "test-harness", "verification", "coverage", "quality", "eval"), - "devtools": ( - "devtools", - "devloop", - "ci", - "release", - "git", - "beads", - "beads-hygiene", - "planning", - "maintenance", - "cleanup", - "status", - ), - "docs/legibility": ("docs", "legibility", "demos", "demo", "architecture", "adoption", "audit"), -} - -KEYWORDS: dict[str, tuple[str, ...]] = { - "storage": ("storage", "sqlite", "index.db", "source.db", "user.db", "schema", "blob", "migration", "table"), - "sources": ("parser", "parse", "ingest", "provider", "origin", "claude code", "codex", "chatgpt", "session"), - "daemon": ("daemon", "polylogued", "converg", "watcher", "cursor", "acquire"), - "insights": ("insight", "cost", "usage", "token", "profile", "timeline", "analytic"), - "cli": ("cli", "command", "query", "click", "filter", "find ", "read --"), - "mcp": ("mcp", "agent", "context", "assertion", "recall", "coordination"), - "web/extension": ("web", "browser", "extension", "workbench", "http", "ui"), - "tests": ("test", "pytest", "fixture", "coverage", "hypothesis", "regression"), - "devtools": ("devtools", "render", "lint", "bead", "ci ", "workflow", "hook"), - "docs/legibility": ("doc", "readme", "demo", "narrative", "legib", "explain"), -} - -TYPE_GLYPH: dict[str, tuple[str, str]] = { - "task": ("▪", "task"), - "bug": ("✕", "bug"), - "feature": ("✦", "feature"), - "epic": ("▣", "epic"), - "chore": ("⚙", "chore"), - "decision": ("⚖", "decision"), - "spike": ("⚡", "spike"), -} -STATUS_GLYPH: dict[str, tuple[str, str]] = { - "open": ("○", "open"), - "in_progress": ("◐", "in progress"), - "deferred": ("◌", "deferred"), - "closed": ("●", "closed"), -} -STATUS_ORDER: tuple[str, ...] = ("open", "in_progress", "deferred", "closed") - -# Mechanical graph-health checks: (badge class, label, Facts attribute, meaning). -HEALTH_CHECKS: tuple[tuple[str, str, str, str], ...] = ( - ("bad", "dangling dependency references", "dangling", "edge points at an id that does not exist"), - ("warn", "dotted id disagrees with parent edge", "id_mismatch", "id prefix names a different parent"), - ("warn", "dotted id with no parent edge", "id_orphan", "looks nested, is structurally a root"), - ( - "warn", - "open parent, all children closed", - "open_parent_all_closed", - "review queue — adjudicate against the parent's own AC", - ), - ( - "warn", - "closed parent with open children", - "closed_parent_open_kids", - "parent retired while decomposed work continues", - ), - ( - "warn", - f"in-progress untouched >{STALE_CLAIM_DAYS} days", - "stale_claims", - "a claim nobody is working — release or finish it", - ), - ("warn", "duplicate titles", "dup_titles", "same work filed twice, independently"), - ("info", "closed without a close reason", "closed_no_reason", "no durable record of why it ended"), - ("info", "open without acceptance criteria", "no_ac_open", "no definition of done"), -) - - -# -------------------------------------------------------------------------- -# loading + primitives -# -------------------------------------------------------------------------- -def load(path: Path) -> tuple[dict[str, IssueDict], list[Edge]]: - """Parse the bd JSONL export into an id->issue map and a dependency edge list.""" - issues: dict[str, IssueDict] = {} - edges: list[Edge] = [] - for line in path.read_text(encoding="utf-8").splitlines(): - line = line.strip() - if not line: - continue - rec = json.loads(line) - if rec.get("_type") == "issue": - issues[rec["id"]] = rec - for dep in rec.get("dependencies") or []: - edges.append( - ( - str(dep["issue_id"]), - str(dep["depends_on_id"]), - str(dep.get("type", "blocks")), - str(dep.get("created_at") or ""), - ) - ) - elif rec.get("_type") == "dependency": - edges.append( - ( - str(rec["issue_id"]), - str(rec["depends_on_id"]), - str(rec.get("type", "blocks")), - str(rec.get("created_at") or ""), - ) - ) - return issues, edges - - -def _blob(rec: IssueDict) -> str: - parts = [str(rec.get("notes") or "")] - parts.extend(str(c.get("text") or "") for c in (rec.get("comments") or [])) - return "\n".join(parts) - - -def _parse_ts(value: str) -> dt.datetime: - parsed = dt.datetime.fromisoformat(value.replace("Z", "+00:00")) - if parsed.tzinfo is None: - parsed = parsed.replace(tzinfo=dt.UTC) - return parsed - - -def _median(values: Sequence[float]) -> float: - if not values: - return 0.0 - ordered = sorted(values) - mid = len(ordered) // 2 - if len(ordered) % 2: - return ordered[mid] - return (ordered[mid - 1] + ordered[mid]) / 2 - - -def _git_show(root: Path, ref: str, rel_path: str) -> str | None: - """Return a file's content at a git ref, or ``None`` if unavailable.""" - try: - proc = subprocess.run( - ["git", "show", f"{ref}:{rel_path}"], - cwd=root, - capture_output=True, - text=True, - timeout=15, - ) - except (OSError, subprocess.SubprocessError): - return None - return proc.stdout if proc.returncode == 0 else None - - -_LIFECYCLE_DECL_RE = re.compile(r"version=(\d+),.*?classes=\(([^)]*)\)", re.DOTALL) - - -def schema_gap_facts(root: Path) -> dict[str, Any]: - """The derived-index schema gap: origin/master's declared - ``INDEX_SCHEMA_VERSION`` vs the live archive's actual ``PRAGMA - user_version``, and which intervening versions are ``SEMANTIC_REPARSE`` - (block a SQL fast-forward -- a merged fix at one of those versions is - real and inert until ``polylogue ops reset --index && polylogued run``). - - Two independent read paths, both best-effort and non-fatal on failure so - a sandbox without network/git-remote access or a live archive still - renders a report (just without this fact): - - - ``git show origin/master:...`` for the declared version and delta - classes, so this is correct even on a branch behind master. - - the operator's configured archive root - (``~/.config/polylogue/polylogue.toml``), read-only ``PRAGMA``, for the - live version. Deliberately does not use ``POLYLOGUE_ARCHIVE_ROOT`` -- - that env var is routinely overridden for sandboxed/test runs, and this - fact is specifically about the real production archive. - """ - result: dict[str, Any] = { - "available": False, - "declared": None, - "live_version": None, - "live_path": None, - "blockers": [], - "blocker_notes": {}, - } - - index_src = _git_show(root, "origin/master", "polylogue/storage/sqlite/archive_tiers/index.py") - if index_src: - m = re.search(r"^INDEX_SCHEMA_VERSION\s*=\s*(\d+)", index_src, re.MULTILINE) - if m: - result["declared"] = int(m.group(1)) - - lifecycle_src = _git_show(root, "origin/master", "polylogue/storage/sqlite/lifecycle.py") or "" - versions: dict[int, tuple[str, ...]] = {} - for vm in _LIFECYCLE_DECL_RE.finditer(lifecycle_src): - v = int(vm.group(1)) - classes = tuple(c.strip().removeprefix("DerivedDeltaClass.") for c in vm.group(2).split(",") if c.strip()) - versions[v] = classes - - live_path: Path | None = None - try: - import tomllib - - cfg_path = Path.home() / ".config/polylogue/polylogue.toml" - if cfg_path.exists(): - cfg = tomllib.loads(cfg_path.read_text(encoding="utf-8")) - root_str = cfg.get("archive", {}).get("root") - if root_str: - live_path = Path(root_str) / "index.db" - except Exception: - live_path = None - - if live_path and live_path.exists(): - result["live_path"] = str(live_path) - try: - proc = subprocess.run( - ["sqlite3", "-readonly", str(live_path), "PRAGMA user_version;"], - capture_output=True, - text=True, - timeout=15, - check=True, - ) - result["live_version"] = int(proc.stdout.strip()) - except Exception: - pass - - if result["declared"] is not None and result["live_version"] is not None: - result["available"] = True - lo, hi = result["live_version"] + 1, result["declared"] - result["blockers"] = [v for v in range(lo, hi + 1) if "SEMANTIC_REPARSE" in versions.get(v, ())] - result["blocker_notes"] = {v: versions.get(v, ()) for v in range(lo, hi + 1)} - return result - - -# -------------------------------------------------------------------------- -# facts: every measured/derived quantity the report renders -# -------------------------------------------------------------------------- -class Facts: - """Every measured quantity the report renders. Nothing here is authored.""" - - def __init__(self, issues: dict[str, IssueDict], edges: list[Edge], now: dt.datetime) -> None: - self.issues = issues - self.edges = edges - self.now = now - self._cluster_adj: dict[str, set[str]] = defaultdict(set) - rows = list(issues.values()) - self.rows = rows - self.total = len(rows) - - self.status = Counter(str(r["status"]) for r in rows) - self.priority = Counter(int(r["priority"]) for r in rows) - self.itype = Counter(str(r["issue_type"]) for r in rows) - self.matrix: dict[tuple[str, int], int] = Counter((str(r["status"]), int(r["priority"])) for r in rows) - self.open_ids = [str(r["id"]) for r in rows if r["status"] != "closed"] - self.open_total = len(self.open_ids) - self.statuses_present = [s for s in STATUS_ORDER if self.status.get(s)] + sorted( - set(self.status) - set(STATUS_ORDER) - ) - - # ---- dependency graph ------------------------------------------- - self.dep_types = Counter(t for _, _, t, _ in edges) - - # relates-to/related vocabulary state. A repair that re-types every - # existing `related` edge does not constrain the field, so a fresh - # `related` edge can land minutes later -- that is a distinct state - # from "never repaired" and from "clean". - _relates, _related = self.dep_types["relates-to"], self.dep_types["related"] - _assoc_total = _relates + _related - if _related == 0: - self.vocab_state = "clean" - elif _assoc_total and _related <= max(3, 0.02 * _assoc_total): - self.vocab_state = "recurring" - else: - self.vocab_state = "split" - - self.blockers: dict[str, set[str]] = defaultdict(set) - self.blocking: dict[str, set[str]] = defaultdict(set) - self.children: dict[str, list[str]] = defaultdict(list) - self.parent: dict[str, str] = {} - # blocks edges with their creation timestamp, for temporal replay - self.block_edges_ts: list[tuple[str, str, str]] = [] - for src, dst, kind, created in edges: - if kind == "blocks": - self.blockers[src].add(dst) - self.blocking[dst].add(src) - self.block_edges_ts.append((src, dst, created)) - elif kind == "parent-child": - self.children[dst].append(src) - self.parent[src] = dst - self.dangling = sorted({(a, b, t) for a, b, t, _ in edges if b not in issues or a not in issues}) - - def is_open(bid: str) -> bool: - rec = issues.get(bid) - return rec is not None and rec["status"] != "closed" - - self.is_open = is_open - self.blocked = [i for i in self.open_ids if any(is_open(b) for b in self.blockers[i])] - self.ready = [i for i in self.open_ids if not any(is_open(b) for b in self.blockers[i])] - self.top_blockers = sorted( - ((sum(1 for x in self.blocking[i] if is_open(x)), i) for i in self.open_ids), - reverse=True, - ) - self.top_blockers = [(n, i) for n, i in self.top_blockers if n > 0] - self.cycles = self._find_cycles() - - # ---- epics ------------------------------------------------------- - self.epics: list[dict[str, Any]] = [] - for pid, kids in self.children.items(): - if pid not in issues: - continue - self.epics.append( - { - "id": pid, - "n": len(kids), - "open": sum(1 for k in kids if is_open(k)), - "kids": kids, - "title": str(issues[pid]["title"]), - "type": str(issues[pid]["issue_type"]), - "status": str(issues[pid]["status"]), - } - ) - self.epics.sort(key=lambda e: (-int(e["n"]), str(e["id"]))) - - # ---- id/edge hierarchy disagreement ------------------------------- - self.id_mismatch = sorted( - (child, self.parent[child]) - for child in self.parent - if "." in child and child.rsplit(".", 1)[0] != self.parent[child] - ) - self.id_orphan = sorted(str(r["id"]) for r in rows if "." in str(r["id"]) and str(r["id"]) not in self.parent) - - # ---- health ------------------------------------------------------- - self.closed_parent_open_kids = sorted( - (pid, [k for k in kids if is_open(k)]) - for pid, kids in self.children.items() - if pid in issues and issues[pid]["status"] == "closed" and any(is_open(k) for k in kids) - ) - self.open_parent_all_closed = sorted( - pid - for pid, kids in self.children.items() - if pid in issues and is_open(pid) and kids and not any(is_open(k) for k in kids) - ) - title_counts = Counter(str(r["title"]).strip().lower() for r in rows) - self.dup_titles = sorted( - (title, [str(r["id"]) for r in rows if str(r["title"]).strip().lower() == title]) - for title, n in title_counts.items() - if n > 1 - ) - self.no_ac_open = [ - str(r["id"]) - for r in rows - if r["status"] != "closed" and not str(r.get("acceptance_criteria") or "").strip() - ] - self.closed_no_reason = [ - str(r["id"]) for r in rows if r["status"] == "closed" and not str(r.get("close_reason") or "").strip() - ] - # in_progress rows whose updated_at is older than the staleness budget - self.stale_claims: list[tuple[str, float]] = sorted( - ( - (str(r["id"]), (now - _parse_ts(str(r["updated_at"]))).total_seconds() / 86400) - for r in rows - if r["status"] == "in_progress" - and r.get("updated_at") - and (now - _parse_ts(str(r["updated_at"]))).total_seconds() / 86400 > STALE_CLAIM_DAYS - ), - key=lambda t: -t[1], - ) - - # ---- time ---------------------------------------------------------- - self.created_day = Counter(str(r["created_at"])[:10] for r in rows) - self.closed_day = Counter(str(r["closed_at"])[:10] for r in rows if r.get("closed_at")) - self.days = sorted(set(self.created_day) | set(self.closed_day)) - self.first_created = min(str(r["created_at"]) for r in rows)[:10] - self.last_created = max(str(r["created_at"]) for r in rows)[:10] - self.span_days = ( - _parse_ts(self.last_created + "T00:00:00Z") - _parse_ts(self.first_created + "T00:00:00Z") - ).days + 1 - - self.lead_by_priority: dict[int, list[float]] = defaultdict(list) - for r in rows: - if r["status"] == "closed" and r.get("closed_at"): - delta = (_parse_ts(str(r["closed_at"])) - _parse_ts(str(r["created_at"]))).total_seconds() / 86400 - self.lead_by_priority[int(r["priority"])].append(delta) - self.age_open = [ - (now - _parse_ts(str(r["created_at"]))).total_seconds() / 86400 for r in rows if r["status"] != "closed" - ] - # age (weeks, capped at 3+) x priority heatmap of open work - self.age_prio: Counter[tuple[int, int]] = Counter( - ( - min(int((now - _parse_ts(str(r["created_at"]))).total_seconds() / 86400 // 7), 3), - int(r["priority"]), - ) - for r in rows - if r["status"] != "closed" - ) - - # trailing velocity windows: [now-14d, now) vs [now-28d, now-14d) - w1_start = now - dt.timedelta(days=VELOCITY_WINDOW_DAYS) - w2_start = now - dt.timedelta(days=2 * VELOCITY_WINDOW_DAYS) - - def _window(day_counter: Counter[str], lo: dt.datetime, hi: dt.datetime) -> int: - return sum(n for day, n in day_counter.items() if lo <= _parse_ts(day + "T00:00:00Z") < hi) - - self.win_recent = (_window(self.created_day, w1_start, now), _window(self.closed_day, w1_start, now)) - self.win_prior = ( - _window(self.created_day, w2_start, w1_start), - _window(self.closed_day, w2_start, w1_start), - ) - - # open P0/P1 older than the aged-urgent budget - self.aged_urgent = sorted( - ( - str(r["id"]) - for r in rows - if r["status"] != "closed" - and int(r["priority"]) <= 1 - and (now - _parse_ts(str(r["created_at"]))).total_seconds() / 86400 > AGED_URGENT_DAYS - ), - ) - - # close-reason coverage over time (first vs second half of closures) - closed_rows = sorted( - (r for r in rows if r["status"] == "closed" and r.get("closed_at")), - key=lambda r: str(r["closed_at"]), - ) - half = len(closed_rows) // 2 - self.reason_rate_halves: tuple[float, float] = ( - (sum(1 for r in closed_rows[:half] if str(r.get("close_reason") or "").strip()) / half if half else 0.0), - ( - sum(1 for r in closed_rows[half:] if str(r.get("close_reason") or "").strip()) - / (len(closed_rows) - half) - if len(closed_rows) - half - else 0.0 - ), - ) - - # ---- labels / themes ------------------------------------------------ - self.labels = Counter(str(x) for r in rows for x in (r.get("labels") or [])) - self.area_labels = [x for x in self.labels if x.startswith("area:")] - self.area_singletons = sum(1 for x in self.area_labels if self.labels[x] == 1) - self.labelled = sum(1 for r in rows if any(str(x).startswith("area:") for x in (r.get("labels") or []))) - # mechanical near-synonym detection: one area suffix a strict prefix of another - suffixes = sorted(x.removeprefix("area:") for x in self.area_labels) - self.label_synonym_pairs = sorted( - (a, b) for a in suffixes for b in suffixes if a != b and b.startswith(a) and len(b) - len(a) <= 8 - ) - - rev_area: dict[str, str] = {} - for canon, folded_suffixes in SUBSYSTEMS.items(): - for suffix in folded_suffixes: - rev_area[suffix] = canon - self.by_label_theme: dict[str, Counter[str]] = defaultdict(Counter) - for r in rows: - seen: set[str] = set() - for raw in r.get("labels") or []: - text = str(raw) - if not text.startswith("area:"): - continue - folded = rev_area.get(text.removeprefix("area:")) - if folded and folded not in seen: - seen.add(folded) - self.by_label_theme[folded][str(r["status"])] += 1 - self.by_kw_theme: dict[str, Counter[str]] = defaultdict(Counter) - for r in rows: - text = (str(r["title"]) + " " + str(r.get("description") or "")).lower() - best, score = "unclassified", 0 - for canon, words in KEYWORDS.items(): - hits = sum(text.count(w) for w in words) - if hits > score: - best, score = canon, hits - self.by_kw_theme[best][str(r["status"])] += 1 - - # ---- verdict subset -------------------------------------------------- - self.verdicts: list[tuple[str, str, str, str]] = [] - loose: Counter[str] = Counter() - for r in rows: - text = _blob(r) - strict = VERDICT_STRICT.search(text) - if strict: - self.verdicts.append((str(r["id"]), strict.group(2).upper(), str(r["status"]), strict.group(1).strip())) - found = VERDICT_LOOSE.findall(text) - if found: - loose[Counter(found).most_common(1)[0][0]] += 1 - self.verdict_counts = Counter(v for _, v, _, _ in self.verdicts) - self.loose_counts = loose - - # A bulk relation re-type stamps every converted edge with the repair - # date. Detect it as a same-day spike on the newest association edge. - rel_days = Counter(created[:10] for _, _, kind, created in edges if kind == "relates-to" and created) - self.retyped_spike = 0 - self.retyped_spike_day = "" - if rel_days and self.vocab_state in ("clean", "recurring"): - newest = max(rel_days) - spike = rel_days[newest] - if spike >= 0.2 * sum(rel_days.values()): - self.retyped_spike = spike - self.retyped_spike_day = newest - - # A `related` edge whose created_at is later than the newest - # relates-to edge on the repair day proves the fork already started - # recurring after the repair. - self.related_recurrence: list[tuple[str, str, str]] = [] - if self.retyped_spike_day: - newest_retype_ts = max( - ( - created - for _, _, kind, created in edges - if kind == "relates-to" and created[:10] == self.retyped_spike_day - ), - default="", - ) - for src, dst, kind, created in edges: - if kind == "related" and created > newest_retype_ts: - self.related_recurrence.append((src, dst, created)) - - # ---- P0 reconciliation pass (2026-07-31) --------------------------- - self.reconciliation: list[tuple[str, str, str]] = [] - self.reconciliation_anchored = 0 - for r in rows: - text = _blob(r) - anchor = RECONCILIATION_ANCHOR.search(text) - if not anchor: - continue - self.reconciliation_anchored += 1 - window = text[anchor.end() : anchor.end() + RECONCILIATION_WINDOW] - token = RECONCILIATION_TOKEN_RE.search(window) - if token: - cut = window.find(token.group()) + len(token.group()) - snippet = window[:cut].strip() - self.reconciliation.append((str(r["id"]), token.group(), snippet)) - self.reconciliation_counts = Counter(v for _, v, _ in self.reconciliation) - - # ---- temporal reconstruction ----------------------------------------- - def open_at(self, rec: IssueDict, as_of: dt.datetime) -> bool: - """Whether a bead existed and was not-closed at ``as_of``. - - Derived from ``created_at``/``closed_at``; a reopened bead's earlier - closure is invisible (closed_at reflects only the latest close), and - deleted beads are absent from the export entirely. - """ - if _parse_ts(str(rec["created_at"])) > as_of: - return False - closed = rec.get("closed_at") - return not (closed and _parse_ts(str(closed)) <= as_of) - - def snapshot(self, as_of: dt.datetime) -> dict[str, int]: - """Key backlog metrics reconstructed as of a past instant.""" - exists = [r for r in self.rows if _parse_ts(str(r["created_at"])) <= as_of] - open_rows = [r for r in exists if self.open_at(r, as_of)] - open_set = {str(r["id"]) for r in open_rows} - blocked_by_open: dict[str, bool] = defaultdict(bool) - for src, dst, created in self.block_edges_ts: - if created and _parse_ts(created) <= as_of and src in open_set and dst in open_set: - blocked_by_open[src] = True - blocked_n = sum(1 for bid in open_set if blocked_by_open[bid]) - return { - "total": len(exists), - "open": len(open_rows), - "closed": len(exists) - len(open_rows), - "p0_open": sum(1 for r in open_rows if int(r["priority"]) == 0), - "blocked": blocked_n, - "ready": len(open_set) - blocked_n, - } - - def daily_series(self) -> list[tuple[str, int, int]]: - """(day, open count, ready count) for every day of the backlog's life.""" - start = _parse_ts(self.first_created + "T00:00:00Z") - out: list[tuple[str, int, int]] = [] - day = start - while day <= self.now: - eod = day + dt.timedelta(days=1) - snap = self.snapshot(eod) - out.append((day.strftime("%Y-%m-%d"), snap["open"], snap["ready"])) - day = eod - return out - - def epic_open_series(self, epic_id: str, days: Sequence[str]) -> list[int]: - """Open-children count per day for one epic (membership as of today).""" - kids = [self.issues[k] for k in self.children.get(epic_id, []) if k in self.issues] - out: list[int] = [] - for day in days: - eod = _parse_ts(day + "T00:00:00Z") + dt.timedelta(days=1) - out.append(sum(1 for k in kids if self.open_at(k, eod))) - return out - - # ---- graph structure -------------------------------------------------- - def _find_cycles(self) -> list[list[str]]: - colour: dict[str, int] = {} - cycles: list[list[str]] = [] - - def visit(start: str) -> None: - stack: list[tuple[str, Iterable[str]]] = [(start, iter(sorted(self.blockers[start])))] - colour[start] = 1 - path = [start] - while stack: - node, it = stack[-1] - advanced = False - for nxt in it: - if nxt not in self.issues: - continue - state = colour.get(nxt) - if state == 1: - cycles.append(path[path.index(nxt) :] + [nxt]) - elif state is None: - colour[nxt] = 1 - path.append(nxt) - stack.append((nxt, iter(sorted(self.blockers[nxt])))) - advanced = True - break - if not advanced: - colour[node] = 2 - stack.pop() - path.pop() - - for bid in sorted(self.issues): - if colour.get(bid) is None: - visit(bid) - return cycles - - def rate(self, priority: int) -> float: - total = self.priority[priority] - closed = self.matrix.get(("closed", priority), 0) - return closed / total if total else 0.0 - - def densest_cluster(self) -> list[str]: - """Largest weakly-connected component of the open-only `blocks` graph.""" - adj: dict[str, set[str]] = defaultdict(set) - open_set = set(self.open_ids) - for src, dst, kind, _ in self.edges: - if kind == "blocks" and src in open_set and dst in open_set: - adj[src].add(dst) - adj[dst].add(src) - seen: set[str] = set() - best: list[str] = [] - for node in sorted(adj): - if node in seen: - continue - comp: list[str] = [] - queue = [node] - seen.add(node) - while queue: - cur = queue.pop() - comp.append(cur) - for nxt in sorted(adj[cur]): - if nxt not in seen: - seen.add(nxt) - queue.append(nxt) - if len(comp) > len(best): - best = comp - self._cluster_adj = adj - return sorted(best) - - def cluster_core(self, component: Sequence[str], min_degree: int = 3) -> list[str]: - """The legible core of a component: nodes with >= min_degree neighbours in it.""" - adj = self._cluster_adj - member = set(component) - return sorted(n for n in component if len({x for x in adj[n] if x in member}) >= min_degree) - - def descendants(self, root: str, max_depth: int = 6) -> set[str]: - """All parent-child descendants of a bead, cycle-safe.""" - out: set[str] = set() - frontier = [root] - for _ in range(max_depth): - nxt: list[str] = [] - for node in frontier: - for kid in self.children.get(node, []): - if kid not in out: - out.add(kid) - nxt.append(kid) - if not nxt: - break - frontier = nxt - return out - - def parallel_frontier(self, top_n: int = 8) -> list[dict[str, Any]]: - """The largest set of big open epics whose open subtrees share no - ``blocks`` component -- work that can run in parallel with no ordering - constraint between the programs. Greedy by open-descendant count. - """ - # component id per open bead in the blocks graph - self.densest_cluster() # populates _cluster_adj - comp_of: dict[str, int] = {} - cid = 0 - for node in sorted(self._cluster_adj): - if node in comp_of: - continue - queue = [node] - comp_of[node] = cid - while queue: - cur = queue.pop() - for nxt in self._cluster_adj[cur]: - if nxt not in comp_of: - comp_of[nxt] = cid - queue.append(nxt) - cid += 1 - - candidates: list[dict[str, Any]] = [] - for e in self.epics: - desc = {d for d in self.descendants(str(e["id"])) if self.is_open(d)} - if len(desc) < 3: - continue - comps = {comp_of[d] for d in desc if d in comp_of} - candidates.append( - { - "id": str(e["id"]), - "title": str(e["title"]), - "open_desc": len(desc), - "desc": desc, - "comps": comps, - "urgent": sum(1 for d in desc if int(self.issues[d]["priority"]) <= 2), - } - ) - candidates.sort(key=lambda c: (-int(c["open_desc"]), str(c["id"]))) - - chosen: list[dict[str, Any]] = [] - used_comps: set[int] = set() - used_beads: set[str] = set() - for cand in candidates[:top_n]: - if cand["comps"] & used_comps or cand["desc"] & used_beads: - continue - chosen.append(cand) - used_comps |= set(cand["comps"]) - used_beads |= set(cand["desc"]) - return chosen - - -# -------------------------------------------------------------------------- -# computed insights: the interpretation layer. Every entry is emitted by a -# condition checked against the data; regeneration cannot leave a stale claim. -# -------------------------------------------------------------------------- -@dataclass -class Insight: - sev: str # bad | warn | info | ok - title: str # html - body: str # html - ev: str = "derived" # measured | derived - chips: list[str] = field(default_factory=list) # bead ids to render as chips - - -_SEV_ORDER = {"bad": 0, "warn": 1, "info": 2, "ok": 3} - - -def compute_insights(facts: Facts, schema_gap: dict[str, Any], snaps: dict[str, dict[str, int]]) -> list[Insight]: - out: list[Insight] = [] - - # -- schema gap: merged fixes inert until rebuild ----------------------- - if schema_gap.get("available") and schema_gap.get("blockers"): - pending = ( - facts.reconciliation_counts["FIXED-PENDING-REBUILD"] + facts.reconciliation_counts["FIXED-PENDING-DEPLOY"] - ) - pending_note = ( - f" The dated reconciliation pass confirmed {pending} beads merged and inert behind it." - if pending - else "" - ) - out.append( - Insight( - "bad", - f"Live archive is {len(schema_gap['blockers'])} SEMANTIC_REPARSE version(s) behind origin/master " - f"(v{schema_gap['live_version']} vs v{schema_gap['declared']})", - "Merged fixes at those versions are real, reviewed, and completely inert until " - "polylogue ops reset --index && polylogued run executes." + pending_note, - ev="measured", - ) - ) - - # -- dangling refs ------------------------------------------------------ - if facts.dangling: - targets = ", ".join(f"{esc(b)}" for _, b, _ in facts.dangling[:6]) - out.append( - Insight( - "bad", - f"{len(facts.dangling)} dependency edge(s) point at ids that do not exist", - f"Unresolvable targets: {targets}. The blocking intent behind each edge is silently absent " - "from every query. Id references in dependency edges are not validated on write.", - ev="measured", - ) - ) - - # -- cycles ------------------------------------------------------------- - if facts.cycles: - first = " → ".join(esc(x) for x in facts.cycles[0]) - out.append( - Insight( - "bad", - f"{len(facts.cycles)} cycle(s) in the blocks graph", - f"A mutual-blocking loop can never become ready. First cycle: {first}.", - ev="measured", - ) - ) - else: - out.append( - Insight( - "ok", - "No cycles in the blocks graph", - f"Verified by DFS colouring over all {facts.total:,} beads and " - f"{facts.dep_types['blocks']:,} blocks edges. Stated explicitly rather than left as an " - "absence: a graph grown this fast by many independent lanes is where mutual-blocking " - "pairs appear by accident.", - ev="measured", - ) - ) - - # -- velocity: trailing window vs the one before it -------------------- - (rc, rx), (pc, px) = facts.win_recent, facts.win_prior - r_net, p_net = rc - rx, pc - px - if r_net > 0 and p_net > 0: - sev, verdict = "warn", "growing in both trailing windows — still in discovery, not burn-down" - elif r_net < 0 <= p_net: - sev, verdict = "ok", "tipped from growth to net drain in the most recent window" - elif r_net < 0 and p_net < 0: - sev, verdict = "ok", "draining in both trailing windows" - elif r_net > 0 >= p_net: - sev, verdict = "warn", "flipped back from drain to growth in the most recent window" - else: - sev, verdict = "info", "roughly flat" - out.append( - Insight( - sev, - f"Backlog is {verdict}", - f"Last {VELOCITY_WINDOW_DAYS}d: {rc} created / {rx} closed (net {r_net:+d}). " - f"Prior {VELOCITY_WINDOW_DAYS}d: {pc} created / {px} closed (net {p_net:+d}). " - "Velocity-derived completion estimates are meaningful only once the trailing window " - "is reliably net-negative.", - ) - ) - - # -- ready fraction ----------------------------------------------------- - if facts.open_total: - frac = len(facts.ready) / facts.open_total - if frac >= 0.6: - out.append( - Insight( - "info", - f"{frac:.0%} of open work is unblocked — throughput-bound, not dependency-bound", - f"{len(facts.ready):,} of {facts.open_total:,} open beads have no open blocker. " - "No amount of unblocking changes the picture; execution capacity is the constraint.", - ev="measured", - ) - ) - elif frac <= 0.35: - out.append( - Insight( - "warn", - f"Only {frac:.0%} of open work is unblocked — the graph is dependency-bound", - f"{len(facts.blocked):,} of {facts.open_total:,} open beads wait on an open blocker. " - "Unblocking the top blockers is the highest-leverage move; see Topology.", - ev="measured", - ) - ) - - # -- priority ladder ---------------------------------------------------- - rates = [facts.rate(p) for p in range(5)] - leads = [_median(facts.lead_by_priority[p]) for p in range(5)] - rate_inversions = [p for p in range(3) if facts.priority[p + 1] and rates[p] < rates[p + 1]] - lead_inversions = [p for p in range(3) if leads[p + 1] and leads[p] > leads[p + 1]] - lead_str = " / ".join(f"{v:.1f}" for v in leads) - rate_str = " / ".join(f"{v:.0%}" for v in rates) - if not rate_inversions and not lead_inversions: - out.append( - Insight( - "info", - "The priority scale is load-bearing — do not renumber it", - f"Median lead time rises monotonically P0→P3 ({lead_str} days for P0–P4) and " - f"closure rate falls monotonically P0→P3 ({rate_str}). An inflated priority label " - "would show a flat curve; this one sorts work at every rung.", - ) - ) - else: - inv = ", ".join(f"P{p}→P{p + 1}" for p in sorted(set(rate_inversions + lead_inversions))) - out.append( - Insight( - "warn", - f"Priority ladder inverts at {inv}", - f"Median lead days P0–P4: {lead_str}. Closure rates: {rate_str}. " - "Where a lower-urgency rung closes faster or more often than the rung above it, the " - "label is not carrying the information a scheduler needs.", - ) - ) - - # -- aged urgency ------------------------------------------------------- - if facts.aged_urgent: - out.append( - Insight( - "warn", - f"{len(facts.aged_urgent)} open P0/P1 bead(s) older than {AGED_URGENT_DAYS} days", - "Urgency that sits is urgency mislabelled or capacity missing — either way it is " - "the first shelf to triage. Oldest listed below.", - chips=facts.aged_urgent[:8], - ) - ) - - # -- stale in-progress claims ------------------------------------------ - if facts.stale_claims: - out.append( - Insight( - "warn", - f"{len(facts.stale_claims)} in-progress claim(s) untouched for over {STALE_CLAIM_DAYS} days", - "A claim without activity blocks other agents from picking the bead up. " - "Release the claim or finish the work.", - ev="measured", - chips=[bid for bid, _ in facts.stale_claims[:8]], - ) - ) - - # -- parallel frontier -------------------------------------------------- - frontier = facts.parallel_frontier() - if len(frontier) >= 2: - total_open_desc = sum(int(c["open_desc"]) for c in frontier) - share = total_open_desc / facts.open_total if facts.open_total else 0.0 - names = ", ".join(f"{esc(str(c['id']))}" for c in frontier) - out.append( - Insight( - "info", - f"{len(frontier)} large programs share no blocking path — they can run in parallel", - f"{names} hold {total_open_desc} open descendants between them ({share:.0%} of all open " - "work) and their open subtrees occupy disjoint components of the blocks " - "graph: no ordering constraint exists between the programs. Detail in Topology.", - ) - ) - - # -- relation vocabulary fork ------------------------------------------ - relates, related = facts.dep_types["relates-to"], facts.dep_types["related"] - if facts.vocab_state == "split": - out.append( - Insight( - "bad", - "Split relation vocabulary: relates-to vs related", - f"{relates} + {related} edges are one relation under two names. Any association query " - "silently returns a partial subset. Pick one spelling, rewrite the other, constrain the field.", - ev="measured", - ) - ) - elif facts.vocab_state == "recurring": - out.append( - Insight( - "warn", - "Relation vocabulary was unified, then forked again", - f"A bulk re-type unified the association relation, but wrote no constraint, and " - f"{related} fresh related edge(s) landed after the repair window. " - "The data was cleaned; the write path is still open.", - ev="measured", - ) - ) - - # -- label vocabulary --------------------------------------------------- - if facts.label_synonym_pairs: - pairs = ", ".join(f"{esc(a)}/{esc(b)}" for a, b in facts.label_synonym_pairs[:6]) - out.append( - Insight( - "warn", - f"{len(facts.area_labels)} distinct area:* labels, " - f"{facts.area_singletons} used once, {len(facts.label_synonym_pairs)} near-synonym pair(s)", - f"Prefix-detected synonym pairs: {pairs}. A free-text label field with no controlled " - "vocabulary degrades into per-author spelling; grouping by raw label undercounts every " - "affected subsystem. The Themes section folds them through an explicit map.", - ev="measured", - ) - ) - - # -- review queue ------------------------------------------------------- - if facts.open_parent_all_closed: - out.append( - Insight( - "warn", - f"{len(facts.open_parent_all_closed)} open parent(s) with every child closed — " - "review, not auto-close", - "Children closing is evidence a parent is ready for adjudication against its own " - "acceptance criteria, never evidence that it is done. Queue in Health.", - ) - ) - - # -- duplicate titles --------------------------------------------------- - if facts.dup_titles: - out.append( - Insight( - "info", - f"{len(facts.dup_titles)} duplicate title pair(s)", - "Independently filed beads describing the same work; listed in Health with both ids.", - ev="measured", - ) - ) - - # -- close-reason discipline ------------------------------------------- - r1, r2 = facts.reason_rate_halves - if facts.status["closed"] >= 20: - direction = "improving" if r2 > r1 + 0.02 else ("degrading" if r2 < r1 - 0.02 else "steady") - sev = "info" if r2 >= 0.9 else "warn" - out.append( - Insight( - sev, - f"Close-reason coverage is {r2:.0%} in the newer half of closures ({direction})", - f"Earlier half {r1:.0%}, newer half {r2:.0%}; " - f"{len(facts.closed_no_reason)} closed bead(s) in total carry no reason. A close without " - "a reason leaves no durable record of why the work ended.", - ) - ) - - # -- open-count trend vs 7d -------------------------------------------- - now_open = snaps["now"]["open"] - week_open = snaps["7d"]["open"] - if week_open: - delta = now_open - week_open - if abs(delta) >= max(5, 0.03 * week_open): - direction = "grew" if delta > 0 else "shrank" - out.append( - Insight( - "info", - f"Open backlog {direction} {abs(delta)} bead(s) in the last 7 days ({week_open} → {now_open})", - "Reconstructed from created/closed timestamps; deleted beads and reopen history " - "are invisible to the reconstruction.", - ) - ) - - out.sort(key=lambda i: _SEV_ORDER.get(i.sev, 9)) - return out - - -# -------------------------------------------------------------------------- -# rendering helpers -# -------------------------------------------------------------------------- -def esc(text: str) -> str: - return html.escape(str(text), quote=True) - - -def chip(facts: Facts, bid: str, *, title: bool = False) -> str: - """The compact bead notation used everywhere in the report.""" - rec = facts.issues.get(bid) - if rec is None: - return f'{esc(bid)}?' - prio = int(rec["priority"]) - status = str(rec["status"]) - sglyph, sname = STATUS_GLYPH.get(status, ("?", status)) - tglyph, tname = TYPE_GLYPH.get(str(rec["issue_type"]), ("?", str(rec["issue_type"]))) - inn = sum(1 for b in facts.blockers[bid] if facts.is_open(b)) - out = sum(1 for b in facts.blocking[bid] if facts.is_open(b)) - deg = "" - if inn or out: - deg = f'{"↑" + str(inn) if inn else ""}{"↓" + str(out) if out else ""}' - tip = f"P{prio} · {sname} · {tname} · {inn} open blocker(s) · blocks {out} open" - text = f' {esc(str(rec["title"])[:64])}' if title else "" - return ( - f'' - f'P{prio}' - f'{sglyph}' - f'{esc(bid)}' - f'{tglyph}{deg}{text}' - ) - - -def delta_chip(now_v: int, then_v: int, *, up_is_bad: bool = False) -> str: - """A signed 7-day delta rendered next to a stat tile value.""" - d = now_v - then_v - if d == 0: - return '±0' - arrow = "▲" if d > 0 else "▼" - bad = (d > 0) == up_is_bad - cls = "worse" if bad else "better" - return f'{arrow}{abs(d)}' - - -def sparkline_svg(values: Sequence[int], *, label: str = "") -> str: - """A tiny inline trend line; viewBox only, currentColor stroke.""" - if len(values) < 2: - return "" - lo, hi = min(values), max(values) - span = (hi - lo) or 1 - w = (len(values) - 1) * 3 - pts = " ".join(f"{i * 3},{18 - 15 * (v - lo) / span:.1f}" for i, v in enumerate(values)) - title = f"{esc(label)} {values[0]} → {values[-1]}" if label else "" - return ( - f'{title}' - f'' - ) - - -def burnup_svg(series: Sequence[tuple[str, int, int]]) -> str: - """Open + ready count over the backlog's whole life, one small chart.""" - if len(series) < 2: - return "" - peak = max(v for _, v, _ in series) or 1 - n = len(series) - w, h = (n - 1) * 10, 60 - - def path(values: Sequence[int]) -> str: - return " ".join(f"{i * 10},{h - 4 - (h - 12) * (v / peak):.1f}" for i, v in enumerate(values)) - - open_pts = path([row[1] for row in series]) - ready_pts = path([row[2] for row in series]) - parts = [ - f'', - f'', - f'', - ] - for i, (day, open_n, ready_n) in enumerate(series): - parts.append( - f'' - f"{esc(day)}: {open_n} open, {ready_n} ready" - ) - parts.append("") - return "".join(parts) - - -def histogram_svg(days: Sequence[str], series: Sequence[tuple[str, Counter[str], str]]) -> str: - """Grouped per-day bar chart, pure SVG, viewBox only.""" - if not days: - return "" - step = 10.0 - width = len(days) * step - height = 46.0 - peak = max((c[day] for _, c, _ in series for day in days), default=1) or 1 - parts = [ - f'' - ] - bw = step / (len(series) + 1) - for si, (_, counts, cls) in enumerate(series): - for di, day in enumerate(days): - value = counts[day] - if not value: - continue - bh = (value / peak) * (height - 8) - x = di * step + si * bw - parts.append( - f'{esc(day)}: {value}' - ) - parts.append("") - return "".join(parts) - - -def cluster_svg(facts: Facts, nodes: Sequence[str]) -> str: - """Layered DAG of one connected `blocks` component. Layer = longest path from a root.""" - node_set = set(nodes) - layer: dict[str, int] = {} - - def depth(bid: str, seen: frozenset[str]) -> int: - if bid in layer: - return layer[bid] - ups = [b for b in facts.blockers[bid] if b in node_set and b not in seen] - value = 0 if not ups else 1 + max(depth(u, seen | {bid}) for u in ups) - layer[bid] = value - return value - - for bid in nodes: - depth(bid, frozenset()) - rows: dict[int, list[str]] = defaultdict(list) - for bid in sorted(nodes): - rows[layer[bid]].append(bid) - pos: dict[str, tuple[float, float]] = {} - col_w, row_h = 210.0, 34.0 - for lv, members in rows.items(): - for i, bid in enumerate(sorted(members)): - pos[bid] = (18 + lv * col_w, 22 + i * row_h) - width = 36 + (max(rows) + 1) * col_w - height = 30 + max(len(v) for v in rows.values()) * row_h - parts = [ - f'', - '', - ] - for src in nodes: - for dst in facts.blockers[src]: - if dst not in node_set: - continue - x1, y1 = pos[dst] - x2, y2 = pos[src] - parts.append( - f'' - ) - for bid, (x, y) in pos.items(): - rec = facts.issues[bid] - prio = int(rec["priority"]) - parts.append( - f'' - f'P{prio} {esc(bid)}' - f"{esc(str(rec['title'])[:110])}" - ) - parts.append("") - return "".join(parts) - - -def tree_html(facts: Facts, root: str, depth: int = 0) -> str: - kids = sorted(facts.children.get(root, [])) - if not kids or depth > 1: - return "" - items = [] - for kid in kids: - if kid not in facts.issues: - continue - sub = tree_html(facts, kid, depth + 1) - title = esc(str(facts.issues[kid]["title"])[:78]) - items.append(f'
  • {chip(facts, kid)} {title}{sub}
  • ') - return f'
      {"".join(items)}
    ' - - -def fill_bar(open_n: int, total: int) -> str: - done = total - open_n - pct = 100 * done / total if total else 0 - return ( - f'' - f'' - f'{done}/{total}' - ) - - -def heatmap_html(facts: Facts) -> str: - """Age (weeks) x priority heatmap of open work; sequential single-hue fill.""" - age_labels = ["0–6d", "7–13d", "14–20d", "21d+"] - peak = max(facts.age_prio.values()) if facts.age_prio else 1 - parts = ['
    '] - for p in range(5): - parts.append(f'') - parts.append('') - for w, label in enumerate(age_labels): - row_total = sum(facts.age_prio.get((w, p), 0) for p in range(5)) - parts.append(f"") - for p in range(5): - n = facts.age_prio.get((w, p), 0) - pct = 58 * n / peak if peak else 0 - parts.append( - f'' - ) - parts.append(f'') - parts.append("
    age \\ priorityP{p}row
    {label}{n or ""}{row_total}
    ") - return "".join(parts) - - -def qa(command: str, label: str) -> str: - """Attach the exact invocation that produced a figure.""" - return ( - '" - ) - - -CSS = """ -:root{ - --bg:#f6f7f9; --panel:#ffffff; --ink:#1a2129; --muted:#5b6773; - --line:#dde3e9; --accent:#2563eb; --accent-ink:#ffffff; - --ok:#0f7b46; --ok-bg:#e2f5eb; --warn:#8a5a00; --warn-bg:#fdf0d3; - --bad:#a01c1c; --bad-bg:#fbe4e4; --info:#1d4ed8; --info-bg:#e3ebfd; - --todo:#6b21a8; --todo-bg:#f1e6fb; --code-bg:#eef1f4; - --p0:#a01c1c; --p1:#b45309; --p2:#1d4ed8; --p3:#0f7b46; --p4:#5b6773; -} -@media (prefers-color-scheme: dark){:root{ - --bg:#12161b; --panel:#1a2027; --ink:#e6ebf0; --muted:#94a1ad; - --line:#2b333c; --accent:#5b8def; --accent-ink:#0d1117; - --ok:#4cc98a; --ok-bg:#12301f; --warn:#e2b93b; --warn-bg:#33290e; - --bad:#ef7070; --bad-bg:#391717; --info:#7ea6f4; --info-bg:#16223b; - --todo:#c793ef; --todo-bg:#2a1738; --code-bg:#232a32; - --p0:#ef7070; --p1:#e2b93b; --p2:#7ea6f4; --p3:#4cc98a; --p4:#94a1ad; -}} -:root[data-theme="light"]{color-scheme:light} -:root[data-theme="dark"]{color-scheme:dark} -:root[data-theme="dark"]{ - --bg:#12161b; --panel:#1a2027; --ink:#e6ebf0; --muted:#94a1ad; - --line:#2b333c; --accent:#5b8def; --accent-ink:#0d1117; - --ok:#4cc98a; --ok-bg:#12301f; --warn:#e2b93b; --warn-bg:#33290e; - --bad:#ef7070; --bad-bg:#391717; --info:#7ea6f4; --info-bg:#16223b; - --todo:#c793ef; --todo-bg:#2a1738; --code-bg:#232a32; - --p0:#ef7070; --p1:#e2b93b; --p2:#7ea6f4; --p3:#4cc98a; --p4:#94a1ad; -} -:root[data-theme="light"]{ - --bg:#f6f7f9; --panel:#ffffff; --ink:#1a2129; --muted:#5b6773; - --line:#dde3e9; --accent:#2563eb; --accent-ink:#ffffff; - --ok:#0f7b46; --ok-bg:#e2f5eb; --warn:#8a5a00; --warn-bg:#fdf0d3; - --bad:#a01c1c; --bad-bg:#fbe4e4; --info:#1d4ed8; --info-bg:#e3ebfd; - --todo:#6b21a8; --todo-bg:#f1e6fb; --code-bg:#eef1f4; - --p0:#a01c1c; --p1:#b45309; --p2:#1d4ed8; --p3:#0f7b46; --p4:#5b6773; -} -:root{--fs:17px} -*{box-sizing:border-box} -html{font-size:var(--fs)} -body{margin:0;background:var(--bg);color:var(--ink); - font:1rem/1.62 system-ui,-apple-system,"Segoe UI",sans-serif} -header.page{position:sticky;top:0;z-index:5;background:var(--panel); - border-bottom:1px solid var(--line);padding:.7rem 1.2rem; - display:flex;flex-wrap:wrap;align-items:baseline;gap:.6rem} -header.page h1{font-size:1.3rem;margin:0} -header.page .spacer{flex:1} -.chip{display:inline-block;padding:.12rem .6rem;border:1px solid var(--line); - border-radius:99px;font-size:.85rem;color:var(--muted);background:var(--bg)} -button.theme,button.fs{border:1px solid var(--line);background:var(--bg);color:var(--ink); - border-radius:.4rem;padding:.2rem .6rem;cursor:pointer;font-size:.85rem} -.layout{display:grid;grid-template-columns:16rem minmax(0,1fr); - max-width:80rem;margin:0 auto;gap:1.2rem;padding:1.2rem} -@media(max-width:62rem){.layout{grid-template-columns:minmax(0,1fr)}nav#toc{display:none}} -nav#toc{position:sticky;top:3.8rem;align-self:start;font-size:.92rem; - border-right:1px solid var(--line);padding-right:.8rem;max-height:85vh;overflow:auto} -nav#toc a{display:block;color:var(--muted);text-decoration:none;padding:.16rem 0} -nav#toc a.h3{padding-left:.9rem;font-size:.87rem} -nav#toc a:hover{color:var(--accent)} -main{min-width:0} -section{background:var(--panel);border:1px solid var(--line);border-radius:.6rem; - padding:1.1rem 1.3rem;margin-bottom:1.1rem} -h2{font-size:1.28rem;margin:.1rem 0 .7rem;line-height:1.3} -h3{font-size:1.07rem;margin:1.1rem 0 .45rem;line-height:1.35} -p{margin:.5rem 0;max-width:76ch} -a{color:var(--accent)} -code,kbd{background:var(--code-bg);border-radius:.3rem;padding:.08rem .35rem; - font:.9em ui-monospace,SFMono-Regular,Menlo,monospace} -pre{background:var(--code-bg);padding:.75rem .9rem;border-radius:.5rem; - overflow-x:auto;font-size:.92rem;line-height:1.5} -pre code{background:none;padding:0;font-size:1em} -.tiles{display:grid;grid-template-columns:repeat(auto-fit,minmax(10.5rem,1fr)); - gap:.7rem;margin:.4rem 0 .9rem} -.tile{border:1px solid var(--line);border-radius:.55rem;padding:.6rem .85rem;background:var(--bg)} -.tile .n{font-size:1.6rem;font-weight:650;display:block;line-height:1.2} -.tile .l{font-size:.85rem;color:var(--muted)} -.tile.qa{position:relative;cursor:help} -.tile.qa::after{content:"\\2315";position:absolute;top:.35rem;right:.45rem;opacity:.35;font-size:.8rem} -.dl{font-size:.8rem;font-weight:700;margin-left:.35rem;vertical-align:.25em} -.dl.worse{color:var(--bad)} .dl.better{color:var(--ok)} .dl.flat{color:var(--muted)} -.badge{display:inline-block;font-size:.8rem;font-weight:600;letter-spacing:.02em; - padding:.1rem .55rem;border-radius:99px;white-space:nowrap} -.ok{color:var(--ok);background:var(--ok-bg)} .warn{color:var(--warn);background:var(--warn-bg)} -.bad{color:var(--bad);background:var(--bad-bg)} .info{color:var(--info);background:var(--info-bg)} -.todo{color:var(--todo);background:var(--todo-bg)} -.tablewrap{overflow-x:auto} -table{border-collapse:collapse;width:100%;font-size:.95rem} -th,td{border-bottom:1px solid var(--line);text-align:left;padding:.42rem .6rem;vertical-align:top} -th{color:var(--muted);font-size:.85rem;text-transform:uppercase;letter-spacing:.04em; - cursor:pointer;user-select:none;white-space:nowrap} -th.nosort{cursor:default} -tr:hover td{background:color-mix(in srgb, var(--accent) 6%, transparent)} -td.num{text-align:right;font-variant-numeric:tabular-nums} -details{border:1px solid var(--line);border-radius:.5rem;padding:.5rem .85rem;margin:.5rem 0;background:var(--bg)} -summary{cursor:pointer;font-weight:600;font-size:1rem} -details[open]{padding-bottom:.6rem} -input.filter{width:100%;max-width:26rem;margin:.2rem 0 .55rem;padding:.35rem .65rem; - border:1px solid var(--line);border-radius:.4rem;background:var(--bg);color:var(--ink);font-size:.95rem} -footer{color:var(--muted);font-size:.88rem;text-align:center;padding:1rem} -.meta{display:grid;grid-template-columns:auto 1fr;gap:.2rem .9rem;font-size:.9rem; - border-left:3px solid var(--line);padding:.1rem 0 .1rem .9rem;margin:.2rem 0 .9rem} -.meta dt{color:var(--muted)} -.meta dd{margin:0} -.ev{display:inline-block;font-size:.72rem;font-weight:700;letter-spacing:.03em; - padding:0 .35rem;border-radius:.25rem;vertical-align:.1em;text-transform:uppercase} -.ev-measured{color:var(--ok);background:var(--ok-bg)} -.ev-derived{color:var(--info);background:var(--info-bg)} -.ev-inferred{color:var(--warn);background:var(--warn-bg)} -.ev-assumed{color:var(--bad);background:var(--bad-bg)} -time.age{color:var(--muted);font-size:.85em;font-variant-numeric:tabular-nums} -time.age.stale{color:var(--warn);font-weight:600} -time.age.stale::after{content:" \\26A0"} -a.path{font:.9em ui-monospace,SFMono-Regular,Menlo,monospace;background:var(--code-bg); - border-radius:.3rem;padding:.08rem .35rem;text-decoration:none;border-bottom:1px dotted var(--accent)} -.pop{position:fixed;z-index:50;max-width:min(46rem,92vw);max-height:70vh;overflow:auto; - background:var(--panel);border:1px solid var(--line);border-radius:.5rem; - box-shadow:0 .6rem 2rem rgba(0,0,0,.28);padding:.7rem .9rem;font-size:.92rem} -.pop h4{margin:0 0 .35rem;font-size:.92rem;font-family:ui-monospace,monospace;color:var(--muted)} -.pop pre{margin:0;max-height:52vh;font-size:.86rem} -.pop .pin{float:right;border:none;background:none;color:var(--muted);cursor:pointer} -blockquote.q{position:relative;margin:.55rem 0 .4rem;padding:.6rem 1.1rem .6rem 2.3rem; - background:var(--panel);border:1px solid var(--line);border-left:3px solid var(--todo); - border-radius:0 .45rem .45rem 0; - font:italic 1rem/1.55 Georgia,"Iowan Old Style","Noto Serif",ui-serif,serif} -blockquote.q::before{content:"\\201C";position:absolute;left:.45rem;top:.15rem; - font:italic 2.2rem/1 Georgia,ui-serif,serif;color:var(--todo);opacity:.5} -blockquote.q cite{display:block;margin-top:.4rem;font:600 .72rem/1 system-ui,sans-serif; - text-transform:uppercase;letter-spacing:.05em;color:var(--muted);font-style:normal} - -/* ---- BEAD NOTATION -------------------------------------------------- */ -.bd{display:inline-flex;align-items:center;gap:.28rem;border:1px solid var(--line); - border-radius:.35rem;padding:.02rem .34rem .02rem .05rem;background:var(--bg); - font:.82rem/1.5 ui-monospace,SFMono-Regular,Menlo,monospace;white-space:nowrap; - border-left:3px solid var(--pc,var(--muted))} -.bd.p0{--pc:var(--p0)} .bd.p1{--pc:var(--p1)} .bd.p2{--pc:var(--p2)} -.bd.p3{--pc:var(--p3)} .bd.p4{--pc:var(--p4)} -.bd .bp{font-weight:700;color:var(--pc);font-size:.78rem;padding-left:.3rem} -.bd .bs{font-size:.82rem;opacity:.85} -.bd.st-closed{opacity:.62} -.bd.st-closed .bi{text-decoration:line-through;text-decoration-thickness:1px} -.bd.st-in_progress{box-shadow:inset 0 0 0 1px color-mix(in srgb,var(--accent) 45%,transparent)} -.bd.st-deferred{opacity:.75;border-style:dashed} -.bd .bi{color:var(--ink)} -.bd .by{opacity:.75} -.bd .bg{color:var(--accent);font-size:.75rem;font-weight:700} -.bd.missing{border-left-color:var(--bad);background:var(--bad-bg)} -.bd .bl{font:.8rem/1.5 system-ui,sans-serif;color:var(--muted);max-width:26rem; - overflow:hidden;text-overflow:ellipsis} -.legend{display:grid;grid-template-columns:repeat(auto-fit,minmax(13rem,1fr));gap:.5rem 1.2rem; - font-size:.9rem;margin:.6rem 0} -.legend div{display:flex;gap:.45rem;align-items:baseline} -.legend b{font:.85rem ui-monospace,monospace;color:var(--accent);min-width:2.2rem} - -/* ---- distribution bars ---- */ -.dist{position:relative;text-align:right;white-space:nowrap} -.dist span{position:absolute;left:0;top:.25rem;bottom:.25rem;width:var(--w); - background:var(--accent);opacity:.18;border-radius:.2rem} -.dist b{position:relative;font-variant-numeric:tabular-nums} - -/* ---- epic fill bars ---- */ -.fill{display:inline-block;width:9rem;height:.62rem;border-radius:99px;background:var(--code-bg); - overflow:hidden;vertical-align:middle;border:1px solid var(--line)} -.fill>span{display:block;height:100%;background:var(--ok)} -.filln{font:.78rem ui-monospace,monospace;color:var(--muted);margin-left:.4rem; - font-variant-numeric:tabular-nums} - -/* ---- matrix / heatmap ---- */ -table.mx td.c{text-align:right;font-variant-numeric:tabular-nums;position:relative} -table.mx td.c i{position:absolute;left:.2rem;top:.25rem;bottom:.25rem;border-radius:.2rem; - background:var(--accent);opacity:.16;font-style:normal} -table.mx td.c b{position:relative} -table.hm td.c{text-align:center;font-variant-numeric:tabular-nums;min-width:3.2rem} -table.hm td,table.hm th{border:1px solid var(--line)} - -/* ---- charts ---- */ -.chart svg{width:100%;height:auto;max-height:7rem} -.bar{fill:var(--accent);opacity:.75} -.bar.closed{fill:var(--ok);opacity:.8} -.ln{stroke-width:2;vector-effect:non-scaling-stroke} -.ln.open{stroke:var(--accent)} -.ln.ready{stroke:var(--ok)} -.hit{fill:transparent} -.spark{width:6.5rem;height:1.1rem;color:var(--accent);vertical-align:middle} -.dag{height:auto;color:var(--muted);min-width:44rem;width:100%} -.dag .dn rect{fill:var(--bg);stroke:var(--pc,var(--muted));stroke-width:1.4} -.dag .dn.p0{--pc:var(--p0)} .dag .dn.p1{--pc:var(--p1)} .dag .dn.p2{--pc:var(--p2)} -.dag .dn.p3{--pc:var(--p3)} .dag .dn.p4{--pc:var(--p4)} -.dag .dn text{fill:var(--ink);font-family:ui-monospace,monospace} -.dagwrap{overflow-x:auto;border:1px solid var(--line);border-radius:.5rem;padding:.4rem;background:var(--bg)} - -/* ---- trees / findings ---- */ -.tree,.tree ul{list-style:none;margin:0;padding-left:1.1rem} -.tree li{position:relative;padding:.1rem 0 .1rem .8rem} -.tree li::before{content:"";position:absolute;left:0;top:0;bottom:0;border-left:1px solid var(--line)} -.tree li::after{content:"";position:absolute;left:0;top:.85rem;width:.65rem;border-top:1px solid var(--line)} -.tree>li:last-child::before{bottom:auto;height:.85rem} -.tree .meta{font-size:.82rem;opacity:.7;margin-left:.4rem} -ul.findings{list-style:none;padding:0;margin:.5rem 0} -ul.findings li{border-left:3px solid var(--line);padding:.4rem 0 .4rem .8rem;margin:.5rem 0;max-width:80ch} -ul.findings li.sev-bad{border-left-color:var(--bad)} -ul.findings li.sev-warn{border-left-color:var(--warn)} -ul.findings li.sev-info{border-left-color:var(--info)} -ul.findings li.sev-ok{border-left-color:var(--ok)} -ul.findings b{display:block;margin-bottom:.15rem} -.digest{display:flex;flex-wrap:wrap;gap:.45rem;margin:.4rem 0 .2rem} -@media print{header.page,nav#toc,button,input.filter,.pop{display:none} - .layout{grid-template-columns:1fr;max-width:none} - section{break-inside:avoid;border:none;padding:0} - details{border:none}details:not([open])>*:not(summary){display:revert}} -""" - -JS = """ -function tgl(){const r=document.documentElement, - d=(r.dataset.theme||(matchMedia('(prefers-color-scheme: dark)').matches?'dark':'light'))==='dark'; - r.dataset.theme=d?'light':'dark';try{localStorage.htmlreport_theme=r.dataset.theme}catch(e){}} -function fs(d){const r=document.documentElement, - cur=parseFloat(getComputedStyle(r).getPropertyValue('--fs'))||17, - next=Math.min(24,Math.max(13,cur+d)); - r.style.setProperty('--fs',next+'px');try{localStorage.htmlreport_fs=next}catch(e){}} -try{if(localStorage.htmlreport_fs)document.documentElement.style.setProperty('--fs',localStorage.htmlreport_fs+'px'); - if(localStorage.htmlreport_theme)document.documentElement.dataset.theme=localStorage.htmlreport_theme}catch(e){} -(()=>{const t=document.getElementById('toc');if(!t)return; -document.querySelectorAll('main h2, main h3').forEach(h=>{ - const s=h.closest('section')||h;if(!h.id)h.id=(h.textContent||'').trim().toLowerCase() - .replace(/[^a-z0-9]+/g,'-').replace(/^-|-$/g,''); - const a=document.createElement('a');a.href='#'+(h.tagName==='H2'&&s.id?s.id:h.id); - a.textContent=h.textContent;if(h.tagName==='H3')a.className='h3';t.appendChild(a)})})(); -document.querySelectorAll('table').forEach(tb=>{ - tb.querySelectorAll('th:not(.nosort)').forEach((th,i)=>th.addEventListener('click',()=>{ - const dir=th.dataset.d=th.dataset.d==='a'?'d':'a'; - const val=td=>td&&td.dataset.v!==undefined?+td.dataset.v:(td?td.textContent.trim():''); - [...tb.tBodies[0].rows].sort((r1,r2)=>{const a=val(r1.cells[i]),b=val(r2.cells[i]); - const c=(typeof a=='number'&&typeof b=='number')?a-b:String(a).localeCompare(String(b)); - return dir==='a'?c:-c}).forEach(r=>tb.tBodies[0].appendChild(r))}))}); -function flt(inp,id){const q=inp.value.toLowerCase(); - document.querySelectorAll('#'+id+' tbody tr').forEach(r=> - r.style.display=r.textContent.toLowerCase().includes(q)?'':'none')} -(()=>{const U=[[31536e6,'y'],[2592e6,'mo'],[6048e5,'w'],[864e5,'d'],[36e5,'h'],[6e4,'m']]; -document.querySelectorAll('time.age').forEach(t=>{ - const d=new Date(t.dateTime);if(isNaN(d))return;const ms=Date.now()-d; - let s='just now';for(const[u,n]of U){if(Math.abs(ms)>=u){s=Math.floor(Math.abs(ms)/u)+n+(ms<0?' ahead':' ago');break}} - const pad=n=>String(n).padStart(2,'0'); - const iso=d.getFullYear()+'-'+pad(d.getMonth()+1)+'-'+pad(d.getDate())+' '+pad(d.getHours())+':'+pad(d.getMinutes()); - if(!t.textContent.trim())t.textContent=iso+' ('+s+')';else t.textContent+=' ('+s+')'; - t.title=d.toString(); - const budget=+(t.dataset.staleDays||0); - if(budget&&ms>budget*864e5)t.classList.add('stale')})})(); -(()=>{let cur=null,pinned=false,timer=null; -const kill=()=>{if(cur&&!pinned){cur.remove();cur=null}}; -const show=(host,html,title)=>{ - if(cur)cur.remove(); - const p=document.createElement('div');p.className='pop'; - p.innerHTML=''+(title?'

    '+title+'

    ':'')+html; - document.body.appendChild(p); - const r=host.getBoundingClientRect(),pr=p.getBoundingClientRect(); - let top=r.bottom+8; if(top+pr.height>innerHeight-8)top=Math.max(8,r.top-pr.height-8); - let left=Math.min(r.left,innerWidth-pr.width-8); - p.style.top=top+'px';p.style.left=Math.max(8,left)+'px'; - p.querySelector('.pin').onclick=()=>{pinned=false;kill()}; - p.onmouseenter=()=>clearTimeout(timer); - p.onmouseleave=()=>{timer=setTimeout(kill,220)}; - cur=p}; -const src=el=>{ - const t=el.querySelector(':scope > template.pop'); - if(t)return[t.innerHTML,el.dataset.popTitle||'']; - return null}; -document.querySelectorAll(':has(> template.pop)').forEach(el=>{ - const s=src(el);if(!s)return; - if(el.tabIndex<0)el.tabIndex=0; - const open=()=>{clearTimeout(timer);pinned=false;show(el,s[0],s[1])}; - el.addEventListener('mouseenter',()=>{clearTimeout(timer);timer=setTimeout(open,180)}); - el.addEventListener('mouseleave',()=>{clearTimeout(timer);timer=setTimeout(kill,220)}); - el.addEventListener('focus',open); - el.addEventListener('click',e=>{ - if(el.tagName==='A'&&(e.metaKey||e.ctrlKey))return; - e.preventDefault();if(!cur)open();pinned=!pinned}); -}); -addEventListener('keydown',e=>{if(e.key==='Escape'){pinned=false;kill()}});})(); -""" - - -# -------------------------------------------------------------------------- -# render -# -------------------------------------------------------------------------- -def render(facts: Facts, source: Path, generated: dt.datetime, schema_gap: dict[str, Any] | None = None) -> str: - schema_gap = schema_gap or {} - snaps = { - "now": facts.snapshot(generated), - "7d": facts.snapshot(generated - dt.timedelta(days=7)), - "14d": facts.snapshot(generated - dt.timedelta(days=14)), - } - series = facts.daily_series() - days_axis = [d for d, _, _ in series] - insights = compute_insights(facts, schema_gap, snaps) - sev_counts = Counter(i.sev for i in insights) - out: list[str] = [] - add = out.append - src = str(source) - - add('\n\n\n') - add('') - add(f"Beads backlog — state of the graph — {generated:%Y-%m-%d}") - add("") - add("\n") - add('

    Beads backlog — state of the graph

    ') - add(f'polylogue{facts.total:,} beads') - add(f'{generated:%Y-%m-%d}') - add('') - add('') - add('
    ') - add('
    ') - - # ---------------- summary ---------------- - add('

    Summary

    ') - add('
    ') - add(f'
    generated
    ') - add( - f'
    data as of
    — a live backlog; re-run to refresh
    ' - ) - add( - '
    basis
    measured every count, matrix, ' - 'tree, and histogram · derived every trend ' - "(reconstructed from timestamps; deletions and reopen history are invisible) and every " - "finding (each emitted by a condition checked against this data)
    " - ) - add(f'
    source
    {esc(src)}
    ') - add("
    regenerate
    devtools workspace beads-state-report --fresh --out <path>
    ") - add("
    ") - - tiles = [ - ( - f"{facts.total:,}", - delta_chip(snaps["now"]["total"], snaps["7d"]["total"]), - "beads, all time", - "wc -l .beads/issues.jsonl", - ), - ( - f"{facts.open_total:,}", - delta_chip(snaps["now"]["open"], snaps["7d"]["open"], up_is_bad=True), - "open + in progress", - "jq -r 'select(._type==\"issue\")|.status' .beads/issues.jsonl | grep -vc closed", - ), - ( - f"{facts.status['closed']:,}", - delta_chip(snaps["now"]["closed"], snaps["7d"]["closed"]), - "closed", - "jq -r 'select(._type==\"issue\")|.status' .beads/issues.jsonl | grep -c closed", - ), - ( - f"{len(facts.ready):,}", - delta_chip(snaps["now"]["ready"], snaps["7d"]["ready"]), - "ready (no open blocker)", - "bd ready --limit 5000 --json | jq length", - ), - ( - str(snaps["now"]["p0_open"]), - delta_chip(snaps["now"]["p0_open"], snaps["7d"]["p0_open"], up_is_bad=True), - "open P0", - "jq -r 'select(._type==\"issue\")|select(.status!=\"closed\")|.priority' .beads/issues.jsonl | grep -c '^0$'", - ), - ( - f"{len(facts.epics)}", - "", - "beads with children", - "jq -r '.dependencies[]?|select(.type==\"parent-child\")|.depends_on_id' " - ".beads/issues.jsonl | sort -u | wc -l", - ), - ] - add('
    ') - for value, delta, label, cmd in tiles: - add( - f'
    {value}{delta}' - f'{label} Δ7d' - f"{qa(cmd, 'measured ' + generated.strftime('%Y-%m-%d'))}
    " - ) - add("
    ") - add( - f"

    The whole bead population, open and closed, as one artifact: {facts.total:,} beads " - f"created over {facts.span_days} days ({facts.first_created} → {facts.last_created}), " - f"{facts.status['closed']:,} already closed. Deltas on the tiles are against the state " - "reconstructed 7 days ago. The findings below are the report's judgement layer — " - "every one is emitted by a condition this generator checked against this file, so a claim " - "that stops being true stops being printed.

    " - ) - add('
    ') - for sev in ("bad", "warn", "info", "ok"): - if sev_counts.get(sev): - add(f'{sev_counts[sev]} {sev}') - add("
    ") - add("
    ") - - # ---------------- pulse ---------------- - add('

    Pulse

    ') - add( - "

    Backlog state over its whole life, reconstructed per day from " - "created_at/closed_at and edge timestamps " - 'derived. ' - '■ open ' - '■ ready. ' - "The gap between the lines is the blocked share; the lines converging means the " - "graph is getting less dependency-bound.

    " - ) - add('
    ' + burnup_svg(series) + "
    ") - add('
    ') - add('') - add('') - for key, label, up_bad in ( - ("open", "open beads", True), - ("ready", "ready (no open blocker)", False), - ("blocked", "blocked by an open bead", True), - ("p0_open", "open P0", True), - ): - d = delta_chip(snaps["now"][key], snaps["7d"][key], up_is_bad=up_bad) - add( - f"" - f"" - f"" - ) - add("
    metric14d ago7d agonowΔ7d
    {label}{snaps['14d'][key]}{snaps['7d'][key]}{snaps['now'][key]}{d}
    ") - add( - "

    Reconstruction caveats, stated once: a deleted bead or edge leaves no trace in the " - "export, and a reopened bead's earlier closure is overwritten by its latest one. Both are " - "rare enough here that the curve's shape is trustworthy; individual day values are " - "±a few.

    " - ) - add("
    ") - - # ---------------- notation ---------------- - add('

    Bead notation

    ') - add( - "

    One bead renders as one chip, everywhere in this report. Left bar and " - "leading number are priority; then status, id, type glyph, and dependency " - "degree. Hover any chip for the expanded reading.

    " - ) - samples = [str(facts.rows[0]["id"])] - for wanted in ("epic", "bug", "feature"): - for r in facts.rows: - if r["issue_type"] == wanted and str(r["id"]) not in samples: - samples.append(str(r["id"])) - break - add("

    " + " ".join(chip(facts, s) for s in samples[:4]) + "

    ") - add('
    ') - add("
    P0–P4priority; also the left bar colour
    ") - for glyph, name in STATUS_GLYPH.values(): - add(f"
    {glyph}{name}
    ") - for glyph, name in TYPE_GLYPH.values(): - add(f"
    {glyph}{name}
    ") - add("
    ↑nn open beads block this one
    ") - add("
    ↓nthis one blocks n open beads
    ") - add("
    struck idclosed
    ") - add("
    blue outlinein progress
    ") - add("
    dashed borderdeferred
    ") - add("
    ") - add( - '

    reading it A chip with a red left bar, a hollow ' - "circle, and ↓9 is an open P0 that nine other open beads are " - "waiting on — the highest-leverage shape in the graph.

    " - ) - add("
    ") - - # ---------------- shape ---------------- - add('

    Shape

    ') - add( - "

    The status × priority matrix, plus per-tier closure rate and median lead " - "time — the fastest read on whether the priority scale is load-bearing. The " - "computed verdict is in Findings.

    " - ) - add("

    Status × priority

    ") - add('
    ') - status_cols = facts.statuses_present - head = "".join(f"" for s in status_cols) - add( - f"{head}" - ) - peak = max(facts.matrix.values()) or 1 - for p in range(5): - row = [f''] - for st in status_cols: - n = facts.matrix.get((st, p), 0) - row.append(f'') - total_p = facts.priority[p] - med = _median(facts.lead_by_priority[p]) - row.append(f'') - row.append(f'') - row.append(f'') - add("".join(row) + "") - add("
    {esc(STATUS_GLYPH.get(s, ('?', s))[1])}
    prioritytotalclosed %median lead (days)
    P{p}{n:,}{total_p:,}{facts.rate(p):.0%}{med:.2f}
    ") - - add("

    Type

    ") - add('
    ') - add("") - peak_t = max(facts.itype.values()) or 1 - for tname, n in facts.itype.most_common(): - closed = sum(1 for r in facts.rows if r["issue_type"] == tname and r["status"] == "closed") - glyph = TYPE_GLYPH.get(tname, ("?", tname))[0] - add( - f"" - f'" - f'' - f'' - ) - add("
    typebeadsshareclosed %
    {glyph} {esc(tname)}' - f"{n:,}{100 * n / facts.total:.1f}%{closed / n:.0%}
    ") - add("
    ") - - # ---------------- structure ---------------- - add('

    Structure

    ') - add( - "

    Parents ranked by child count, from parent-child edges (the dotted " - "id is not the parent link — the two disagree for " - f"{len(facts.id_mismatch)} beads and {len(facts.id_orphan)} dotted beads have no edge " - "at all; both are counted in Health). The trend column is each epic's open-children " - 'count over the backlog\'s life derived — ' - "a falling line is a draining program, a flat high line is a parked one, a rising " - "line is a program still being decomposed.

    " - ) - add('
    ') - add('') - for e in facts.epics[:24]: - eid = str(e["id"]) - spark = sparkline_svg(facts.epic_open_series(eid, days_axis), label=f"{eid} open children") - add( - f"" - f"" - f'" - f"" - f'' - f'' - ) - add("
    parenttitleprogresstrendchildrenopen
    {chip(facts, eid)}{esc(str(e['title'])[:74])}' - f"{fill_bar(int(e['open']), int(e['n']))}{spark}{e["n"]}{e["open"]}
    ") - - # the two computed extremes: most-finished and least-started large parents - big = [e for e in facts.epics if int(e["n"]) >= 8] - if big: - most_done = min(big, key=lambda e: int(e["open"]) / int(e["n"])) - least_started = max(big, key=lambda e: int(e["open"]) / int(e["n"])) - add("

    Trees: the two extremes

    ") - add( - f"

    Computed from the table above: the most-finished large parent " - f"({chip(facts, str(most_done['id']))}, {int(most_done['n']) - int(most_done['open'])} of " - f"{most_done['n']} children closed) against the least-started " - f"({chip(facts, str(least_started['id']))}, {least_started['open']} of " - f"{least_started['n']} still open). A flat backlog list renders those identically; " - "the fill bars are why this section exists.

    " - ) - for e in (most_done, least_started): - eid = str(e["id"]) - if eid not in facts.issues: - continue - rec = facts.issues[eid] - add( - f"
    {chip(facts, eid)}  {esc(str(rec['title'])[:80])}" - + tree_html(facts, eid) - + "
    " - ) - add("
    All other parents with 8+ children") - skip = {str(most_done["id"]), str(least_started["id"])} - for e in big: - if str(e["id"]) in skip: - continue - add( - f"
    {chip(facts, str(e['id']))}  {esc(str(e['title'])[:80])} " - f" {fill_bar(int(e['open']), int(e['n']))}" - + tree_html(facts, str(e["id"])) - + "
    " - ) - add("
    ") - add("
    ") - - # ---------------- topology ---------------- - add('

    Topology

    ') - ready_pct = f"{len(facts.ready) / facts.open_total:.0%}" if facts.open_total else "n/a" - add( - f"

    The blocks graph: {facts.dep_types['blocks']:,} edges over " - f"{facts.open_total:,} open beads, of which {len(facts.ready):,} are ready " - f"({ready_pct}) and {len(facts.blocked)} wait on an open blocker.

    " - ) - add('
    ') - for value, label, cmd in [ - ( - f"{facts.dep_types['blocks']:,}", - "blocks edges", - "jq -r '.dependencies[]?|.type' .beads/issues.jsonl | sort | uniq -c", - ), - (f"{len(facts.ready):,}", "ready", "bd ready --limit 5000 --json | jq length"), - (f"{len(facts.blocked)}", "blocked by an open bead", "bd blocked --json | jq length"), - (f"{len(facts.cycles)}", "cycles in blocks", "DFS colouring over the blocks graph (this generator)"), - ]: - add( - f'
    {value}{label}' - f"{qa(cmd, 'measured')}
    " - ) - add("
    ") - - frontier = facts.parallel_frontier() - if len(frontier) >= 2: - add("

    The parallel frontier

    ") - add( - "

    Computed: the largest programs whose open subtrees occupy disjoint " - "components of the blocks graph — no dependency path of any " - "length connects them, so they can run as parallel lanes with zero coordination " - "on ordering. Greedy by open-descendant count.

    " - ) - add('
    ') - add("") - for c in frontier: - add( - f"" - f'' - f'' - ) - add("
    programtitleopen descendantsopen P0–P2
    {chip(facts, str(c['id']))}{esc(str(c['title'])[:70])}{c["open_desc"]}{c["urgent"]}
    ") - - add("

    Relation vocabulary

    ") - add('
    ') - add("") - first_last: dict[str, tuple[str, str]] = {} - for _, _, kind, created in facts.edges: - when = created[:10] - if not when: - continue - lo, hi = first_last.get(kind, (when, when)) - first_last[kind] = (min(lo, when), max(hi, when)) - peak_r = max(facts.dep_types.values()) or 1 - for kind, n in facts.dep_types.most_common(): - lo, hi = first_last.get(kind, ("", "")) - flag = "" - if kind in ("relates-to", "related"): - flag = { - "split": ' split vocab', - "recurring": ' unified, then recurred', - "clean": ' unified', - }[facts.vocab_state] - add( - f"" - f'' - f"" - ) - add("
    relationedgesfirst writtenlast written
    {esc(kind)}{flag}{n:,}{esc(lo)}{esc(hi)}
    ") - if facts.retyped_spike: - add( - f"

    A bulk re-type is visible in the timestamps: {facts.retyped_spike:,} " - f"relates-to edges carry {esc(facts.retyped_spike_day)} as " - "their created_at — converted edges lose their original assertion " - "dates, so any “when were associations asserted?” analysis must treat that " - 'date as an artifact spike, not a real burst derived.

    ' - ) - if facts.related_recurrence: - add( - "

    Post-repair recurrence, exactly: the edge(s) below were written after " - "the newest re-typed edge, proving the write path is still unconstrained.

    " - ) - add('
    ') - add("") - for src_id, dst_id, ts in facts.related_recurrence: - add(f"") - add("
    fromrelates tocreated_at
    {chip(facts, src_id)}{chip(facts, dst_id)}{esc(ts)}
    ") - - add("

    Highest-leverage blockers

    ") - add( - "

    Each row releases a fan-out no other bead does; if the graph is " - "dependency-bound these are the lever, and if it is throughput-bound they are " - "merely the tidiest finish-first candidates.

    " - ) - add('
    ') - add("") - for n, bid in facts.top_blockers[:20]: - add( - f"" - f'' - ) - add("
    beadtitleblocks (open)
    {chip(facts, bid)}{esc(str(facts.issues[bid]['title'])[:80])}{n}
    ") - - cluster = facts.densest_cluster() - core = facts.cluster_core(cluster) - if core: - add("

    Densest blocking cluster

    ") - add( - f"

    Rendering all {facts.total:,} beads as a graph produces a hairball, so this is " - f"a slice: the largest connected component of the open-only blocks graph " - f"holds {len(cluster)} beads, and this is its {len(core)}-node core " - "(members with ≥3 neighbours inside it; the dropped nodes are single-edge leaves). " - "Arrows point blocker → blocked; columns are longest-path depth, so leftmost is " - "furthest upstream. Hover a node for its title.

    " - ) - add('
    ' + cluster_svg(facts, core) + "
    ") - add("
    ") - - # ---------------- time ---------------- - add('

    Time

    ') - peak_day, peak_n = max(facts.created_day.items(), key=lambda kv: kv[1]) - add( - f"

    Filing is spiky ({peak_n} beads on {esc(peak_day)}, the busiest day, against a " - f"median of {_median(list(facts.created_day.values())):.0f}/day) — the signature " - "of audit-lane batch filing rather than continuous discovery. Closure runs at a median " - f"of {_median(list(facts.closed_day.values())):.0f}/day.

    " - ) - add('
    ') - add( - histogram_svg( - facts.days, - [("created", facts.created_day, "created"), ("closed", facts.closed_day, "closed")], - ) - ) - add( - f"
    beads per day, {facts.first_created} → {facts.last_created} " - f'· ■ created ' - f'■ closed · ' - f'measured
    ' - ) - add("
    ") - - (rc, rx), (pc, px) = facts.win_recent, facts.win_prior - add('
    ') - add('') - for name, created_n, closed_n in ( - (f"prior {VELOCITY_WINDOW_DAYS}d", pc, px), - (f"last {VELOCITY_WINDOW_DAYS}d", rc, rx), - ): - cls = "bad" if created_n - closed_n > 0 else "ok" - add( - f'' - f'' - ) - add("
    windowcreatedclosednet
    {name}{created_n}{closed_n}{created_n - closed_n:+d}
    ") - - add("

    Age × priority of open work

    ") - add( - "

    Where open work sits in the age-urgency plane. Mass in the bottom-left " - "(old, urgent) is the shelf to triage first; mass in the bottom-right (old, P3/P4) " - "is parking, which is fine as long as it was chosen.

    " - ) - add(heatmap_html(facts)) - add( - f"

    Median open bead is {_median(facts.age_open):.0f} days old; the oldest is " - f"{max(facts.age_open):.0f} days — against a tracker that is itself only " - f"{facts.span_days} days old.

    " - ) - add("
    ") - - # ---------------- health ---------------- - add('

    Health

    ') - add( - f"

    {len(HEALTH_CHECKS)} mechanical checks over the graph. These are a " - "review queue, not a batch-fix list — especially the " - "“open parent, all children closed” row, where children closing means the " - "parent is ready for adjudication against its own acceptance criteria, never that it " - "is done.

    " - ) - add('
    ') - add("") - for cls, name, attr, meaning in HEALTH_CHECKS: - n = len(getattr(facts, attr)) - badge = f'{n}' - add(f"") - add("
    checkcountmeaning
    {name}{badge}{meaning}
    ") - - if facts.dangling: - add("

    Dangling references

    ") - add( - "

    A genuine data bug, not a judgement call: these edges point at ids that " - "resolve to nothing, so the blocking intent behind them is silently absent from " - "every query.

    " - ) - add('
    ') - add("") - for a, b, kind in facts.dangling: - add( - f"" - f'' - ) - add("
    frompoints atrelation
    {chip(facts, a)}{esc(b)}' - f'{esc(kind)}
    ") - - if facts.stale_claims: - add("

    Stale in-progress claims

    ") - add('
    ') - add("") - for bid, days_stale in facts.stale_claims: - add( - f"" - f'' - ) - add("
    beadtitledays since update
    {chip(facts, bid)}{esc(str(facts.issues[bid]['title'])[:76])}{days_stale:.0f}
    ") - - add("

    Review queue: open parents whose children are all closed

    ") - add('
    ') - add('') - for pid in facts.open_parent_all_closed: - rec = facts.issues[pid] - has_ac = bool(str(rec.get("acceptance_criteria") or "").strip()) - flag = 'yes — read it' if has_ac else 'no' - add( - f"" - f'' - f"" - ) - add("
    parenttitlechildrenhas own AC
    {chip(facts, pid)}{esc(str(rec['title'])[:76])}{len(facts.children[pid])}{flag}
    ") - - if facts.dup_titles: - add("

    Duplicate titles

    ") - add('
    ') - add('') - for title, ids in facts.dup_titles: - add(f"") - add("
    titlebeads
    {esc(title[:86])}" + " ".join(chip(facts, i) for i in ids) + "
    ") - - # schema gap: a live archive/code divergence fact, rendered when measurable - if schema_gap.get("available"): - add("

    Archive/code schema gap

    ") - add( - f"

    The live archive is at index-schema v{schema_gap['live_version']}; " - f"origin/master declares v{schema_gap['declared']}. " - f"{len(schema_gap['blockers'])} intervening version(s) are " - "SEMANTIC_REPARSE — no clone-safe SQL fast-forward exists, so every " - "merged fix at those versions is inert until " - "polylogue ops reset --index && polylogued run. A closed bead " - "here can still describe a live bug against the data an operator queries right now.

    " - ) - add('
    ') - for value, label, cmd in [ - ( - f"v{schema_gap['live_version']}", - "live archive schema", - f"sqlite3 -readonly {schema_gap.get('live_path', '/index.db')} 'PRAGMA user_version;'", - ), - ( - f"v{schema_gap['declared']}", - "origin/master declares", - "git show origin/master:polylogue/storage/sqlite/archive_tiers/index.py | grep INDEX_SCHEMA_VERSION", - ), - ( - str(len(schema_gap["blockers"])), - "SEMANTIC_REPARSE versions blocking fast-forward", - "git show origin/master:polylogue/storage/sqlite/lifecycle.py", - ), - ]: - add( - f'
    {esc(value)}{esc(label)}' - f"{qa(cmd, 'measured ' + generated.strftime('%Y-%m-%d'))}
    " - ) - add("
    ") - if schema_gap["blockers"]: - add('
    ') - add('') - for v in range(schema_gap["live_version"] + 1, schema_gap["declared"] + 1): - classes = schema_gap["blocker_notes"].get(v, ()) - cls_badges = ( - " ".join( - f'{esc(c)}' - for c in classes - ) - or 'undeclared' - ) - add(f"") - add("
    versiondelta class(es)
    v{v}{cls_badges}
    ") - else: - add( - '

    schema gap not measured Needs both ' - "git show origin/master:… and a readable live archive at the " - "operator's configured archive root; at least one was unavailable, so the tiles are " - "omitted rather than guessed.

    " - ) - add("
    ") - - # ---------------- themes ---------------- - add('

    Themes

    ') - add( - f"

    Two independent readings of where open work concentrates, because neither alone is " - f"trustworthy: the label view folds the repository's own area:* labels " - f"through an explicit synonym map (covers {facts.labelled:,} of {facts.total:,} beads and " - f"says nothing about the rest), and the keyword view classifies every bead from title and " - "description (full coverage, will misfile loose titles). Where the two agree, the " - "concentration is real.

    " - ) - add('
    ') - add( - "" - "" - ) - names = sorted(set(facts.by_label_theme) | set(facts.by_kw_theme)) - peak_o = ( - max( - (facts.by_label_theme[n]["open"] + facts.by_label_theme[n]["in_progress"] for n in names), - default=1, - ) - or 1 - ) - for name in names: - lab = facts.by_label_theme.get(name, Counter()) - kw = facts.by_kw_theme.get(name, Counter()) - lab_open = lab["open"] + lab["in_progress"] - kw_open = kw["open"] + kw["in_progress"] - add( - f"" - f'' - f'' - f'' - f'' - ) - add("
    subsystemopen (labels)closed (labels)open (keywords)closed (keywords)
    {esc(name)}{lab_open}{lab["closed"]}{kw_open}{kw["closed"]}
    ") - add("
    The synonym map used to fold area:* labels") - add('
    ') - add('') - for canon, suffixes in SUBSYSTEMS.items(): - add(f"") - add("
    subsystemfolded area labels
    {esc(canon)}" + " ".join(f"area:{esc(s)}" for s in suffixes) + "
    ") - add( - "

    This map is a proposal written by hand, not something the repository declares. " - "Disagreeing with a row changes the table above it.

    " - ) - add("
    ") - add("
    Raw area:* labels, top 30") - add('
    ') - add("") - area_counts = [(x, facts.labels[x]) for x in facts.area_labels] - area_counts.sort(key=lambda kv: (-kv[1], kv[0])) - peak_l = area_counts[0][1] if area_counts else 1 - for name, n in area_counts[:30]: - add( - f"" - f'' - ) - add("
    labelbeads
    {esc(name)}{n}
    ") - add("
    ") - - # ---------------- dated review passes ---------------- - strict_total = sum(facts.verdict_counts.values()) - loose_total = sum(facts.loose_counts.values()) - if strict_total or facts.reconciliation: - add('

    Dated review passes

    ') - add( - "

    Two hand-verification passes left machine-parseable markers in bead notes; this " - "generator extracts them mechanically. They are dated evidence — " - "ground truth as of the day each pass ran, ageing at the speed the code moves, not " - "recomputed facts.

    " - ) - if strict_total: - add('

    The verified subset pass of 2026-07-31

    ') - add( - f"

    {strict_total} beads carry a machine-parseable " - "VERIFICATION (…): STALE|PARTIAL|LIVE verdict written by a human " - f"reviewer — {facts.verdict_counts['LIVE']} LIVE / " - f"{facts.verdict_counts['PARTIAL']} PARTIAL / {facts.verdict_counts['STALE']} STALE. " - "It is the only subset of this backlog where “is this claim still true?” " - "was answered by inspection rather than assumed. Preserve it as a labelled " - "evaluation set.

    " - ) - add( - f"

    Extraction gap, stated rather than papered over: the operator's own count of " - f"the pass is {OPERATOR_VERIFIED_COUNT} beads; the strict pattern finds " - f"{strict_total}, and relaxing to any bare verdict token finds {loose_total} " - f"({facts.loose_counts['LIVE']}/{facts.loose_counts['PARTIAL']}/" - f"{facts.loose_counts['STALE']}), which brackets the figure but includes incidental " - "prose. The difference is verdicts phrased outside the parseable form — a " - "review this valuable should write its verdict one way, every time.

    " - ) - add('') - add('
    ') - add("") - badge_of = {"LIVE": "ok", "PARTIAL": "warn", "STALE": "bad"} - for bid, verdict, _status, sweep in sorted(facts.verdicts, key=lambda v: (v[1], v[0])): - add( - f"" - f'' - f"" - f"" - ) - add("
    beadverdictsweeptitle
    {chip(facts, bid)}{verdict}{esc(sweep[:34])}{esc(str(facts.issues[bid]['title'])[:70])}
    ") - - if facts.reconciliation: - add('

    P0 reconciliation pass of 2026-07-31

    ') - add( - f"

    A hand-verified pass over the P0 shelf, read against origin/master " - "and the live archive. Its sharpest product is a split a flat status field cannot " - "express: “open” work with no code fix yet versus work that is " - "done and merged but inert behind the archive/code schema gap (see Health). " - f"Marker coverage: {facts.reconciliation_anchored} beads carry the anchor; the " - f"pass's own summary covered {RECONCILIATION_EXPECTED}.

    " - ) - add('
    ') - for value, label in [ - (str(facts.reconciliation_counts["FIXED-AND-EFFECTIVE"]), "FIXED-AND-EFFECTIVE — closed"), - (str(facts.reconciliation_counts["FIXED-PENDING-REBUILD"]), "FIXED-PENDING-REBUILD"), - (str(facts.reconciliation_counts["FIXED-PENDING-DEPLOY"]), "FIXED-PENDING-DEPLOY"), - (str(facts.reconciliation_counts["MISFRAMED"]), "MISFRAMED — demoted"), - (str(facts.reconciliation_counts["GENUINELY OPEN"]), "GENUINELY OPEN"), - ]: - add(f'
    {value}{label}
    ') - add("
    ") - rebuild_rows = [ - (bid, verdict, snippet) - for bid, verdict, snippet in facts.reconciliation - if verdict in ("FIXED-PENDING-REBUILD", "FIXED-PENDING-DEPLOY") - ] - if rebuild_rows: - add( - "

    The rebuild-blocked set: merged, confirmed inert by direct query, needing no " - "further engineering — the honest measure of how much apparent open work is " - "actually done.

    " - ) - add('
    ') - add('') - for bid, verdict, snippet in sorted(rebuild_rows, key=lambda t: (t[1], t[0])): - title = esc(str(facts.issues[bid]["title"])[:70]) if bid in facts.issues else "" - add( - f"" - f'' - f"" - ) - add("
    beadtitleverdictevidence
    {chip(facts, bid)}{title}{esc(verdict)}{esc(snippet[-220:])}
    ") - add("
    Full reconciliation table") - add('') - add('
    ') - add("") - rc_badge = { - "FIXED-AND-EFFECTIVE": "ok", - "FIXED-PENDING-REBUILD": "warn", - "FIXED-PENDING-DEPLOY": "warn", - "MISFRAMED": "info", - "GENUINELY OPEN": "bad", - } - for bid, verdict, _snippet in sorted(facts.reconciliation, key=lambda t: (t[1], t[0])): - title = esc(str(facts.issues[bid]["title"])[:74]) if bid in facts.issues else "" - add( - f"" - f'' - ) - add("
    beadtitleverdict
    {chip(facts, bid)}{title}{esc(verdict)}
    ") - if strict_total or facts.reconciliation: - add("
    ") - - # ---------------- findings ---------------- - add('

    Findings

    ') - add( - "

    The judgement layer, ordered by severity. Every entry was emitted by a condition " - "this generator checked against the input file at render time — there is no " - "hand-authored claim here that can silently outlive its evidence. " - 'measured marks direct counts; ' - 'derived marks reconstructions and computed ' - "verdicts.

    " - ) - add('
      ') - for ins in insights: - chips = (" " + " ".join(chip(facts, c) for c in ins.chips)) if ins.chips else "" - add( - f'
    • {ins.sev} ' - f'{ins.title} {ins.ev}{ins.body}{chips}
    • ' - ) - add("
    ") - add("
    ") - - # ---------------- method ---------------- - add('

    Method & regeneration

    ') - add( - "

    Everything numeric on this page is computed by the generator named below from a " - "single input file; nothing was transcribed. Interpretation is computed too: findings " - "are conditional generators, not prose — if a number here disagrees with " - "bd, the export is stale, not the arithmetic; regenerate with " - "--fresh.

    " - ) - add("

    Provenance

    ") - add('
    ') - add('') - for part, origin in [ - ("every count, percentage, matrix cell, median, heatmap cell", "measured from the JSONL"), - ("epic trees, fill bars, dependency degree on each chip", "measured (parent-child + blocks edges)"), - ("per-day histogram, velocity windows", "measured"), - ( - "pulse snapshots, burnup, epic sparklines, 7d deltas", - "derived (timestamp reconstruction; deletions invisible)", - ), - ( - "ready / blocked / cycles / densest cluster / parallel frontier", - "measured (graph traversal in the generator)", - ), - ("health checks and review queues", "measured"), - ("keyword theme classification", "measured, using a hand-written keyword list"), - ("area-label folding", "measured, using a hand-written synonym map"), - ("verdict + reconciliation marker extraction", "measured by regex over notes and comments"), - ("schema gap (live vs origin/master)", "measured: git show + read-only PRAGMA"), - ("findings", "derived: conditional generators over all of the above"), - ("section framing sentences and the notation legend", "authored, deliberately data-free"), - ( - f"two constants from dated passes ({OPERATOR_VERIFIED_COUNT}, {RECONCILIATION_EXPECTED})", - "operator-supplied, provenance attached where used", - ), - ]: - cls = "todo" if origin.startswith(("authored", "operator")) else "info" - add(f'') - add("
    part of the reportorigin
    {esc(part)}{esc(origin)}
    ") - add("

    Regenerate

    ") - add( - "
    cd /realm/project/polylogue\n"
    -        "devtools workspace beads-state-report --fresh \\\n"
    -        f"  --out {esc(str(source.parent))}/beads-state.html
    " - ) - add( - "

    --fresh re-runs bd export -o .beads/issues.jsonl " - "first. Without it the report describes whatever the file last held, which after " - "any uncommitted bd mutation is not the live state. The exported " - "file is contended across sessions and should not be committed as part of " - "generating a report.

    " - ) - add("
    ") - - add("
    ") - add( - f"
    generated by devtools workspace beads-state-report on " - f"{generated:%Y-%m-%d %H:%M} from {esc(src)} · {facts.total:,} beads, " - f"{len(facts.edges):,} dependency edges
    " - ) - add("") - add("\n") - return "\n".join(out) - - -# -------------------------------------------------------------------------- -# entry point -# -------------------------------------------------------------------------- -def json_payload(facts: Facts, schema_gap: dict[str, Any], generated: dt.datetime) -> dict[str, Any]: - snaps = { - "now": facts.snapshot(generated), - "7d": facts.snapshot(generated - dt.timedelta(days=7)), - "14d": facts.snapshot(generated - dt.timedelta(days=14)), - } - insights = compute_insights(facts, schema_gap, snaps) - return { - "generated": generated.isoformat(), - "total": facts.total, - "status": dict(facts.status), - "priority": {str(k): v for k, v in facts.priority.items()}, - "open_total": facts.open_total, - "ready": len(facts.ready), - "blocked": len(facts.blocked), - "cycles": facts.cycles, - "dangling": [list(d) for d in facts.dangling], - "open_parent_all_closed": facts.open_parent_all_closed, - "stale_claims": [[bid, round(days, 2)] for bid, days in facts.stale_claims], - "aged_urgent": facts.aged_urgent, - "dup_titles": [[t, ids] for t, ids in facts.dup_titles], - "snapshots": snaps, - "velocity": { - "recent": {"created": facts.win_recent[0], "closed": facts.win_recent[1]}, - "prior": {"created": facts.win_prior[0], "closed": facts.win_prior[1]}, - "window_days": VELOCITY_WINDOW_DAYS, - }, - "parallel_frontier": [ - {"id": c["id"], "open_desc": c["open_desc"], "urgent": c["urgent"]} for c in facts.parallel_frontier() - ], - "verdicts": {"strict": dict(facts.verdict_counts), "loose": dict(facts.loose_counts)}, - "reconciliation": facts.reconciliation, - "vocab_state": facts.vocab_state, - "schema_gap": schema_gap, - "insights": [{"sev": i.sev, "title": i.title, "body": i.body, "ev": i.ev} for i in insights], - } - - -def main(argv: list[str] | None = None) -> int: - root = _get_root() - parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("path", nargs="?", default=None, help="issues.jsonl (default: .beads/issues.jsonl)") - parser.add_argument("--out", default=None, help="write HTML here (default: .local/beads-state.html)") - parser.add_argument("--fresh", action="store_true", help="run `bd export` before reading") - parser.add_argument("--json", action="store_true", help="print the computed facts as JSON instead of HTML") - args = parser.parse_args(argv) - - path = Path(args.path) if args.path else root / ".beads/issues.jsonl" - if args.fresh: - subprocess.run(["bd", "export", "-o", str(path)], check=True, capture_output=True) - if not path.exists(): - print(f"no such file: {path}", file=sys.stderr) - return 1 - - issues, edges = load(path) - if not issues: - print(f"no issues parsed from {path}", file=sys.stderr) - return 1 - now = dt.datetime.now(dt.UTC) - facts = Facts(issues, edges, now) - schema_gap = schema_gap_facts(root) - - if args.json: - print(json.dumps(json_payload(facts, schema_gap, now), indent=2, sort_keys=True)) - return 0 - - out = Path(args.out) if args.out else root / ".local/beads-state.html" - out.parent.mkdir(parents=True, exist_ok=True) - out.write_text(render(facts, path, now, schema_gap), encoding="utf-8") - print(f"wrote {out} ({out.stat().st_size:,} bytes) from {facts.total:,} beads") - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/benchmark_campaign.py b/devtools/benchmark_campaign.py deleted file mode 100644 index 9fbd2ec4d0..0000000000 --- a/devtools/benchmark_campaign.py +++ /dev/null @@ -1,415 +0,0 @@ -"""Run and compare benchmark campaigns with durable artifacts.""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import tempfile -import time -from collections.abc import Iterable, Mapping -from dataclasses import asdict, dataclass, field -from datetime import UTC, datetime -from pathlib import Path -from typing import TypedDict - -from devtools import repo_root as _get_root -from polylogue.core.json import JSONDocument, json_document -from polylogue.scenarios import ExecutionKind, ScenarioMetadata, pytest_execution, run_execution - -from .authored_scenario_catalog import get_authored_scenario_catalog -from .benchmark_catalog import BenchmarkCampaignEntry -from .benchmark_results import ( - BenchmarkStat, - BenchmarkStatRecord, - benchmark_stat_record, - parse_pytest_benchmark_stats, -) - -ROOT = _get_root() -ARTIFACT_DIR = Path(".local/benchmark-campaigns") -STATUS_IGNORE_PREFIXES = (f"{ARTIFACT_DIR.as_posix()}/",) -DEFAULT_WARN_PCT = 10.0 -DEFAULT_FAIL_PCT = 20.0 -CAMPAIGNS = get_authored_scenario_catalog().benchmark_campaign_index() - - -class RegressionRecord(TypedDict): - fullname: str - baseline_mean: float - current_mean: float - delta_pct: float - - -@dataclass(frozen=True) -class Regression: - fullname: str - baseline_mean: float - current_mean: float - delta_pct: float - - -@dataclass(frozen=True) -class CampaignResult: - campaign: str - description: str - commit: str - worktree_dirty: bool - created_at: str - workspace: str - command: list[str] - tests: list[str] - notes: list[str] - benchmark_count: int - runtime_seconds: float - exit_code: int - machine_info: JSONDocument - benchmarks: list[BenchmarkStatRecord] - slowest: list[BenchmarkStatRecord] - compare_to: str | None - warn_pct: float - fail_pct: float - regressions: list[RegressionRecord] - worst_regression_pct: float | None - origin: str = "authored" - path_targets: list[str] = field(default_factory=list) - artifact_targets: list[str] = field(default_factory=list) - operation_targets: list[str] = field(default_factory=list) - maintenance_targets: list[str] = field(default_factory=list) - tags: list[str] = field(default_factory=list) - - -def _git_output(*args: str) -> str: - return subprocess.check_output(["git", *args], cwd=ROOT, text=True).strip() - - -def _worktree_dirty() -> bool: - status = subprocess.check_output(["git", "status", "--short"], cwd=ROOT, text=True) - for line in status.splitlines(): - path = line[3:] - if any(path.startswith(prefix) for prefix in STATUS_IGNORE_PREFIXES): - continue - if path: - return True - return False - - -def _default_artifact_path(campaign: str, suffix: str) -> Path: - date = datetime.now(UTC).strftime("%Y-%m-%d") - return ROOT / ARTIFACT_DIR / f"{date}-{campaign}.{suffix}" - - -def _number_value(value: object, *, field: str) -> str | int | float: - if isinstance(value, bool) or not isinstance(value, (str, int, float)): - raise ValueError(f"Benchmark payload field {field!r} is not numeric: {value!r}") - return value - - -def _regression_record(regression: Regression) -> RegressionRecord: - return { - "fullname": regression.fullname, - "baseline_mean": regression.baseline_mean, - "current_mean": regression.current_mean, - "delta_pct": regression.delta_pct, - } - - -def _load_campaign_result(path: Path) -> CampaignResult: - return CampaignResult(**json.loads(path.read_text())) - - -def _compare_results(current: list[BenchmarkStat], baseline: Iterable[Mapping[str, object]]) -> list[Regression]: - baseline_map = {str(item["fullname"]): item for item in baseline} - regressions: list[Regression] = [] - for bench in current: - previous = baseline_map.get(bench.fullname) - if previous is None: - continue - baseline_mean = float(_number_value(previous["mean"], field="mean")) - if baseline_mean <= 0: - continue - delta_pct = ((bench.mean - baseline_mean) / baseline_mean) * 100.0 - regressions.append( - Regression( - fullname=bench.fullname, - baseline_mean=baseline_mean, - current_mean=bench.mean, - delta_pct=delta_pct, - ) - ) - regressions.sort(key=lambda item: item.delta_pct, reverse=True) - return regressions - - -def _render_markdown(result: CampaignResult) -> str: - lines = [ - f"# Benchmark Campaign: {result.campaign}", - "", - f"- Description: {result.description}", - f"- Commit: `{result.commit}`", - f"- Worktree dirty: {'yes' if result.worktree_dirty else 'no'}", - f"- Created: `{result.created_at}`", - f"- Runtime: `{result.runtime_seconds:.2f}s`", - f"- Command: `{' '.join(result.command)}`", - f"- Tests: `{', '.join(result.tests)}`", - f"- Benchmarks: `{result.benchmark_count}`", - f"- Warn threshold: `{result.warn_pct:.1f}%`", - f"- Fail threshold: `{result.fail_pct:.1f}%`", - "", - ] - if ( - result.path_targets - or result.artifact_targets - or result.operation_targets - or result.maintenance_targets - or result.tags - ): - lines.extend(["## Scenario Metadata", ""]) - lines.append(f"- Origin: `{result.origin}`") - if result.path_targets: - lines.append(f"- Path targets: `{', '.join(result.path_targets)}`") - if result.artifact_targets: - lines.append(f"- Artifact targets: `{', '.join(result.artifact_targets)}`") - if result.operation_targets: - lines.append(f"- Operation targets: `{', '.join(result.operation_targets)}`") - if result.maintenance_targets: - lines.append(f"- Maintenance targets: `{', '.join(result.maintenance_targets)}`") - if result.tags: - lines.append(f"- Tags: `{', '.join(result.tags)}`") - lines.append("") - lines.extend( - [ - "## Slowest Benchmarks", - "", - "| Benchmark | Mean (s) | Median (s) | Ops/s | Rounds |", - "| --- | ---: | ---: | ---: | ---: |", - ] - ) - for bench in result.slowest: - ops = "-" if bench["ops"] is None else f"{bench['ops']:.2f}" - lines.append( - f"| `{bench['fullname']}` | {bench['mean']:.6f} | {bench['median']:.6f} | {ops} | {bench['rounds']} |" - ) - if result.regressions: - lines.extend( - [ - "", - "## Largest Regressions vs Baseline", - "", - "| Benchmark | Delta % | Baseline Mean (s) | Current Mean (s) |", - "| --- | ---: | ---: | ---: |", - ] - ) - for regression in result.regressions[:10]: - lines.append( - f"| `{regression['fullname']}` | {regression['delta_pct']:.2f}% | {regression['baseline_mean']:.6f} | {regression['current_mean']:.6f} |" - ) - if result.notes: - lines.extend(["", "## Notes", ""]) - lines.extend([f"- {note}" for note in result.notes]) - return "\n".join(lines) + "\n" - - -def run_campaign( - campaign: BenchmarkCampaignEntry, - *, - json_out: Path | None, - markdown_out: Path | None, - compare_to: Path | None, - warn_pct: float | None, - fail_pct: float | None, -) -> CampaignResult: - if campaign.execution is None or campaign.execution.kind is not ExecutionKind.PYTEST: - raise ValueError(f"Benchmark campaign {campaign.name!r} must use pytest execution") - - artifact_json = json_out or _default_artifact_path(campaign.name, "json") - artifact_md = markdown_out or _default_artifact_path(campaign.name, "md") - artifact_json.parent.mkdir(parents=True, exist_ok=True) - artifact_md.parent.mkdir(parents=True, exist_ok=True) - - with tempfile.TemporaryDirectory(prefix=f"benchmark-{campaign.name}-") as tmpdir: - raw_json = Path(tmpdir) / "pytest-benchmark.json" - benchmark_execution = pytest_execution( - "-q", - "--override-ini=addopts=-ra", - "-n", - "0", - "-p", - "no:randomly", - "--benchmark-enable", - f"--benchmark-json={raw_json}", - *campaign.execution.pytest_targets, - ) - start = time.monotonic() - completed = run_execution(benchmark_execution, cwd=ROOT) - runtime_seconds = time.monotonic() - start - command = list(completed.command) - if completed.exit_code != 0 and (not raw_json.exists() or raw_json.stat().st_size == 0): - raise SystemExit(completed.exit_code) - if not raw_json.exists(): - raise SystemExit(f"Benchmark run for {campaign.name} produced no JSON artifact") - payload = json.loads(raw_json.read_text()) - - payload_document = json_document(payload) - benchmarks = parse_pytest_benchmark_stats(payload_document) - benchmarks.sort(key=lambda item: item.mean, reverse=True) - baseline_result = _load_campaign_result(compare_to) if compare_to else None - regressions = _compare_results(benchmarks, baseline_result.benchmarks) if baseline_result else [] - warn_threshold = campaign.warn_pct if warn_pct is None else warn_pct - fail_threshold = campaign.fail_pct if fail_pct is None else fail_pct - worst_regression = regressions[0].delta_pct if regressions else None - - metadata = ScenarioMetadata.from_object(campaign) - result = CampaignResult( - campaign=campaign.name, - description=campaign.description, - commit=_git_output("rev-parse", "HEAD"), - worktree_dirty=_worktree_dirty(), - created_at=datetime.now(UTC).isoformat(), - workspace=str(ROOT), - command=command, - tests=list(campaign.tests), - notes=list(campaign.notes), - benchmark_count=len(benchmarks), - runtime_seconds=runtime_seconds, - exit_code=completed.exit_code, - machine_info=json_document(payload_document.get("machine_info")), - benchmarks=[benchmark_stat_record(bench) for bench in benchmarks], - slowest=[benchmark_stat_record(bench) for bench in benchmarks[:10]], - compare_to=str(compare_to) if compare_to else None, - warn_pct=warn_threshold, - fail_pct=fail_threshold, - regressions=[_regression_record(item) for item in regressions], - worst_regression_pct=worst_regression, - origin=metadata.origin, - path_targets=list(metadata.path_targets), - artifact_targets=list(metadata.artifact_targets), - operation_targets=list(metadata.operation_targets), - maintenance_targets=list(metadata.maintenance_targets), - tags=list(metadata.tags), - ) - - artifact_json.write_text(json.dumps(asdict(result), indent=2, sort_keys=True) + "\n") - artifact_md.write_text(_render_markdown(result)) - - if completed.exit_code != 0: - raise SystemExit(completed.exit_code) - if worst_regression is not None and worst_regression > fail_threshold: - raise SystemExit( - f"Benchmark regression exceeded fail threshold: {worst_regression:.2f}% > {fail_threshold:.2f}%" - ) - return result - - -def render_index() -> str: - artifact_dir = ROOT / ARTIFACT_DIR - rows: list[tuple[str, str, str, str, str, str, str]] = [] - for json_path in sorted(artifact_dir.glob("*.json")): - result = _load_campaign_result(json_path) - md_path = json_path.with_suffix(".md") - worst = "-" if result.worst_regression_pct is None else f"{result.worst_regression_pct:.2f}%" - rows.append( - ( - result.created_at[:10], - result.campaign, - result.commit[:12], - str(result.benchmark_count), - f"{result.runtime_seconds:.2f}s", - worst, - md_path.name if md_path.exists() else "-", - ) - ) - lines = [ - "# Benchmark Campaign Artifacts", - "", - "Use `devtools bench campaign list` to see campaign definitions.", - "Use `devtools bench campaign run ` to record a fresh artifact.", - "Use `devtools bench campaign compare ` to compare two artifacts.", - "", - "| Date | Campaign | Commit | Benchmarks | Runtime | Worst Regression | Markdown |", - "| --- | --- | --- | ---: | ---: | ---: | --- |", - ] - for row in rows: - date, campaign, commit, count, runtime, worst, markdown = row - md_link = f"[{markdown}](./{markdown})" if markdown != "-" else "-" - lines.append(f"| {date} | `{campaign}` | `{commit}` | {count} | {runtime} | {worst} | {md_link} |") - return "\n".join(lines) + "\n" - - -def compare_artifacts(baseline: Path, candidate: Path, fail_pct: float) -> int: - baseline_result = _load_campaign_result(baseline) - candidate_result = _load_campaign_result(candidate) - regressions = _compare_results( - [BenchmarkStat(**entry) for entry in candidate_result.benchmarks], - baseline_result.benchmarks, - ) - if not regressions: - print("No overlapping benchmarks to compare.") - return 0 - print(f"Comparing {candidate.name} against {baseline.name}:") - for regression in regressions[:10]: - print( - f" {regression.fullname}: {regression.delta_pct:.2f}% " - f"({regression.baseline_mean:.6f}s -> {regression.current_mean:.6f}s)" - ) - worst = regressions[0].delta_pct - if worst > fail_pct: - print(f"FAIL: worst regression {worst:.2f}% exceeds {fail_pct:.2f}%") - return 1 - return 0 - - -def parse_args(argv: list[str] | None = None) -> argparse.Namespace: - parser = argparse.ArgumentParser(description=__doc__) - subparsers = parser.add_subparsers(dest="command", required=True) - - subparsers.add_parser("list", help="List benchmark campaigns") - subparsers.add_parser("index", help="Regenerate benchmark artifact index markdown") - - run_parser = subparsers.add_parser("run", help="Run a benchmark campaign and write durable artifacts") - run_parser.add_argument("campaign", choices=sorted(CAMPAIGNS)) - run_parser.add_argument("--json-out", type=Path) - run_parser.add_argument("--markdown-out", type=Path) - run_parser.add_argument("--compare-to", type=Path) - run_parser.add_argument("--warn-pct", type=float) - run_parser.add_argument("--fail-pct", type=float) - - compare_parser = subparsers.add_parser("compare", help="Compare two existing benchmark artifacts") - compare_parser.add_argument("baseline", type=Path) - compare_parser.add_argument("candidate", type=Path) - compare_parser.add_argument("--fail-pct", type=float, default=DEFAULT_FAIL_PCT) - - return parser.parse_args(argv) - - -def main(argv: list[str] | None = None) -> int: - args = parse_args(argv) - if args.command == "list": - for campaign in CAMPAIGNS.values(): - print(f"{campaign.name}: {campaign.description}") - for test in campaign.tests: - print(f" - {test}") - return 0 - if args.command == "index": - path = ROOT / ARTIFACT_DIR / "README.md" - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text(render_index()) - print(path) - return 0 - if args.command == "run": - run_campaign( - CAMPAIGNS[args.campaign], - json_out=args.json_out, - markdown_out=args.markdown_out, - compare_to=args.compare_to, - warn_pct=args.warn_pct, - fail_pct=args.fail_pct, - ) - return 0 - if args.command == "compare": - return compare_artifacts(args.baseline, args.candidate, args.fail_pct) - raise AssertionError(f"Unhandled command: {args.command}") - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/benchmark_campaigns.py b/devtools/benchmark_campaigns.py deleted file mode 100644 index f7b9df24ff..0000000000 --- a/devtools/benchmark_campaigns.py +++ /dev/null @@ -1,113 +0,0 @@ -"""Long-haul benchmark campaign runner. - -Executes reproducible benchmark campaigns against synthetic archives -and produces durable JSON + Markdown reports under .local/benchmark-campaigns/. -""" - -from __future__ import annotations - -from pathlib import Path - -from polylogue.scenarios import CorpusSourceKind, RunnerInvocation, dispatch_runner_execution - -from .authored_scenario_catalog import get_authored_scenario_catalog -from .benchmark_catalog import BenchmarkCampaignEntry -from .campaign_archive_location import CampaignArchiveLocation -from .synthetic_benchmark_runtime import CampaignResult, resolve_synthetic_benchmark_runner - -SYNTHETIC_CAMPAIGNS: dict[str, BenchmarkCampaignEntry] = ( - get_authored_scenario_catalog().synthetic_benchmark_campaign_index() -) - - -async def run_synthetic_benchmark_campaign(name: str, db_path: Path) -> CampaignResult: - """Dispatch one synthetic benchmark campaign by authored scenario id.""" - - campaign = SYNTHETIC_CAMPAIGNS[name] - if campaign.execution is None: - raise ValueError(f"Synthetic benchmark campaign {campaign.name!r} has no execution") - result = await dispatch_runner_execution( - campaign.execution, - runner_resolver=resolve_synthetic_benchmark_runner, - invocation=RunnerInvocation(args=(db_path,)), - ) - result.origin = campaign.origin - result.path_targets = list(campaign.path_targets) - result.artifact_targets = list(campaign.artifact_targets) - result.operation_targets = list(campaign.operation_targets) - result.tags = list(campaign.tags) - return result - - -async def run_full_campaign( - scale_level: str, - output_dir: Path, - *, - corpus_source: CorpusSourceKind | str = CorpusSourceKind.DEFAULT, -) -> list[CampaignResult]: - """Run all benchmark campaigns at a given scale level. - - Generates a synthetic archive at the specified scale, then runs - each campaign against the resulting database. - - Args: - scale_level: One of "small", "medium", "large", "stretch". - output_dir: Directory for archive and report output. - - Returns: - List of CampaignResult for all campaigns. - """ - from devtools.large_archive_generator import ( - ScaleLevel, - generate_archive, - get_default_spec, - ) - - level = ScaleLevel(scale_level) - spec = get_default_spec(level) - - archive_dir = output_dir / f"archive-{scale_level}" - source_kind = CorpusSourceKind(corpus_source) - - # Acquired once and held for the entire campaign run: generation and - # ownership must stay stable from archive generation through every - # subsequent benchmark reopen below (polylogue-ovme.3). A wrong/unowned - # archive_dir fails here, before any archive generation or measurement - # work starts. - with CampaignArchiveLocation.acquire(archive_dir) as location: - print( - f"Generating {scale_level} archive from {source_kind.value} corpus source " - f"({spec.sessions} sessions, ~{spec.message_count} messages)..." - ) - archive_metrics = await generate_archive(spec, archive_dir, corpus_source=source_kind, location=location) - print( - f"Archive generated in {archive_metrics.wall_time_s:.1f}s " - f"({archive_metrics.session_count} convs, " - f"{archive_metrics.message_count} msgs, " - f"{archive_metrics.db_size_bytes / 1024 / 1024:.1f} MB)" - ) - - results: list[CampaignResult] = [] - - for campaign in SYNTHETIC_CAMPAIGNS.values(): - if campaign.scale_targets and scale_level not in campaign.scale_targets: - continue - print(f"Running {campaign.name} campaign...") - # Re-fetched on every reopen (not cached from generation above) - # so a stale/foreign generation is caught before this campaign - # opens a connection against it. - result = await run_synthetic_benchmark_campaign(campaign.name, location.active_index_path) - result.scale_level = scale_level - results.append(result) - metric_value = result.metrics.get(campaign.summary_metric, 0) - print(f" -> {metric_value:.4f}{campaign.summary_label}") - - return results - - -__all__ = [ - "CampaignResult", - "SYNTHETIC_CAMPAIGNS", - "run_full_campaign", - "run_synthetic_benchmark_campaign", -] diff --git a/devtools/benchmark_catalog.py b/devtools/benchmark_catalog.py deleted file mode 100644 index 9db3352615..0000000000 --- a/devtools/benchmark_catalog.py +++ /dev/null @@ -1,22 +0,0 @@ -"""Typed benchmark campaign catalog shared across control-plane surfaces.""" - -from __future__ import annotations - -from .benchmark_models import BenchmarkCampaignEntry -from .benchmark_scenario_catalog import BENCHMARK_SCENARIOS -from .synthetic_benchmark_catalog import SYNTHETIC_BENCHMARK_SCENARIOS - - -def build_benchmark_entries() -> tuple[BenchmarkCampaignEntry, ...]: - return tuple(sorted(BENCHMARK_SCENARIOS, key=lambda item: item.name)) - - -def build_synthetic_benchmark_entries() -> tuple[BenchmarkCampaignEntry, ...]: - return tuple(sorted(SYNTHETIC_BENCHMARK_SCENARIOS, key=lambda item: item.name)) - - -__all__ = [ - "BenchmarkCampaignEntry", - "build_benchmark_entries", - "build_synthetic_benchmark_entries", -] diff --git a/devtools/benchmark_models.py b/devtools/benchmark_models.py deleted file mode 100644 index aa2115c57d..0000000000 --- a/devtools/benchmark_models.py +++ /dev/null @@ -1,40 +0,0 @@ -"""Shared benchmark campaign metadata for control-plane catalogs.""" - -from __future__ import annotations - -from dataclasses import dataclass - -from polylogue.scenarios import ExecutableScenario, ScenarioProjectionSourceKind - - -@dataclass(frozen=True, kw_only=True) -class BenchmarkCampaignEntry(ExecutableScenario): - notes: tuple[str, ...] = () - summary_metric: str = "" - summary_label: str = "" - scale_targets: tuple[str, ...] = () - projection_kind: ScenarioProjectionSourceKind = ScenarioProjectionSourceKind.BENCHMARK_CAMPAIGN - - @property - def projection_source_kind(self) -> ScenarioProjectionSourceKind: - return self.projection_kind - - @property - def warn_pct(self) -> float: - return self.assertion.resolved_benchmark_warn_pct() - - @property - def fail_pct(self) -> float: - return self.assertion.resolved_benchmark_fail_pct() - - -def compile_benchmark_campaigns( - campaigns: tuple[BenchmarkCampaignEntry, ...], -) -> dict[str, BenchmarkCampaignEntry]: - return {campaign.name: campaign for campaign in campaigns} - - -__all__ = [ - "BenchmarkCampaignEntry", - "compile_benchmark_campaigns", -] diff --git a/devtools/benchmark_scenario_catalog.py b/devtools/benchmark_scenario_catalog.py deleted file mode 100644 index 8ada96f334..0000000000 --- a/devtools/benchmark_scenario_catalog.py +++ /dev/null @@ -1,115 +0,0 @@ -"""Authored durable benchmark campaigns shared across control-plane surfaces.""" - -from __future__ import annotations - -from devtools.benchmark_models import BenchmarkCampaignEntry, compile_benchmark_campaigns -from polylogue.scenarios import pytest_execution - -BENCHMARK_SCENARIOS: tuple[BenchmarkCampaignEntry, ...] = ( - BenchmarkCampaignEntry( - name="search-filters", - description="FTS and SessionFilter benchmark domain", - execution=pytest_execution("tests/benchmarks/test_search_filters.py"), - notes=( - "Canonical search/filter latency domain.", - "Keep on session-seeded DB fixtures for comparability.", - ), - origin="authored.benchmark-domain", - artifact_targets=("session_query_results", "message_fts"), - operation_targets=("query-sessions", "benchmark.query.search-filters"), - tags=("benchmark", "search", "filters"), - ), - BenchmarkCampaignEntry( - name="storage", - description="Repository/backend list/get-many/save benchmark domain", - execution=pytest_execution("tests/benchmarks/test_storage.py"), - notes=("Canonical storage CRUD and batch-write latency domain.",), - origin="authored.benchmark-domain", - conceptual_artifact_targets=("session_rows", "message_rows", "raw_rows"), - operation_targets=("benchmark.storage.crud",), - tags=("benchmark", "storage"), - ), - BenchmarkCampaignEntry( - name="pipeline", - description="Index rebuild/update plus hashing and semantic helper benchmark domain", - execution=pytest_execution("tests/benchmarks/test_pipeline.py"), - notes=("Covers indexing and hot helper throughput.",), - origin="authored.benchmark-domain", - conceptual_artifact_targets=("index_state", "pipeline_helpers"), - operation_targets=("benchmark.pipeline.index-and-helpers",), - tags=("benchmark", "pipeline"), - ), - BenchmarkCampaignEntry( - name="reader-api", - description="Reader HTTP API list/get/facets/context-image/cost-rollup benchmark domain", - execution=pytest_execution("tests/benchmarks/test_reader_api.py"), - notes=( - "Covers reader read-path latency for list, get, facets, and context operations.", - "SLO catalog gates are defined in docs/plans/slo-catalog.yaml.", - ), - origin="authored.benchmark-domain", - conceptual_artifact_targets=("reader_list_results", "reader_facets"), - operation_targets=("benchmark.reader.api",), - tags=("benchmark", "reader", "api"), - ), - BenchmarkCampaignEntry( - name="session-digest", - description="Deterministic session digest transform/render benchmark domain", - execution=pytest_execution("tests/benchmarks/test_session_digest.py"), - notes=( - "Covers session digest transform compilation and report rendering over tool-heavy sessions.", - "Keeps #1880 session digest artifact shape in the generated benchmark inventory.", - ), - origin="authored.benchmark-domain", - path_targets=("session-digest-transform-loop",), - artifact_targets=("session_digest", "forensic_index", "resume_bundle", "session_report_markdown"), - operation_targets=( - "compile-session-digest", - "render-session-report", - "benchmark.transform.session-digest", - "benchmark.transform.session-report", - ), - tags=("benchmark", "transform", "session-analysis"), - ), - BenchmarkCampaignEntry( - name="daemon-convergence", - description="Daemon ingest convergence at synthetic scale tiers — single-file and multi-session", - execution=pytest_execution("tests/benchmarks/test_daemon_convergence.py"), - notes=( - "Measures convergence stage timing at small/medium/large synthetic tiers.", - "Run with --benchmark-enable -p no:xdist -o 'addopts='", - ), - origin="authored.benchmark-domain", - conceptual_artifact_targets=("daemon_convergence_timing",), - operation_targets=("benchmark.daemon.convergence",), - tags=("benchmark", "daemon", "convergence"), - ), - BenchmarkCampaignEntry( - name="archive-maintenance", - description="Archive backup planning, blob-GC dry-run, and space-report benchmark domain", - execution=pytest_execution("tests/benchmarks/test_archive_maintenance.py"), - notes=( - "Covers read-only archive maintenance performance artifacts.", - "Backup runtime copy/restore semantics are intentionally out of scope.", - ), - origin="authored.benchmark-domain", - artifact_targets=("archive_readiness",), - operation_targets=( - "benchmark.archive.backup-plan", - "benchmark.archive.blob-gc-dry-run", - "benchmark.archive.space-report", - ), - maintenance_targets=("orphaned_blobs",), - tags=("benchmark", "archive", "maintenance", "backup", "gc"), - ), -) - -BENCHMARK_SCENARIO_INDEX: dict[str, BenchmarkCampaignEntry] = compile_benchmark_campaigns(BENCHMARK_SCENARIOS) - - -__all__ = [ - "BENCHMARK_SCENARIO_INDEX", - "BENCHMARK_SCENARIOS", - "BenchmarkCampaignEntry", - "compile_benchmark_campaigns", -] diff --git a/devtools/campaign_report.py b/devtools/campaign_report.py deleted file mode 100644 index 2a025e3bbb..0000000000 --- a/devtools/campaign_report.py +++ /dev/null @@ -1,182 +0,0 @@ -"""Generate Markdown and JSON reports from campaign results.""" - -from __future__ import annotations - -import json -from dataclasses import asdict -from datetime import UTC, datetime -from pathlib import Path - -from devtools.synthetic_benchmark_runtime import CampaignResult - - -def generate_campaign_markdown(results: list[CampaignResult]) -> str: - """Produce a Markdown report from a list of campaign results. - - The report includes scale level, per-campaign timing, DB size, - row counts, and a comparison table for easy diffing. - """ - if not results: - return "# Benchmark Campaign Report\n\nNo results.\n" - - scale_level = results[0].scale_level or "unknown" - timestamp = results[0].timestamp - - lines = [ - "# Benchmark Campaign Report", - "", - f"- **Scale level**: {scale_level}", - f"- **Generated**: {timestamp}", - "", - "## Summary", - "", - "| Campaign | Key Metric | Value |", - "| --- | --- | ---: |", - ] - - for result in results: - key_metric, value = _pick_key_metric(result) - lines.append(f"| {result.campaign_name} | {key_metric} | {value} |") - - lines.extend(["", "## Detailed Results", ""]) - - for result in results: - lines.extend(_render_campaign_section(result)) - - # DB stats comparison table - lines.extend(["", "## Database Statistics", ""]) - lines.extend(_render_db_stats_table(results)) - - lines.append("") - return "\n".join(lines) - - -def _pick_key_metric(result: CampaignResult) -> tuple[str, str]: - """Pick the single most important metric from a campaign result.""" - m = result.metrics - match result.campaign_name: - case "fts-rebuild": - return "rebuild_wall_s", f"{m.get('rebuild_wall_s', 0):.3f}s" - case "incremental-index": - return "total_wall_s", f"{m.get('total_wall_s', 0):.3f}s" - case "filter-scan": - return "list_50_wall_s", f"{m.get('list_50_wall_s', 0):.4f}s" - case "startup-readiness": - return "total_readiness_s", f"{m.get('total_readiness_s', 0):.4f}s" - case _: - if m: - key = next(iter(m)) - return key, f"{m[key]}" - return "n/a", "n/a" - - -def _render_campaign_section(result: CampaignResult) -> list[str]: - """Render a single campaign's detailed section.""" - lines = [ - f"### {result.campaign_name}", - "", - ] - if result.path_targets or result.artifact_targets or result.operation_targets or result.tags: - lines.extend( - [ - "| Scenario metadata | Value |", - "| --- | --- |", - f"| origin | `{result.origin}` |", - ] - ) - if result.path_targets: - lines.append(f"| path targets | `{', '.join(result.path_targets)}` |") - if result.artifact_targets: - lines.append(f"| artifact targets | `{', '.join(result.artifact_targets)}` |") - if result.operation_targets: - lines.append(f"| operation targets | `{', '.join(result.operation_targets)}` |") - if result.tags: - lines.append(f"| tags | `{', '.join(result.tags)}` |") - lines.append("") - - if result.metrics: - lines.extend( - [ - "| Metric | Value |", - "| --- | ---: |", - ] - ) - for key, value in sorted(result.metrics.items()): - if isinstance(value, float): - lines.append(f"| {key} | {value:.4f} |") - else: - lines.append(f"| {key} | {value} |") - lines.append("") - - return lines - - -def _render_db_stats_table(results: list[CampaignResult]) -> list[str]: - """Render a DB statistics comparison table.""" - # Collect all unique stat keys - all_keys: list[str] = [] - seen: set[str] = set() - for result in results: - for key in result.db_stats: - if key not in seen: - all_keys.append(key) - seen.add(key) - - if not all_keys: - return ["_No database statistics collected._"] - - # Header - campaign_names = [r.campaign_name for r in results] - header = "| Stat | " + " | ".join(campaign_names) + " |" - separator = "| --- | " + " | ".join("---:" for _ in campaign_names) + " |" - lines = [header, separator] - - for key in all_keys: - values = [] - for result in results: - val = result.db_stats.get(key, "") - if isinstance(val, int) and key.endswith("_bytes"): - values.append(f"{val / 1024 / 1024:.1f} MB") - else: - values.append(str(val)) - lines.append(f"| {key} | " + " | ".join(values) + " |") - - return lines - - -def generate_campaign_json(results: list[CampaignResult]) -> str: - """Produce a JSON report from a list of campaign results.""" - payload = { - "generated_at": datetime.now(UTC).isoformat(), - "scale_level": results[0].scale_level if results else "unknown", - "campaigns": [asdict(r) for r in results], - } - return json.dumps(payload, indent=2, sort_keys=True) + "\n" - - -def save_campaign_reports(results: list[CampaignResult], output_dir: Path) -> list[Path]: - """Save both Markdown and JSON reports to the output directory. - - Returns: - List of paths to saved report files. - """ - output_dir.mkdir(parents=True, exist_ok=True) - - scale = results[0].scale_level if results else "unknown" - date = datetime.now(UTC).strftime("%Y-%m-%d") - stem = f"{date}-{scale}" - - md_path = output_dir / f"{stem}.md" - json_path = output_dir / f"{stem}.json" - - md_path.write_text(generate_campaign_markdown(results), encoding="utf-8") - json_path.write_text(generate_campaign_json(results), encoding="utf-8") - - return [md_path, json_path] - - -__all__ = [ - "generate_campaign_json", - "generate_campaign_markdown", - "save_campaign_reports", -] diff --git a/devtools/chatgpt_lifecycle_anchor_audit.py b/devtools/chatgpt_lifecycle_anchor_audit.py deleted file mode 100644 index afdae3e6ab..0000000000 --- a/devtools/chatgpt_lifecycle_anchor_audit.py +++ /dev/null @@ -1,375 +0,0 @@ -"""Read-only ChatGPT lifecycle-anchor census through the production parser route. - -This command audits whether quarantined ChatGPT revisions currently exhibit -the historical mapping-order failure: two exports with equal transcript and -lifecycle content but a different generation-lifecycle anchor. It does not -change archive state. The only optional write is a caller-selected, -sanitized JSON receipt outside the archive. -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import os -import sqlite3 -import subprocess -from collections import Counter, defaultdict -from collections.abc import Iterable -from dataclasses import dataclass -from pathlib import Path -from typing import Literal, TextIO - -from polylogue.archive.session_revision_membership import MembershipRevision, _relation, classify_membership_revisions -from polylogue.core.enums import Provider -from polylogue.core.hashing import hash_payload -from polylogue.pipeline.ids import _event_content_payload, session_revision_projection -from polylogue.sources.parsers.base import ParsedSession, ParsedSessionEvent -from polylogue.sources.revision_backfill import _parse_one -from polylogue.storage.blob_store import BlobStore - -_Relation = Literal["equal", "a_contains_b", "b_contains_a", "conflict"] - -SCHEMA = "polylogue.chatgpt-lifecycle-anchor-audit.v2" -TARGET_PREDICATE = ( - "A pair in one persisted logical_source_key cohort where each parsed session has exactly one " - "generation_lifecycle event (other session events are allowed), their source_message_provider_id " - "anchors differ, message_contents, attachment_identities, and attachment_contents are equal, " - "generation_lifecycle event " - "content hashes after removing source_message_provider_id are equal, all normalized event content " - "is equal after that same lifecycle-only exception, and the production _relation is conflict." -) -SELECTION_SQL = """ -SELECT r.raw_id, r.source_path, lower(hex(r.blob_hash)) AS blob_hash, - m.logical_source_key, m.provider_session_id -FROM raw_sessions AS r -JOIN raw_session_memberships AS m ON m.raw_id = r.raw_id -WHERE r.origin = 'chatgpt-export' AND r.revision_authority = 'quarantined' -ORDER BY m.logical_source_key, r.raw_id -""".strip() -POPULATION_SQL = """ -SELECT raw_id -FROM raw_sessions -WHERE origin = 'chatgpt-export' AND revision_authority = 'quarantined' -ORDER BY raw_id -""".strip() - - -@dataclass(frozen=True, slots=True) -class _RawMember: - raw_id: str - source_path: str - blob_hash: str - logical_source_key: str - provider_session_id: str - - -@dataclass(frozen=True, slots=True) -class _ParsedMember: - revision: MembershipRevision - session: ParsedSession - - -def _connect_read_only(path: Path) -> sqlite3.Connection: - return sqlite3.connect(f"file:{path}?mode=ro", uri=True) - - -def _database_provenance(conn: sqlite3.Connection, path: Path) -> dict[str, int]: - stat = path.stat() - return { - "size_bytes": stat.st_size, - "mtime_ns": stat.st_mtime_ns, - "sqlite_schema_version": int(conn.execute("PRAGMA schema_version").fetchone()[0]), - "sqlite_user_version": int(conn.execute("PRAGMA user_version").fetchone()[0]), - } - - -def _git_provenance() -> dict[str, object]: - repo_root = Path(__file__).resolve().parents[1] - try: - revision = subprocess.check_output( - ["git", "-C", os.fspath(repo_root), "rev-parse", "--verify", "HEAD"], - text=True, - stderr=subprocess.DEVNULL, - timeout=5, - ).strip() - status = subprocess.run( - ["git", "-C", os.fspath(repo_root), "status", "--porcelain=v1", "--untracked-files=all"], - capture_output=True, - check=True, - text=True, - timeout=5, - ).stdout - except (OSError, subprocess.CalledProcessError, subprocess.TimeoutExpired) as error: - raise RuntimeError("ChatGPT lifecycle-anchor audit requires a readable git producer checkout") from error - return { - "git_revision": revision, - "working_tree_clean": not bool(status), - "working_tree_status_sha256": hashlib.sha256(status.encode("utf-8")).hexdigest(), - } - - -def _generation_events(session: ParsedSession) -> list[ParsedSessionEvent]: - return [event for event in session.session_events if event.event_type == "generation_lifecycle"] - - -def _anchor_independent_event_content(event: ParsedSessionEvent) -> bytes: - """Hash one lifecycle event without its provider-message anchor.""" - payload = _event_content_payload(event) - payload.pop("source_message_provider_id", None) - return bytes.fromhex(hash_payload(payload)) - - -def _event_content_signature(event: ParsedSessionEvent) -> bytes: - """Hash normalized event content, retaining anchors except for lifecycle events.""" - if event.event_type == "generation_lifecycle": - return _anchor_independent_event_content(event) - return bytes.fromhex(hash_payload(_event_content_payload(event))) - - -def _session_event_content_signatures(session: ParsedSession) -> Counter[bytes]: - """Return normalized event content as a multiset, independent of array order.""" - return Counter(_event_content_signature(event) for event in session.session_events) - - -def _blob_store_snapshot(blob_store: BlobStore) -> dict[str, object]: - """Capture a deterministic, read-only identity and integrity scan of blobs.""" - snapshot_digest = hashlib.sha256() - integrity_digest = hashlib.sha256() - canonical_blob_count = 0 - canonical_blob_bytes = 0 - verified_blob_count = 0 - hash_mismatch_count = 0 - invalid_namespace_entry_count = 0 - - for entry in blob_store.iter_namespace(): - if entry.hash_hex is None: - invalid_namespace_entry_count += 1 - record = { - "kind": entry.kind.value, - "issue": entry.issue.value if entry.issue is not None else None, - "relative_path": entry.relative_path, - } - encoded = json.dumps(record, sort_keys=True, separators=(",", ":")).encode("utf-8") - snapshot_digest.update(encoded) - snapshot_digest.update(b"\n") - integrity_digest.update(encoded) - integrity_digest.update(b"\n") - continue - - size_bytes = entry.path.stat().st_size - actual_digest = hashlib.sha256() - with entry.path.open("rb") as blob: - while chunk := blob.read(1024 * 1024): - actual_digest.update(chunk) - verified = actual_digest.hexdigest() == entry.hash_hex - canonical_blob_count += 1 - canonical_blob_bytes += size_bytes - verified_blob_count += int(verified) - hash_mismatch_count += int(not verified) - snapshot_record = {"hash": entry.hash_hex, "size_bytes": size_bytes} - integrity_record = { - **snapshot_record, - "verified": verified, - "observed_sha256": actual_digest.hexdigest(), - } - snapshot_encoded = json.dumps(snapshot_record, sort_keys=True, separators=(",", ":")).encode("utf-8") - integrity_encoded = json.dumps(integrity_record, sort_keys=True, separators=(",", ":")).encode("utf-8") - snapshot_digest.update(snapshot_encoded) - snapshot_digest.update(b"\n") - integrity_digest.update(integrity_encoded) - integrity_digest.update(b"\n") - - return { - "snapshot_sha256": snapshot_digest.hexdigest(), - "canonical_blob_count": canonical_blob_count, - "canonical_blob_bytes": canonical_blob_bytes, - "integrity": { - "scan": "full_read_only_namespace_and_content_hash", - "verified_blob_count": verified_blob_count, - "hash_mismatch_count": hash_mismatch_count, - "invalid_namespace_entry_count": invalid_namespace_entry_count, - "integrity_sha256": integrity_digest.hexdigest(), - }, - } - - -def _matches_target(left: _ParsedMember, right: _ParsedMember, relation: _Relation) -> bool: - left_generation_events = _generation_events(left.session) - right_generation_events = _generation_events(right.session) - if len(left_generation_events) != 1 or len(right_generation_events) != 1: - return False - left_event, right_event = left_generation_events[0], right_generation_events[0] - left_projection = left.revision.projection - right_projection = right.revision.projection - return ( - left_event.source_message_provider_id != right_event.source_message_provider_id - and left_projection.message_contents == right_projection.message_contents - and left_projection.attachment_identities == right_projection.attachment_identities - and left_projection.attachment_contents == right_projection.attachment_contents - and _session_event_content_signatures(left.session) == _session_event_content_signatures(right.session) - and _anchor_independent_event_content(left_event) == _anchor_independent_event_content(right_event) - and relation == "conflict" - ) - - -def _load_existing_heads(index_conn: sqlite3.Connection) -> dict[str, str]: - return { - str(row[0]): str(row[1]) - for row in index_conn.execute("SELECT logical_source_key, accepted_raw_id FROM raw_revision_heads") - } - - -def _parse_member(member: _RawMember, blob_store: BlobStore, archive_root: Path) -> _ParsedMember: - sessions = _parse_one( - Provider.CHATGPT, - blob_store.read_all(member.blob_hash), - member.source_path, - archive_root=archive_root, - fallback_id_override=member.provider_session_id, - ) - matches = [session for session in sessions if session.provider_session_id == member.provider_session_id] - if len(matches) != 1: - raise RuntimeError( - "ChatGPT lifecycle-anchor audit expected one parsed session for a persisted membership row, " - f"got {len(matches)}" - ) - session = matches[0] - return _ParsedMember(MembershipRevision(member.raw_id, session_revision_projection(session)), session) - - -def _cohorts(rows: Iterable[_RawMember]) -> dict[str, list[_RawMember]]: - grouped: dict[str, list[_RawMember]] = defaultdict(list) - for row in rows: - grouped[row.logical_source_key].append(row) - return dict(grouped) - - -def run_audit(archive_root: Path) -> dict[str, object]: - """Run the full current-corpus census without opening an archive writer.""" - producer = _git_provenance() - source_db = archive_root / "source.db" - index_db = archive_root / "index.db" - blob_store = BlobStore(archive_root / "blob") - source_conn = _connect_read_only(source_db) - index_conn = _connect_read_only(index_db) - try: - population_raw_ids = {str(row[0]) for row in source_conn.execute(POPULATION_SQL)} - rows = [_RawMember(*map(str, row)) for row in source_conn.execute(SELECTION_SQL)] - rows_by_raw_id: dict[str, list[_RawMember]] = defaultdict(list) - for row in rows: - rows_by_raw_id[row.raw_id].append(row) - duplicated_membership_raw_count = sum(1 for members in rows_by_raw_id.values() if len(members) != 1) - if duplicated_membership_raw_count: - raise RuntimeError("ChatGPT lifecycle-anchor audit requires exactly one membership row per selected raw") - cohorts = _cohorts(rows) - relation_counts: Counter[str] = Counter() - classifier_counts: Counter[str] = Counter() - target_pair_count = 0 - parsed_raw_count = 0 - heads = _load_existing_heads(index_conn) - blob_snapshot = _blob_store_snapshot(blob_store) - for logical_source_key in sorted(cohorts): - revisions = [ - _parse_member(member, blob_store, archive_root) - for member in sorted(cohorts[logical_source_key], key=lambda member: member.raw_id) - ] - parsed_raw_count += len(revisions) - for index, left in enumerate(revisions): - for right in revisions[index + 1 :]: - relation = _relation(left.revision.projection, right.revision.projection) - relation_counts[relation] += 1 - if _matches_target(left, right, relation): - target_pair_count += 1 - classification = classify_membership_revisions( - [revision.revision for revision in revisions], existing_accepted_raw_id=heads.get(logical_source_key) - ) - classifier_counts["cohorts_with_accepted_raw"] += bool(classification.accepted_raw_ids) - classifier_counts["cohorts_with_equivalent_raw"] += bool(classification.equivalent_raw_ids) - classifier_counts["cohorts_with_ambiguous_raw"] += bool(classification.ambiguous_raw_ids) - cohort_sizes = Counter(len(members) for members in cohorts.values()) - return { - "schema": SCHEMA, - "provenance": { - "archive_access": "SQLite source.db and index.db opened mode=ro; blob files read only; no archive writer created.", - "producer_git_revision": producer["git_revision"], - "producer_working_tree_clean": producer["working_tree_clean"], - "producer_working_tree_status_sha256": producer["working_tree_status_sha256"], - "production_route": [ - "polylogue.sources.revision_backfill._parse_one", - "polylogue.pipeline.ids.session_revision_projection", - "polylogue.archive.session_revision_membership._relation", - "polylogue.archive.session_revision_membership.classify_membership_revisions", - ], - "source_db": _database_provenance(source_conn, source_db), - "index_db": _database_provenance(index_conn, index_db), - "blob_store": blob_snapshot, - }, - "selection": {"sql": SELECTION_SQL, "population_sql": POPULATION_SQL}, - "target_predicate": TARGET_PREDICATE, - "denominators": { - "selected_quarantined_chatgpt_raw_count": len(population_raw_ids), - "selected_membership_row_count": len(rows), - "membershipless_selected_raw_count": len(population_raw_ids - set(rows_by_raw_id)), - "logical_source_key_count": len(cohorts), - "singleton_cohort_count": cohort_sizes[1], - "multi_candidate_cohort_count": sum(count for size, count in cohort_sizes.items() if size > 1), - "raws_in_multi_candidate_cohorts": sum( - size * count for size, count in cohort_sizes.items() if size > 1 - ), - "parsed_and_projected_raw_count": parsed_raw_count, - }, - "outcomes": { - "pair_relation_counts": { - name: relation_counts[name] for name in ("equal", "a_contains_b", "b_contains_a", "conflict") - }, - "target_pair_count": target_pair_count, - "classifier_cohort_counts": dict(sorted(classifier_counts.items())), - }, - "scope": { - "sanitized": "No raw ids, native ids, source paths, blob hashes, titles, or payload content are emitted.", - "conclusion_limit": ( - "A zero target_pair_count describes only this current parser-and-corpus snapshot. It does not establish " - "the historical pre-fix replay required to reclassify or remove any graph gate." - ), - }, - } - finally: - index_conn.close() - source_conn.close() - - -def _write_receipt(path: Path, receipt: dict[str, object]) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n") - - -def main(argv: list[str] | None = None, *, stdout: TextIO | None = None) -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--archive-root", type=Path, required=True, help="Archive root to inspect without mutation.") - parser.add_argument("--receipt", type=Path, help="Optional worktree-local path for the sanitized JSON receipt.") - parser.add_argument( - "--json", - action="store_true", - help="Emit the machine-readable JSON audit report (the default output format).", - ) - args = parser.parse_args(argv) - archive_root = args.archive_root.resolve() - if args.receipt is not None: - receipt_path = args.receipt.resolve() - try: - receipt_path.relative_to(archive_root) - except ValueError: - pass - else: - parser.error("--receipt must resolve outside --archive-root") - receipt = run_audit(archive_root) - if args.receipt is not None: - _write_receipt(args.receipt.resolve(), receipt) - print(json.dumps(receipt, indent=2, sort_keys=True), file=stdout) - return 0 - - -if __name__ == "__main__": # pragma: no cover - raise SystemExit(main()) diff --git a/devtools/claim_vs_evidence.py b/devtools/claim_vs_evidence.py index 06342c8dfb..6a2beca24a 100644 --- a/devtools/claim_vs_evidence.py +++ b/devtools/claim_vs_evidence.py @@ -15,7 +15,8 @@ from typing import Any from polylogue.archive.actions.followup import classify_failed_followup_evidence -from polylogue.config import Config, get_config +from polylogue.config import Config, active_archive_root, get_config +from polylogue.storage.index_generation import RebuildLease from polylogue.storage.sqlite.connection_profile import open_readonly_connection _WORDLESS_CONTINUATION_TEXT_CHAR_LIMIT = 40 @@ -110,7 +111,8 @@ def _parser() -> argparse.ArgumentParser: help=( "Register this run's structured-failure selection, matched rows, and headline numbers " "as durable query/result-set/finding evidence in the archive's user tier (polylogue-rxdo.13). " - "Off by default; report generation stays read-only unless explicitly requested." + "Off by default; report generation stays read-only unless explicitly requested, and the " + "materializing form requires exclusive offline writer ownership." ), ) parser.add_argument("--json", action="store_true", help="Emit JSON report to stdout.") @@ -131,6 +133,11 @@ def _config_with_archive_root(config: Config, archive_root: Path | None) -> Conf ) +def _report_config(args: argparse.Namespace) -> Config: + """Resolve one file-set authority for both report reads and optional writes.""" + return _config_with_archive_root(get_config(), args.archive_root) + + def _user_version(conn: Connection) -> int: row = conn.execute("PRAGMA user_version").fetchone() return int(row[0]) if row else 0 @@ -449,6 +456,7 @@ def _failure_outcome_rows(conn: Connection, *, limit: int, origin: str | None) - SELECT r.session_id, r.message_id AS tool_result_message_id, + r.block_id AS tool_result_block_id, r.tool_id AS tool_result_tool_id, s.origin, r.tool_result_is_error AS is_error, @@ -463,6 +471,7 @@ def _failure_outcome_rows(conn: Connection, *, limit: int, origin: str | None) - SELECT r.session_id, r.message_id AS tool_result_message_id, + r.block_id AS tool_result_block_id, r.tool_id AS tool_result_tool_id, s.origin, r.tool_result_is_error AS is_error, @@ -479,6 +488,7 @@ def _failure_outcome_rows(conn: Connection, *, limit: int, origin: str | None) - SELECT r.session_id, r.message_id AS tool_result_message_id, + r.block_id AS tool_result_block_id, r.tool_id AS tool_result_tool_id, s.origin, r.tool_result_is_error AS is_error, @@ -494,7 +504,7 @@ def _failure_outcome_rows(conn: Connection, *, limit: int, origin: str | None) - ) SELECT * FROM failed - ORDER BY session_id, tool_result_tool_id, order_message_id + ORDER BY session_id, tool_result_tool_id, order_message_id, tool_result_block_id LIMIT ? """, params, @@ -510,7 +520,7 @@ def _paired_failure_rows( paired_rows: list[dict[str, object]] = [] for start in range(0, len(failure_rows), chunk_size): chunk = failure_rows[start : start + chunk_size] - placeholders = ",".join("(?, ?, ?, ?, ?, ?, ?, ?)" for _ in chunk) + placeholders = ",".join("(?, ?, ?, ?, ?, ?, ?, ?, ?)" for _ in chunk) params: list[object] = [] for offset, row in enumerate(chunk, start=start): params.extend( @@ -518,6 +528,7 @@ def _paired_failure_rows( offset, row["session_id"], row["tool_result_message_id"], + row["tool_result_block_id"], row["tool_result_tool_id"], row["origin"], row["is_error"], @@ -533,6 +544,7 @@ def _paired_failure_rows( sort_index, session_id, tool_result_message_id, + tool_result_block_id, tool_result_tool_id, origin, is_error, @@ -545,11 +557,12 @@ def _paired_failure_rows( SELECT w.sort_index, w.session_id, - u.message_id, + a.message_id, w.tool_result_message_id, + w.tool_result_block_id, w.tool_result_tool_id, - u.tool_name, - u.tool_command, + a.tool_name, + a.tool_command, w.origin, w.is_error, w.exit_code, @@ -559,23 +572,33 @@ def _paired_failure_rows( SELECT nm.message_id FROM messages AS nm WHERE nm.session_id = w.session_id - AND nm.role = 'assistant' + AND nm.material_origin = 'assistant_authored' AND nm.position > rm.position - ORDER BY nm.position + AND nm.position < COALESCE( + ( + SELECT MIN(next_human.position) + FROM messages AS next_human + WHERE next_human.session_id = w.session_id + AND next_human.material_origin = 'human_authored' + AND next_human.position > rm.position + ), + 9223372036854775807 + ) + ORDER BY nm.position, nm.variant_index, nm.message_id LIMIT 1 ) AS next_message_id FROM wanted AS w - JOIN blocks AS u INDEXED BY idx_blocks_tool_id - ON u.tool_id = w.tool_result_tool_id - AND u.session_id = w.session_id - AND u.block_type = 'tool_use' - JOIN messages AS m ON m.message_id = u.message_id + JOIN actions AS a + ON a.session_id = w.session_id + AND a.tool_result_block_id = w.tool_result_block_id + JOIN messages AS m ON m.message_id = a.message_id JOIN messages AS rm ON rm.message_id = w.tool_result_message_id ) SELECT p.session_id, p.message_id, p.tool_result_message_id, + p.tool_result_block_id, p.tool_result_tool_id, p.tool_name, p.tool_command, @@ -622,7 +645,7 @@ def _assistant_window_details( SELECT MIN(next_user.position) FROM messages AS next_user WHERE next_user.session_id = w.session_id - AND next_user.role = 'user' + AND next_user.material_origin = 'human_authored' AND next_user.position > w.result_position ) AS next_user_position FROM wanted AS w @@ -632,14 +655,15 @@ def _assistant_window_details( b.sort_index, m.message_id, m.position, - ROW_NUMBER() OVER ( + m.variant_index, + DENSE_RANK() OVER ( PARTITION BY b.sort_index ORDER BY m.position ) AS assistant_rank FROM bounded AS b JOIN messages AS m ON m.session_id = b.session_id - AND m.role = 'assistant' + AND m.material_origin = 'assistant_authored' AND m.position > b.result_position AND ( b.next_user_position IS NULL @@ -660,7 +684,7 @@ def _assistant_window_details( COALESCE(b.text, '') AS text FROM top_window AS w LEFT JOIN blocks AS b ON b.message_id = w.message_id - ORDER BY w.sort_index, w.assistant_rank, b.position + ORDER BY w.sort_index, w.assistant_rank, w.variant_index, w.message_id, b.position """, [*params, window_size], ) @@ -946,7 +970,7 @@ def _calibration_labels_path(args: argparse.Namespace) -> Path | None: return candidate if candidate.exists() else None -def build_report(args: argparse.Namespace) -> dict[str, Any]: +def build_report(args: argparse.Namespace, *, config: Config | None = None) -> dict[str, Any]: if args.limit < 1: raise ValueError("--limit must be positive") if args.sample_limit < 1: @@ -955,7 +979,8 @@ def build_report(args: argparse.Namespace) -> dict[str, Any]: raise ValueError("--n-min must be positive") if args.calibration_size < 0: raise ValueError("--calibration-size must be non-negative") - config = _config_with_archive_root(get_config(), args.archive_root) + config = config or _report_config(args) + file_set_root = active_archive_root(config) index_db = config.db_path conn = open_readonly_connection(index_db) try: @@ -1015,6 +1040,7 @@ def build_report(args: argparse.Namespace) -> dict[str, Any]: "session_ref": f"session:{row['session_id']}", "tool_message_ref": f"message:{row['message_id']}", "tool_result_message_ref": f"message:{row['tool_result_message_id']}", + "tool_result_block_ref": f"block:{row['tool_result_block_id']}", "tool_result_tool_id": row["tool_result_tool_id"], "next_message_ref": f"message:{row['next_message_id']}" if row["next_message_id"] else None, "tool_name": tool, @@ -1115,7 +1141,7 @@ def build_report(args: argparse.Namespace) -> dict[str, Any]: "report_version": 1, "captured_at": datetime.now(UTC).isoformat(), "command": "devtools workspace claim-vs-evidence", - "archive_root": str(config.archive_root), + "archive_root": str(file_set_root), "index_db": str(index_db), "index_schema_version": schema_version, "limit": args.limit, @@ -1131,7 +1157,7 @@ def build_report(args: argparse.Namespace) -> dict[str, Any]: "then proportional fill by origin failure count; each origin candidate frame is bounded " "before pairing to tool-use rows" ), - "selection_order": "origin, session_id, tool_id, tool_result_message_id", + "selection_order": "origin, session_id, tool_id, tool_result_message_id, tool_result_block_id", "failure_predicate": "tool_result_is_error = 1 OR tool_result_exit_code != 0", "classification_scope": "immediately following assistant message only", "sensitivity_scope": "next 3 assistant messages after the failed result, stopping before the next user message", @@ -1184,7 +1210,7 @@ def build_report(args: argparse.Namespace) -> dict[str, Any]: "other": "Any tool name outside the explicit benign/consequential methodology sets.", }, "evidence": { - "member_refs": sorted({f"message:{row['tool_result_message_id']}" for row in rows}), + "member_refs": sorted(f"block:{row['tool_result_block_id']}" for row in rows), }, "calibration": calibration, "calibration_sample": [ @@ -1659,9 +1685,8 @@ def _write_readme(path: Path, report: dict[str, Any]) -> None: "```bash", "devtools workspace claim-vs-evidence \\", " --limit 5000 \\", - " --out-dir .agent/demos/claim-vs-evidence \\", + " --out-dir .local/evidence/claim-vs-evidence \\", " --json", - "devtools workspace demo-shelf", "```", "", "## Files", @@ -1672,7 +1697,7 @@ def _write_readme(path: Path, report: dict[str, Any]) -> None: f"- `{_COLD_READER_GATE_FILE}` — cold-reader prompt and passing-answer checklist.", f"- `{_CALIBRATION_SAMPLE_FILE}` — deterministic sample for marker calibration.", f"- `{_CALIBRATION_LABELS_FILE}` — optional human labels consumed on regeneration.", - "- `summary.json` — current demo-shelf claim/non-claim/proof/caveat summary.", + "- `summary.json` — local claim/non-claim/proof/caveat summary.", "- `README.md` — this human-readable packet.", "", ] @@ -1681,19 +1706,27 @@ def _write_readme(path: Path, report: dict[str, Any]) -> None: def main(argv: list[str] | None = None) -> int: parsed = _parser().parse_args(argv) + config = _report_config(parsed) + evidence: object | None = None try: - report = build_report(parsed) + if parsed.materialize_evidence: + from devtools.claim_vs_evidence_evidence import materialize_claim_vs_evidence_evidence_under_lease + + file_set_root = active_archive_root(config) + with RebuildLease(file_set_root): + report = build_report(parsed, config=config) + evidence = materialize_claim_vs_evidence_evidence_under_lease( + report, + archive_root=file_set_root, + now_ms=int(datetime.now(UTC).timestamp() * 1000), + ) + else: + report = build_report(parsed, config=config) except ValueError as exc: print(f"claim-vs-evidence: {exc}", file=sys.stderr) return 2 if parsed.materialize_evidence: - from devtools.claim_vs_evidence_evidence import materialize_claim_vs_evidence_evidence - - evidence = materialize_claim_vs_evidence_evidence( - report, - archive_root=Path(report["archive_root"]), - now_ms=int(datetime.now(UTC).timestamp() * 1000), - ) + assert evidence is not None print(f"materialized evidence: {json.dumps(evidence, indent=2, sort_keys=True)}", file=sys.stderr) if parsed.json: sys.stdout.write(json.dumps(report, indent=2, sort_keys=True) + "\n") diff --git a/devtools/claim_vs_evidence_evidence.py b/devtools/claim_vs_evidence_evidence.py index 475aee44bd..af03d64e44 100644 --- a/devtools/claim_vs_evidence_evidence.py +++ b/devtools/claim_vs_evidence_evidence.py @@ -12,8 +12,9 @@ It writes through the same production primitives the daemon's own standing-query convergence stage uses -(``polylogue/daemon/convergence_standing_queries.py``) via -``open_daemon_connection`` -- not a new generic finding registry, not a +(``polylogue/daemon/convergence_standing_queries.py``), but only after taking +the archive's exclusive offline-writer lease -- not through a second live +SQLite writer, not a new generic finding registry, not a metric/pattern/cohort/experiment definition system, and not a scheduler. A finding is written with ``public_claim=None`` (no ``PublicClaimDeclaration``) unless the run's own construct-validity gates (``n_min``, non-zero classified @@ -36,6 +37,8 @@ from polylogue.core.query_identity import JsonValue from polylogue.core.query_identity import query_ref as _query_object_ref from polylogue.core.query_identity import result_set_ref as _result_set_object_ref +from polylogue.storage.archive_identity import archive_file_set_root +from polylogue.storage.index_generation import RebuildLease from polylogue.storage.sqlite.archive_tiers.user_write import ( ArchiveAssertionEnvelope, FindingAssertion, @@ -61,7 +64,7 @@ ANALYSIS_TARGET_REF = "analysis:claim-vs-evidence" _QUERY_GRAIN = "structured-failure-followup" _QUERY_LANE = "analysis" -_QUERY_RANK_POLICY = "origin,session_id,tool_id,tool_result_message_id" +_QUERY_RANK_POLICY = "origin,session_id,tool_id,tool_result_message_id,tool_result_block_id" class MaterializedEvidence(TypedDict): @@ -100,7 +103,7 @@ def build_query_definition(report: dict[str, Any]) -> dict[str, JsonValue]: def build_result_set_members(report: dict[str, Any]) -> tuple[str, ...]: - """Return the sorted ``message:`` refs the run actually classified.""" + """Return one sorted ``block:`` ref per failed outcome classified.""" return tuple(report["evidence"]["member_refs"]) @@ -222,18 +225,13 @@ def build_findings( ] -def materialize_claim_vs_evidence_evidence( +def materialize_claim_vs_evidence_evidence_under_lease( report: dict[str, Any], *, archive_root: Path, now_ms: int, ) -> MaterializedEvidence: - """Register one report run's query, result set, receipt, and findings. - - Writes through ``open_daemon_connection`` (the same connection helper the - daemon's own standing-query convergence stage uses), so this coexists - with the running daemon's single-writer discipline instead of bypassing - it with a bare ``sqlite3.connect``. + """Register one report while the caller holds ``RebuildLease``. The AnalysisDefinition (query) and its matched-row ResultSetManifest are content-addressed: identical selection logic and identical matched rows @@ -245,7 +243,16 @@ def materialize_claim_vs_evidence_evidence( should carry that a re-verification happened under a later tier state -- not silently collapse repeated regenerations into one row. """ - index_db = Path(report["index_db"]) + archive_root = archive_root.resolve() + report_root = Path(report["archive_root"]).resolve() + index_db = Path(report["index_db"]).resolve() + if report_root != archive_root: + raise ValueError(f"report archive root {report_root} does not match materialization root {archive_root}") + index_file_set_root = archive_file_set_root(archive_root=archive_root, db_path=index_db).resolve() + if index_file_set_root != archive_root: + raise ValueError( + f"report index {index_db} belongs to {index_file_set_root}, not materialization root {archive_root}" + ) query_definition = build_query_definition(report) member_refs = build_result_set_members(report) conn = open_daemon_connection(archive_root / "user.db", timeout=30.0) @@ -259,7 +266,19 @@ def materialize_claim_vs_evidence_evidence( created_at_ms=now_ms, ) query_reference = _query_object_ref(query.query_hash).format() - result_set_id = f"finding-{membership_merkle_root(member_refs)}" + corpus_epoch = _index_epoch(index_db) + result_set_digest = hash_payload( + ( + query.query_hash, + _QUERY_GRAIN, + corpus_epoch, + membership_merkle_root(member_refs), + hash_payload(list(member_refs)), + "capped", + "finding", + ) + ) + result_set_id = f"finding-{result_set_digest}" result_set: ResultSetManifest | None = get_result_set(conn, result_set_id) if result_set is None: result_set = put_result_set( @@ -267,7 +286,7 @@ def materialize_claim_vs_evidence_evidence( result_set_id=result_set_id, query_hash=query.query_hash, grain=_QUERY_GRAIN, - corpus_epoch=_index_epoch(index_db), + corpus_epoch=corpus_epoch, member_refs=member_refs, exactness="capped", persistence_class="finding", @@ -305,3 +324,25 @@ def materialize_claim_vs_evidence_evidence( "finding_assertion_ids": [envelope.assertion_id for envelope in envelopes], "public_claim_written": any(finding.public_claim is not None for finding in findings), } + + +def materialize_claim_vs_evidence_evidence( + report: dict[str, Any], + *, + archive_root: Path, + now_ms: int, +) -> MaterializedEvidence: + """Materialize an already-collected report under exclusive writer ownership. + + The CLI acquires this lease before report collection and calls the internal + under-lease function directly. This public helper remains safe for callers + that already hold a report: it excludes every live writer before opening + ``user.db`` and validates that the report's index and durable tiers belong + to the same archive file set. + """ + with RebuildLease(archive_root): + return materialize_claim_vs_evidence_evidence_under_lease( + report, + archive_root=archive_root, + now_ms=now_ms, + ) diff --git a/devtools/cli_surface_audit.py b/devtools/cli_surface_audit.py deleted file mode 100644 index 148a08f253..0000000000 --- a/devtools/cli_surface_audit.py +++ /dev/null @@ -1,379 +0,0 @@ -"""Capture a current-curated CLI surface audit demo.""" - -from __future__ import annotations - -import argparse -import json -import shutil -import sqlite3 -import subprocess -import sys -import time -from dataclasses import dataclass -from datetime import datetime -from pathlib import Path -from typing import TypedDict - - -@dataclass(frozen=True, slots=True) -class AuditCommand: - name: str - argv: tuple[str, ...] - notes: str - - -class ArchiveCounts(TypedDict): - schema_version: int | None - sessions: int | None - messages: int | None - - -class CommandRecord(TypedDict): - name: str - argv: list[str] - exit_code: int - duration_ms: float - stdout_bytes: int - stderr_bytes: int - stdout_lines: int - stderr_lines: int - json_top_level_bytes: dict[str, int] | None - notes: str - - -class AuditPayload(TypedDict): - generated_at: str - archive_root: str - archive: ArchiveCounts - include_unbounded_dialogue: bool - commands: list[CommandRecord] - - -DEFAULT_COMMANDS: tuple[AuditCommand, ...] = ( - AuditCommand("root_help", ("polylogue", "--help"), "Dense but coherent root help."), - AuditCommand("find_help", ("polylogue", "find", "--help"), "Short query workflow help."), - AuditCommand("read_help", ("polylogue", "read", "--help"), "Projection, delivery, and read-view options."), - AuditCommand( - "read_views_json", - ("polylogue", "read", "--views", "--format", "json"), - "Machine-discoverable read-view registry.", - ), - AuditCommand("status_plain", ("polylogue", "--plain", "status"), "Compact archive/daemon status."), - AuditCommand( - "ops_status_json", - ("polylogue", "--plain", "ops", "status", "--json"), - "Machine status payload.", - ), - AuditCommand( - "find_explain_read_json", - ( - "polylogue", - "--plain", - "--format", - "json", - "--explain", - "find", - "repo:polylogue", - "then", - "read", - "--view", - "messages", - "--limit", - "1", - ), - "Query explain JSON for a messages read.", - ), - AuditCommand( - "find_select_json", - ( - "polylogue", - "--plain", - "--format", - "json", - "find", - "repo:polylogue", - "then", - "select", - "--limit", - "3", - ), - "Bounded candidate identity selection.", - ), - AuditCommand( - "read_dialogue_bounded_json", - ( - "polylogue", - "--plain", - "find", - "repo:polylogue", - "then", - "read", - "--view", - "dialogue", - "--format", - "json", - "--limit", - "1", - "--max-tokens", - "120", - ), - "Projection-bounded dialogue payload with omission accounting.", - ), - AuditCommand( - "read_temporal_spec_json", - ( - "polylogue", - "--plain", - "find", - "repo:polylogue", - "then", - "read", - "--view", - "temporal,chronicle", - "--render", - "layout:context-image,timestamps:omit,format:json,destination:stdout", - "--projection", - "max-tokens:240,redact-paths:true,include-assertions:false", - "--spec", - "--limit", - "1", - ), - "Projection/render spec path with compact projection and render expressions.", - ), - AuditCommand( - "facets_json", - ("polylogue", "--plain", "facets", "--query", "repo:polylogue", "--format", "json"), - "Archive-backed facets payload.", - ), -) - -UNBOUNDED_DIALOGUE_COMMAND = AuditCommand( - "read_dialogue_unbounded_json", - ( - "polylogue", - "--plain", - "find", - "repo:polylogue", - "then", - "read", - "--view", - "dialogue", - "--format", - "json", - "--limit", - "1", - ), - "Opt-in diagnostic for the large unbounded dialogue payload.", -) - - -def _parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser( - prog="devtools workspace cli-surface-audit", - description="Capture a current-curated CLI surface audit demo.", - ) - parser.add_argument("--out-dir", type=Path, default=Path(".agent/demos/cli-surface-audit/current")) - parser.add_argument("--archive-root", type=Path, default=Path.home() / ".local/share/polylogue") - parser.add_argument("--include-unbounded-dialogue", action="store_true") - parser.add_argument("--timeout", type=int, default=60) - parser.add_argument("--json", action="store_true") - return parser - - -def _archive_counts(archive_root: Path) -> ArchiveCounts: - index_db = archive_root / "index.db" - if not index_db.exists(): - return {"schema_version": None, "sessions": None, "messages": None} - with sqlite3.connect(index_db) as conn: - version = int(conn.execute("PRAGMA user_version").fetchone()[0]) - sessions = int(conn.execute("SELECT COUNT(*) FROM sessions").fetchone()[0]) - messages = int(conn.execute("SELECT COUNT(*) FROM messages").fetchone()[0]) - return {"schema_version": version, "sessions": sessions, "messages": messages} - - -def _commands(include_unbounded_dialogue: bool) -> tuple[AuditCommand, ...]: - if include_unbounded_dialogue: - return (*DEFAULT_COMMANDS, UNBOUNDED_DIALOGUE_COMMAND) - return DEFAULT_COMMANDS - - -def _line_count(text: str) -> int: - return text.count("\n") + (1 if text and not text.endswith("\n") else 0) - - -def _json_top_level_bytes(text: str) -> dict[str, int] | None: - try: - payload = json.loads(text) - except json.JSONDecodeError: - return None - if isinstance(payload, list): - return { - "$array_bytes": len(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()), - "$array_items": len(payload), - } - if not isinstance(payload, dict): - return {"$scalar_bytes": len(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode())} - return { - key: len(json.dumps(value, sort_keys=True, separators=(",", ":")).encode()) - for key, value in sorted(payload.items()) - } - - -def _run_commands(commands: tuple[AuditCommand, ...], *, outdir: Path, timeout: int) -> list[CommandRecord]: - summary: list[CommandRecord] = [] - for command in commands: - started = time.perf_counter() - proc = subprocess.run(command.argv, text=True, capture_output=True, timeout=timeout) - duration_ms = round((time.perf_counter() - started) * 1000, 3) - (outdir / f"{command.name}.stdout").write_text(proc.stdout, encoding="utf-8") - (outdir / f"{command.name}.stderr").write_text(proc.stderr, encoding="utf-8") - summary.append( - { - "name": command.name, - "argv": list(command.argv), - "exit_code": proc.returncode, - "duration_ms": duration_ms, - "stdout_bytes": len(proc.stdout.encode()), - "stderr_bytes": len(proc.stderr.encode()), - "stdout_lines": _line_count(proc.stdout), - "stderr_lines": _line_count(proc.stderr), - "json_top_level_bytes": _json_top_level_bytes(proc.stdout), - "notes": command.notes, - } - ) - return summary - - -def _json_breakdown_section(commands: list[CommandRecord]) -> str: - large_json_commands = [ - command for command in commands if command["json_top_level_bytes"] and command["stdout_bytes"] >= 8192 - ] - if not large_json_commands: - return "" - sections = ["## Large JSON Payloads", ""] - for command in large_json_commands: - top_level_bytes = command["json_top_level_bytes"] - if top_level_bytes is None: - continue - top = sorted( - top_level_bytes.items(), - key=lambda item: item[1], - reverse=True, - )[:8] - sections.append(f"### `{command['name']}`") - sections.append("") - sections.append(f"- stdout: {command['stdout_bytes']} bytes") - sections.append("- top-level sections:") - for name, size in top: - sections.append(f" - `{name}`: {size} bytes") - sections.append("") - return "\n".join(sections) - - -def _readme(*, generated_at: str, archive_root: Path, archive: ArchiveCounts, commands: list[CommandRecord]) -> str: - rows = [ - "| Command | Exit | Duration | Size | Notes |", - "| --- | ---: | ---: | ---: | --- |", - ] - for command in commands: - rows.append( - "| `{}` | {} | {:.0f} ms | {} bytes | {} |".format( - " ".join(str(part) for part in command["argv"]), - command["exit_code"], - command["duration_ms"], - command["stdout_bytes"], - command["notes"], - ) - ) - matrix = "\n".join(rows) - json_breakdown = _json_breakdown_section(commands) - return f"""# CLI Surface Audit - -Generated: {generated_at} -Archive root: `{archive_root}` -Archive state: index schema v{archive["schema_version"]}, {archive["sessions"]} sessions, {archive["messages"]} messages - -## What This Proves - -This artifact is a live command audit over representative Polylogue CLI -surfaces: help, read-view discovery, daemon/archive status, query explain, -select, bounded dialogue read, temporal spec, and facets. It is not a complete -CLI certification; it is a dogfood slice that turns real command behavior into -concrete product work. - -Raw command outputs are in `outputs/`. Timings, byte counts, exit codes, and -command notes are in `command-matrix.json`. - -## Command Matrix - -{matrix} - -{json_breakdown} - -## Current Curation Policy - -The default audit intentionally excludes the unbounded dialogue JSON export. -That diagnostic remains available with `--include-unbounded-dialogue`, but the -current demo shelf should prefer bounded dialogue output because it demonstrates -the product path an operator should actually use for large devloop sessions. - -## Regeneration - -```bash -devtools workspace cli-surface-audit --out-dir .agent/demos/cli-surface-audit/current -``` -""" - - -def run_audit( - *, - out_dir: Path, - archive_root: Path, - include_unbounded_dialogue: bool, - timeout: int, -) -> AuditPayload: - out_dir = out_dir.expanduser() - outputs = out_dir / "outputs" - if outputs.exists(): - shutil.rmtree(outputs) - outputs.mkdir(parents=True, exist_ok=True) - generated_at = datetime.now().astimezone().strftime("%Y-%m-%d %H:%M:%S %Z") - archive = _archive_counts(archive_root) - commands = _run_commands(_commands(include_unbounded_dialogue), outdir=outputs, timeout=timeout) - payload: AuditPayload = { - "generated_at": generated_at, - "archive_root": str(archive_root), - "archive": archive, - "include_unbounded_dialogue": include_unbounded_dialogue, - "commands": commands, - } - out_dir.mkdir(parents=True, exist_ok=True) - (out_dir / "command-matrix.json").write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8") - (out_dir / "README.md").write_text( - _readme(generated_at=generated_at, archive_root=archive_root, archive=archive, commands=commands), - encoding="utf-8", - ) - return payload - - -def main(argv: list[str] | None = None) -> int: - args = _parser().parse_args(argv) - try: - payload = run_audit( - out_dir=args.out_dir, - archive_root=args.archive_root, - include_unbounded_dialogue=args.include_unbounded_dialogue, - timeout=args.timeout, - ) - except (OSError, sqlite3.Error, subprocess.SubprocessError) as exc: - print(f"cli-surface-audit: {exc}", file=sys.stderr) - return 2 - if args.json: - print(json.dumps(payload, indent=2, sort_keys=True)) - else: - print(f"captured {len(payload['commands'])} CLI command(s) under {args.out_dir}") - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/click_dispatch.py b/devtools/click_dispatch.py index 371b0d5d48..5d469378c6 100644 --- a/devtools/click_dispatch.py +++ b/devtools/click_dispatch.py @@ -179,10 +179,8 @@ def callback(args: tuple[str, ...], json_flag: bool = False, inner_help: bool = callback=callback, params=params, ) - # Subcommands that use argparse internally need unknown options forwarded - # as-is rather than rejected by Click's option parser. This allows - # modules like devtools/task_history.py with their own sub-subcommands and - # --flags to work transparently. + # Subcommands use argparse internally, so unknown options must be forwarded + # as-is rather than rejected by Click's option parser. cmd.allow_extra_args = True cmd.ignore_unknown_options = True return cmd @@ -270,18 +268,7 @@ def _dispatch(argv: list[str]) -> int: def main(argv: list[str] | None = None) -> int: - """Entry point for programmatic use of the Click-based devtools CLI. - - Converts argv to Click invocation and returns the exit code. Every - invocation appends a JSONL record to ``.agent/task-history/tasks.jsonl`` so - agent task history is self-populating (see ``devtools workspace tasks``). Set - ``POLYLOGUE_TASK_HISTORY_DISABLE=1`` to opt out (also suppressed during a - ``devtools workspace tasks replay`` to avoid double-logging the outer wrapper). - """ - import os - import time - - from devtools import task_history as task_history_mod + """Entry point for programmatic use of the Click-based devtools CLI.""" try: assert_polylogue_matches_checkout(_REPO_ROOT, context="devtools") @@ -289,34 +276,4 @@ def main(argv: list[str] | None = None) -> int: sys.stderr.write(f"{exc}\n") return 125 - args_list = list(argv or []) - if not args_list or args_list[0].startswith("-"): - # Bare invocation or root option only — skip auto-log. - return _dispatch(args_list) - - command_name = args_list[0] - inner_args = args_list[1:] - for spec in sorted(COMMAND_SPECS, key=lambda item: len(item.command_path), reverse=True): - path = spec.command_path - if tuple(args_list[: len(path)]) == path: - command_name = " ".join(path) - inner_args = args_list[len(path) :] - break - - if task_history_mod.auto_log_disabled(): - return _dispatch(args_list) - - started = time.perf_counter() - exit_code = 0 - try: - exit_code = _dispatch(args_list) - return exit_code - finally: - duration_ms = (time.perf_counter() - started) * 1000.0 - task_history_mod.record_invocation( - command=command_name, - args=inner_args, - duration_ms=duration_ms, - exit_code=exit_code, - cwd=os.getcwd(), - ) + return _dispatch(list(argv or [])) diff --git a/devtools/command_catalog.py b/devtools/command_catalog.py index f6d46cf21e..a6c690501b 100644 --- a/devtools/command_catalog.py +++ b/devtools/command_catalog.py @@ -9,43 +9,6 @@ CommandMain = Callable[[list[str] | None], int] CONTROL_PLANE = "devtools" -VERIFICATION_LAB_COMMAND_NAMES: tuple[str, ...] = ( - "lab graph", - "lab lanes", - "lab policy backlog-hygiene", - "lab policy bead-graph", - "lab policy acceptance-contracts", - "lab policy acceptance-contract-reconcile", - "lab policy campaign-archive-boundaries", - "lab policy demo-packet-registry", - "lab policy demo-tour-freshness", - "lab policy docs-drift", - "lab policy insight-honesty", - "lab policy schema-versioning", - "lab policy timestamp-doctrine", - "lab provider completeness", - "lab probe capture-regression", - "lab probe cost-reconciliation", - "lab probe pipeline", - "lab probe turso", - "lab projections", - "lab run", - "lab smoke", - "lab schema audit", - "lab schema commit", - "lab schema compare", - "lab schema explain", - "lab schema generate", - "lab schema list", - "lab schema parser-diff", - "lab schema promote", - "lab schema roundtrip", - "lab seed-receipt-compare", - "lab snapshot read-surface", - "lab test-economics", - "lab testmon-blind-spots", - "lab testmon-proof", -) CATEGORY_ORDER: tuple[str, ...] = ( "core", @@ -102,26 +65,6 @@ def to_dict(self) -> dict[str, object]: return data -@dataclass(frozen=True, slots=True) -class WorkspaceCommandDisposition: - name: str - disposition: str - evidence: str - replacement: str - replacement_command: str | None = None - - -@dataclass(frozen=True, slots=True) -class CatalogBypassSite: - path: str - marker: str - command_name: str | None - disposition: str - reason: str - occurrence_line: int | None = None - expected_occurrences: int = 0 - - COMMAND_SPECS: tuple[CommandSpec, ...] = ( CommandSpec( "status", @@ -217,29 +160,12 @@ class CatalogBypassSite: "Render the command catalog inside docs/devtools.md.", "devtools.render_devtools_reference", ), - CommandSpec( - "render demo-corpus-datasheet", - "generated surfaces", - "Render docs/plans/demo-corpus-construct-audit.md from the demo family registry and a measured seed archive.", - "devtools.render_demo_corpus_datasheet", - use_when="Refresh or verify the deterministic demo corpus construct datasheet after changing demo families, constructs, or seed semantics.", - examples=( - "devtools render demo-corpus-datasheet", - "devtools render demo-corpus-datasheet --check", - ), - ), CommandSpec( "render docs-surface", "generated surfaces", "Render docs/README.md and the README documentation table.", "devtools.render_docs_surface", ), - CommandSpec( - "render quality-reference", - "generated surfaces", - "Render docs/test-quality-workflows.md from executable lane, mutation, and benchmark registries.", - "devtools.render_quality_reference", - ), CommandSpec( "render query-discovery", "generated surfaces", @@ -251,69 +177,6 @@ class CatalogBypassSite: ), examples=("devtools render query-discovery", "devtools render query-discovery --check"), ), - CommandSpec( - "render api-operation-parity", - "generated surfaces", - "Render the committed semantic-operation parity matrix and Python facade reference.", - "devtools.render_api_operation_parity", - use_when=( - "Refresh or verify stable Python API operation IDs, cross-surface bindings, intentional exclusions, " - "and the signature-aware generated section in docs/library-api.md." - ), - examples=("devtools render api-operation-parity", "devtools render api-operation-parity --check"), - ), - CommandSpec( - "render mcp-equivalence", - "generated surfaces", - "Render docs/generated/mcp-equivalence.json from executable MCP declarations.", - "devtools.render_mcp_equivalence", - use_when=( - "Refresh or verify MCP discovery names, input/output contracts, role gates, operation owners, " - "Python parity expectations, and disjoint migration ownership after changing the compatibility surface." - ), - examples=("devtools render mcp-equivalence", "devtools render mcp-equivalence --check"), - ), - CommandSpec( - "render mcp-tool-index", - "generated surfaces", - "Render the generated exhaustive tool-name appendix into docs/mcp-reference.md.", - "devtools.render_mcp_tool_index", - use_when=( - "Keep every registered MCP tool name individually reachable from the docs tree " - "(tests/infra/mcp.py:EXPECTED_TOOL_NAMES) after adding or removing a tool, so " - "`devtools verify docs-coverage` stays clean without hand-duplicating the list." - ), - examples=("devtools render mcp-tool-index", "devtools render mcp-tool-index --check"), - ), - CommandSpec( - "render product-workflows", - "generated surfaces", - "Render docs/product/workflows.md from executable query-action workflow registries.", - "devtools.render_product_workflows", - use_when=( - "Refresh or verify the product query-action workflow contract after changing action contracts, " - "read-view surfaces, completion behavior, or workflow golden paths (#2305)." - ), - examples=( - "devtools render product-workflows", - "devtools render product-workflows --check", - ), - ), - CommandSpec( - "render public-claims", - "generated surfaces", - "Render README, launch, findings-page, and verified-export claim views from FINDING assertions.", - "devtools.render_public_claims", - use_when=( - "Refresh public claim Markdown/JSON and the generated YAML compatibility view after changing " - "FINDING seeds, judgments, capability declarations, or 37t.14 verdict receipts." - ), - examples=( - "devtools render public-claims", - "devtools render public-claims --check", - "devtools render public-claims --archive-root /path/to/archive --verdicts /path/to/verdicts.json", - ), - ), CommandSpec( "render pages", "generated surfaces", @@ -331,7 +194,6 @@ class CatalogBypassSite: examples=( "devtools render visual-tapes", "devtools render visual-tapes --capture", - "devtools render visual-tapes --check", ), ), CommandSpec( @@ -344,18 +206,20 @@ class CatalogBypassSite: featured=True, ), CommandSpec( - "verify public-claims", + "verify ci-commands", "verification", - "Verify generated public-claim views, preset parity, sanitized refs, coverage markers, and retired copy.", - "devtools.public_claims", - use_when=( - "Check externally visible claims after changing README, demos, findings, FINDING assertions, " - "capability declarations, or evidence-integrity receipts." - ), - examples=( - "devtools verify public-claims", - "devtools verify public-claims --json", - ), + "Validate devtools invocations in structured CI run fields.", + "devtools.verify_ci_commands", + use_when="Catch CI scripts that reference a removed or misspelled devtools command.", + examples=("devtools verify ci-commands", "devtools verify ci-commands --json"), + ), + CommandSpec( + "verify doc-commands", + "verification", + "Validate executable documentation examples against live command inventories.", + "devtools.verify_doc_commands", + use_when="Catch README and documentation examples that reference an unknown command path or flag.", + examples=("devtools verify doc-commands", "devtools verify doc-commands --json"), ), CommandSpec( "verify corpus-fidelity", @@ -387,21 +251,6 @@ class CatalogBypassSite: "--ground-truth-root codex-session=/path/to/codex --receipt /path/to/schema-inference-gate-receipt.json --json", ), ), - CommandSpec( - "release readiness", - "release", - "Validate the externally-presentable release gate definition.", - "devtools.release_readiness", - use_when=( - "Check that the release-readiness gate document, required local commands, " - "and release PR evidence template are still coherent before touching a release PR." - ), - examples=( - "devtools release readiness", - "devtools release readiness --json", - "devtools release readiness --release-body-file /tmp/release-pr-body.md", - ), - ), CommandSpec( "lab provider completeness", "verification lab", @@ -431,17 +280,6 @@ class CatalogBypassSite: ), featured=True, ), - CommandSpec( - "reindex-canary", - "verification", - "Run the product's representative inactive-generation reindex canary.", - "devtools.reindex_canary", - use_when="Exercise the bounded semantic reindex canary before paying for a full rebuild.", - examples=( - "devtools reindex-canary --archive-root /path/to/isolated-archive --input /path/to/index.db --schema-inference-receipt /path/to/schema-inference-gate-receipt.json --sample 100 --report /path/to/canary.json --no-promote", - "devtools reindex-canary --archive-root /path/to/isolated-archive --schema-inference-receipt /path/to/schema-inference-gate-receipt.json --pathology-session-id codex-session:whale --report /path/to/canary.json --no-promote", - ), - ), CommandSpec( "verify coverage", "verification", @@ -457,7 +295,7 @@ class CatalogBypassSite: CommandSpec( "verify mutation-freshness", "verification", - "Verify fresh mutation campaigns meet their declared kill-rate thresholds.", + "Verify executable mutation campaigns meet the selected freshness and kill-rate thresholds.", "devtools.verify_mutation_freshness", use_when=( "Enforce mutation campaign freshness and kill-rate thresholds after a rotating CI campaign " @@ -465,47 +303,18 @@ class CatalogBypassSite: ), examples=( "devtools verify mutation-freshness --enforce-kill-rate", - "devtools verify mutation-freshness --yaml .local/mutation-campaigns/index.yaml --strict", - ), - ), - CommandSpec( - "lab lanes", - "verification lab", - "Run named validation lanes.", - "devtools.run_validation_lanes", - use_when="List, dry-run, or execute authored validation lanes from the executable lane registry.", - examples=( - "devtools lab lanes --list", - "devtools lab lanes --lane frontier-local", - "devtools lab lanes --lane live-archive-smoke --dry-run", + "devtools verify mutation-freshness --strict --default-freshness-days 30", ), ), CommandSpec( "lab graph", "verification lab", - "Render the runtime artifact, operation, and scenario-coverage map.", + "Render the runtime artifact and operation graph.", "devtools.artifact_graph", - use_when="Inspect the authored runtime graph and see which scenarios currently cover declared artifacts and operations.", + use_when="Inspect declared runtime artifacts, operations, paths, and maintenance targets.", examples=( "devtools lab graph", "devtools lab graph --json", - "devtools lab graph --strict", - ), - ), - CommandSpec( - "lab test-economics", - "verification lab", - "Report per-package coverage/fix-density/test-cost economics (polylogue-9e5.11).", - "devtools.test_economics_report", - use_when=( - "Decide where test-writing effort or test-suite pruning actually pays off, by " - "cross-referencing coverage percent, historical fix-commit density, testmon " - "wall-time cost exposure, and testmon selection fan-out per top-level package." - ), - examples=( - "devtools lab test-economics", - "devtools lab test-economics --json", - "devtools lab test-economics --write docs/test-economics.md", ), ), CommandSpec( @@ -520,53 +329,6 @@ class CatalogBypassSite: ), examples=("devtools lab testmon-proof", "devtools lab testmon-proof --json"), ), - CommandSpec( - "lab testmon-blind-spots", - "verification lab", - "Audit coverage-known files that are absent from the testmon fingerprint graph.", - "devtools.testmon_blind_spot_audit", - use_when=( - "Inspect an existing coverage JSON report against an existing pytest-testmon database. " - "Declaration-only modules are reported separately from executable validator risk; this command " - "does not run pytest or regenerate coverage." - ), - examples=( - "devtools lab testmon-blind-spots", - "devtools lab testmon-blind-spots --json", - "devtools lab testmon-blind-spots --coverage-json path/to/coverage.json --testmon-db path/to/testmondata", - ), - ), - CommandSpec( - "lab pytest-witness-repetitions", - "verification lab", - "Repeat the exact optimize, WAL, and embedding seed-hang witnesses with durable receipts.", - "devtools.pytest_witness_repetitions", - use_when=( - "Establish that the historical periodic optimize, WAL checkpoint, and embedding backlog " - "lifecycle witnesses survive consecutive isolated and xdist runs. Each attempt uses the ordinary " - "managed pytest/supervisor path; failures and timeouts are retained rather than retried." - ), - examples=( - "devtools lab pytest-witness-repetitions", - "devtools lab pytest-witness-repetitions --attempts 2 --xdist-workers 3 --json", - ), - ), - CommandSpec( - "lab seed-receipt-compare", - "verification lab", - "Compare two workload receipts for a clean, like-for-like seed/incident proof (polylogue-b054.1.1.3).", - "devtools.seed_receipt_compare", - use_when=( - "Judge a seed-testmon or focused pytest run's resource receipt against a named baseline " - "receipt (e.g. a prior incident) — confirms both runs share workload identity and terminated " - "cleanly, then scores wall-time speedup and peak-PSS ceiling targets, naming a blocker and " - "linked follow-up for any unmet target instead of silently dropping it." - ), - examples=( - "devtools lab seed-receipt-compare --baseline incident.json --candidate current.json", - "devtools lab seed-receipt-compare --baseline incident.json --candidate current.json --json", - ), - ), CommandSpec( "bench ingest-amplification", "benchmarking", @@ -603,13 +365,6 @@ class CatalogBypassSite: "devtools bench ingest-throughput --batches 20 --seed 2391", ), ), - CommandSpec( - "bench coordination-latency", - "benchmarking", - "Measure compact coordination status p50/p95 with raw stage samples.", - "devtools.coordination_latency_probe", - examples=("devtools bench coordination-latency --samples 21 --out .local/coordination-latency.json",), - ), CommandSpec( "lab snapshot read-surface", "verification lab", @@ -657,40 +412,6 @@ class CatalogBypassSite: "devtools workspace worktree-gc --apply --force", ), ), - CommandSpec( - "workspace backlog-calibration", - "workspace", - "Measured lead-time/discovery/staleness distributions over the bead corpus.", - "devtools.backlog_calibration", - use_when=( - "Re-fit the numbers a backlog-execution plan rests on instead of guessing them: " - "closed-bead lead-time percentiles split by priority/type/epic-membership/dependency " - "degree, a censoring-honest survival view (closed-only medians are survivorship-" - "biased), close-reason classification measuring how much of the backlog closes with " - "no implementation (already-satisfied/obsolete/duplicate), and created-vs-closed " - "discovery dynamics (does the backlog drain?). Optionally calibrates PR open->merge " - "latency by size from a gh dump. Answers a different question than " - "`workspace beads-state-report` (population shape and graph hygiene, point-in-time) " - "and `workspace bead-cluster` (what to batch next): this is duration *models* over " - "history, meant to be re-run as the corpus grows so plans stop quoting stale guesses." - ), - examples=( - "devtools workspace backlog-calibration", - "devtools workspace backlog-calibration --input beads.jsonl --json", - "devtools workspace backlog-calibration --prs prs.json", - ), - ), - CommandSpec( - "workspace bead-batch-show", - "workspace", - "Batch-show beads: id, status, prio, title, desc head, deps, notes tail.", - "devtools.bead_batch_show", - use_when=( - "Skim several beads at once (e.g. a fanout cluster or a discovered-follow-up batch) " - "without a separate `bd show` round-trip per id." - ), - examples=("devtools workspace bead-batch-show polylogue-kapb polylogue-2yax",), - ), CommandSpec( "workspace bead-cluster", "workspace", @@ -701,10 +422,7 @@ class CatalogBypassSite: "footprint (cluster into one branch/PR sweep) vs. which are genuinely disjoint " "(safe to run as parallel worktree lanes), and flag contention -- beads that touch " "the same durable migration slot or the same generated surface even without an " - "exact file-path overlap. Answers a different question than " - "`workspace delivery-gate-status` (gate-progress board): this is footprint/overlap " - "clustering over the same live bead data, not gate percentage. Footprints are " - "advisory -- verify on claim." + "exact file-path overlap. Footprints are advisory, so verify them when claiming work." ), examples=( "devtools workspace bead-cluster", @@ -714,25 +432,6 @@ class CatalogBypassSite: "devtools workspace bead-cluster --input ready.json --validate-roster", ), ), - CommandSpec( - "workspace lane-brief", - "workspace", - "Generate a dispatch brief for a bead lane with live footprint/prior-art evidence.", - "devtools.lane_brief", - use_when=( - "Before dispatching a batch of bead ids as one lane/branch to a subagent, produce a " - "markdown brief carrying the full bead records plus LIVE evidence: every file path " - "mentioned in the bead text is checked against the working tree (exists? line count? " - "last 3 commits?) so a dispatcher does not hand a subagent stale bead prose, and closed " - "beads mentioning the same paths are surfaced as prior art. Sections the tool cannot " - "fill (measured baseline, non-goals) are emitted as explicit placeholders, never " - "silently dropped." - ), - examples=( - "devtools workspace lane-brief polylogue-abc polylogue-def", - "devtools workspace lane-brief polylogue-ei94 --out .agent/scratch/lane-ei94.md", - ), - ), CommandSpec( "workspace lane-init", "workspace", @@ -793,7 +492,7 @@ class CatalogBypassSite: CommandSpec( "workspace merge-gate", "workspace", - "Structural pre-merge safety check: fresh local-verification receipt + no late review comments.", + "Structural pre-merge safety check: fresh local verification + resolved review threads.", "devtools.merge_gate", use_when=( "Immediately before squash-merging any PR in a merge train, replacing coordinator memory " @@ -801,17 +500,17 @@ class CatalogBypassSite: 'per-PR) with a check that fails closed. `record --command "..."` requires the current ' "checkout to already be the PR's exact head commit with a clean tree (it refuses otherwise), " "first validates the same versioned PR-scope carrier CircleCI checks, then runs a local verification " - "command, and persists a receipt flagging commands that look like they skip tests (e.g. " - "`verify --quick`). `check ` polls review comments across a real grace window (default " - "3x20s, covering CodeRabbit's 30-60s late-arrival window) and BLOCKs unless a receipt exists " - "for the CURRENT head sha within a freshness window with exit_code 0, and no review comment's " - "created_at is newer than the head commit's timestamp unless explicitly `ack`'d for that exact " - "head sha. The receipt binds the carrier digest plus a fresh head- and Bead-bound scope attestation, so a changed scope or Bead state requires re-recording. Motivated by two 2026-08-01 incidents: PR #3502 merged before CodeRabbit's findings " + "command and persists its typed verification scope. `check ` polls structured GitHub " + "review-thread state across a real grace window (default 3x20s, covering CodeRabbit's " + "30-60s late-arrival window) and BLOCKs unless a receipt exists for the CURRENT head sha " + "within a freshness window with exit_code 0, every review thread is resolved, and GitHub's " + "review decision is not CHANGES_REQUESTED. The receipt binds the carrier digest plus a fresh " + "head- and Bead-bound scope attestation, so a changed scope or Bead state requires re-recording. " + "Motivated by two 2026-08-01 incidents: PR #3502 merged before CodeRabbit's findings " "posted, and PR #3517 nearly merged with a 43-test regression no CI check or review comment " "ever flagged -- plus review findings on this tool itself (recording from an unrelated " - "checkout, a --quick example that would have missed its own motivating regression, a single " - "comment snapshot instead of a grace-period poll, and no way to triage a false-positive late " - "comment without an empty commit). `check --post-status` (polylogue-1cbeh) posts the " + "checkout, a --quick example that would have missed its own motivating regression, and a single " + "review snapshot instead of a grace-period poll). `check --post-status` (polylogue-1cbeh) posts the " "verdict as a GitHub commit status (`context=merge-gate`, success/failure) on the PR's " "current head sha via `gh api repos/{owner}/{repo}/statuses/{sha}` -- this is what lets " "branch protection or `gh pr merge --auto` gate on the same verdict instead of a " @@ -822,7 +521,6 @@ class CatalogBypassSite: "devtools workspace merge-gate check 3517", "devtools workspace merge-gate check 3517 --json --max-age-s 7200 --poll-rounds 1", "devtools workspace merge-gate check 3517 --post-status", - 'devtools workspace merge-gate ack 3517 123456789 --reason "already fixed upstream, false positive"', ), ), CommandSpec( @@ -854,26 +552,6 @@ class CatalogBypassSite: 'devtools workspace merge record-full-verify --command "devtools verify --all"', ), ), - CommandSpec( - "workspace merge-conductor", - "workspace", - "Mechanical-conflict triage for the PR merge train (dry-run by default).", - "devtools.merge_conductor", - use_when=( - "Before or during a merge train, classify each roster PR's conflicting/modified files " - "into AUTO-RESOLVABLE mechanical classes (.beads/issues.jsonl take-master, regenerable " - "surfaces) vs ESCALATE classes (schema migrations, hooks config, anything else) and " - "detect cross-PR contention (two PRs claiming the same migration slot or generated-" - "surface family). Implements the mechanical slice of the polylogue-ei94 merge-conductor " - "design. --execute only touches AUTO-RESOLVABLE PRs, in a scratch worktree, and aborts " - "back to ESCALATE on any deviation; escalate-class PRs are never touched under --execute." - ), - examples=( - "devtools workspace merge-conductor --pr 3301 --pr 3302", - "devtools workspace merge-conductor --pr 3301 --json", - "devtools workspace merge-conductor --pr 3301 --execute", - ), - ), CommandSpec( "workspace bead-reimport-guard", "workspace", @@ -894,24 +572,6 @@ class CatalogBypassSite: "devtools workspace bead-reimport-guard export /tmp/issues-snapshot.jsonl", ), ), - CommandSpec( - "workspace delivery-gate-status", - "workspace", - "Per-release-gate progress board over .beads/issues.jsonl (delivery: / lane: overlay).", - "devtools.delivery_gate_status", - use_when=( - "Check overall delivery-gate progress (R0-normalize through N-horizon) -- percent " - "complete, ready/blocked/in-progress counts, and the active frontier gate's exit " - "criterion. --fresh re-exports .beads/issues.jsonl first since bd updates do not " - "immediately re-export." - ), - examples=( - "devtools workspace delivery-gate-status", - "devtools workspace delivery-gate-status --fresh", - "devtools workspace delivery-gate-status --json", - "devtools workspace delivery-gate-status --gate delivery:A-trust-floor", - ), - ), CommandSpec( "demo real-slice-screen", "workspace", @@ -957,22 +617,6 @@ class CatalogBypassSite: "devtools workspace dev-loop --inspect-run .cache/dev-loop/ --json", ), ), - CommandSpec( - "workspace frontier", - "workspace", - "Derive a complete, non-mutating execution focus from live Beads state.", - "devtools.frontier_report", - use_when=( - "During Direction, Velocity, or wait-ahead windows, distinguish the full ambition, admitted " - "active set, current claims, dependency-ready work, and resource-permitted execution focus before " - "claiming or dispatching work." - ), - examples=( - "devtools workspace frontier", - "devtools workspace frontier --json", - "devtools workspace frontier --out .agent/task-history/frontier-latest.md", - ), - ), CommandSpec( "workspace deployment-smoke", "workspace", @@ -1072,20 +716,6 @@ class CatalogBypassSite: "devtools workspace raw-live-source-reconciliation --limit 500 --sample-limit 20", ), ), - CommandSpec( - "workspace chatgpt-lifecycle-anchor-audit", - "workspace", - "Census the current quarantined ChatGPT corpus for lifecycle-anchor conflicts.", - "devtools.chatgpt_lifecycle_anchor_audit", - use_when=( - "Run the read-only parser-to-classifier census behind the historical ChatGPT mapping-order defect. " - "It emits only aggregate, sanitized evidence and never reclassifies source rows or graph gates." - ), - examples=( - "devtools workspace chatgpt-lifecycle-anchor-audit --archive-root /path/to/archive", - "devtools workspace chatgpt-lifecycle-anchor-audit --archive-root /path/to/archive --receipt docs/audits/receipt.json", - ), - ), CommandSpec( "workspace raw-live-source-reconciliation-apply", "workspace", @@ -1496,21 +1126,6 @@ class CatalogBypassSite: "devtools workspace temporal-read-profile --query 'repo:polylogue devloop' --limit 3 --out .local/temporal-profile.json", ), ), - CommandSpec( - "workspace temporal-devloop", - "workspace", - "Compose git and operating-log events into a temporal evidence window.", - "devtools.devloop_temporal", - use_when=( - "Dogfood temporal analysis on the current devloop without inventing a bespoke report shape: " - "git commits and OPERATING-LOG headings are normalized as event families and projected through " - "the shared TemporalEvidenceWindow." - ), - examples=( - "devtools workspace temporal-devloop --since 2026-06-30T00:00:00+02:00 --json", - "devtools workspace temporal-devloop --out .agent/demos/14-devloop-temporal-dogfood/devloop-events.json", - ), - ), CommandSpec( "workspace temporal-archive-aggregates", "workspace", @@ -1523,7 +1138,7 @@ class CatalogBypassSite: ), examples=( "devtools workspace temporal-archive-aggregates --json", - "devtools workspace temporal-archive-aggregates --out-dir .agent/demos/01-real-archive-temporal-devloops", + "devtools workspace temporal-archive-aggregates --out-dir .local/temporal-archive-aggregates", ), ), CommandSpec( @@ -1539,7 +1154,7 @@ class CatalogBypassSite: examples=( "devtools workspace lineage-validation --json", "devtools workspace lineage-validation --sample-prefix-sharing 100 --json", - "devtools workspace lineage-validation --out-dir .agent/demos/lineage-validation/current", + "devtools workspace lineage-validation --out-dir .local/evidence/lineage-validation/current", ), ), CommandSpec( @@ -1554,24 +1169,22 @@ class CatalogBypassSite: examples=( "devtools workspace affordance-usage --days 7 --json", "devtools workspace affordance-usage --detail-pattern codebase-memory --detail-pattern search_code --days 30", - "devtools workspace affordance-usage --out-dir .agent/demos/agent-affordance-usage", + "devtools workspace affordance-usage --out-dir .local/evidence/agent-affordance-usage", ), ), CommandSpec( - "workspace demo-shelf", + "workspace claim-vs-evidence", "workspace", - "Refresh or verify current demo shelf indexes.", - "devtools.demo_shelf", + "Analyze structured failures and the assistant behavior that followed.", + "devtools.claim_vs_evidence", use_when=( - "After curating .agent/demos or another explicit current demo shelf, regenerate " - "MANIFEST.readable.json and SUMMARY_INDEX.json from one deterministic helper. " - "Declarative read packages own portable readable bundles; the shelf is the best current set, " - "not an append-only archive." + "Measure bounded, origin-stratified follow-up behavior from structural tool-result failures; " + "the report preserves ambiguous outcomes, calibration, sensitivity windows, and separate " + "usage/cost lanes instead of treating prose as the failure oracle." ), examples=( - "devtools workspace demo-shelf", - "devtools workspace demo-shelf --check", - "devtools workspace demo-shelf --root /realm/project/sinex/.agent/demos --json", + "devtools workspace claim-vs-evidence --json", + "devtools workspace claim-vs-evidence --limit 5000 --out-dir .local/evidence/claim-vs-evidence", ), ), CommandSpec( @@ -1586,37 +1199,7 @@ class CatalogBypassSite: ), examples=( "devtools workspace degraded-archive-proof --json", - "devtools workspace degraded-archive-proof --out-dir .agent/demos/degraded-archive-proof/current --json", - ), - ), - CommandSpec( - "workspace cli-surface-audit", - "workspace", - "Capture a current-curated CLI surface audit demo.", - "devtools.cli_surface_audit", - use_when=( - "Refresh the CLI surface audit shelf through one reusable command that captures representative " - "help, status, query, read, and facets outputs while keeping large unbounded diagnostics opt-in." - ), - examples=( - "devtools workspace cli-surface-audit", - "devtools workspace cli-surface-audit --out-dir .agent/demos/cli-surface-audit/current --json", - "devtools workspace cli-surface-audit --include-unbounded-dialogue", - ), - ), - CommandSpec( - "workspace claim-vs-evidence", - "workspace", - "Build a structured failure follow-up claim-vs-evidence demo.", - "devtools.claim_vs_evidence", - use_when=( - "Produce a fast, bounded report over structured tool failures and the immediately following " - "assistant turn, using tool_result is_error/exit_code as the evidence anchor instead of " - "prose-mined outcome claims." - ), - examples=( - "devtools workspace claim-vs-evidence --json", - "devtools workspace claim-vs-evidence --limit 5000 --out-dir .agent/demos/claim-vs-evidence", + "devtools workspace degraded-archive-proof --out-dir .local/evidence/degraded-archive-proof/current --json", ), ), CommandSpec( @@ -1634,18 +1217,6 @@ class CatalogBypassSite: "devtools workspace read-package --spec package.yaml --session-id 019f... --out-dir product-read --dry-run --json", ), ), - CommandSpec( - "lab projections", - "verification lab", - "Render the authored scenario-bearing verification projections.", - "devtools.scenario_projections", - use_when="Inspect the unified projection inventory that feeds runtime coverage, generated docs, and control-plane maps.", - examples=( - "devtools lab projections", - "devtools lab projections --source-kind validation-lane --artifact-target session_insight_rows", - "devtools lab projections --json", - ), - ), CommandSpec( "verify agent-integration", "verification", @@ -1658,17 +1229,6 @@ class CatalogBypassSite: "devtools verify agent-integration --require-live", ), ), - CommandSpec( - "verify closure-matrix", - "verification", - "Verify docs/plans/test-closure-matrix.yaml stays grounded in the realized tree.", - "devtools.verify_closure_matrix", - use_when=( - "Keep the per-domain test-closure matrix honest — fails when a declared target file or " - "representative test path is missing, or when a row violates the gate schema." - ), - examples=("devtools verify closure-matrix", "devtools verify closure-matrix --json"), - ), CommandSpec( "bench slo", "benchmarking", @@ -1703,79 +1263,6 @@ class CatalogBypassSite: "devtools bench help-latency --repeats 5 --out .local/help-latency.json", ), ), - CommandSpec( - "verify manifests", - "verification", - "Verify internal consistency across all docs/plans/*.yaml manifest files.", - "devtools.verify_manifests", - use_when=( - "Catch malformed manifests, duplicate rule IDs, missing required fields, " - "and cross-manifest reference inconsistencies." - ), - examples=("devtools verify manifests",), - ), - CommandSpec( - "verify doc-commands", - "verification", - "Verify README/docs command examples resolve to live polylogue, polylogued, and devtools commands.", - "devtools.verify_doc_commands", - use_when=( - "Catch doc drift away from the daemon-first command surface. " - "Fails when README.md or any docs/**/*.md references a subcommand " - "that is not registered, or a stale invocation like " - "'polylogued run --enable-api' / 'polylogue run --source'." - ), - examples=("devtools verify doc-commands", "devtools verify doc-commands --json"), - ), - CommandSpec( - "verify docs-coverage", - "verification", - "Verify every public CLI command, MCP tool, config key, and stable daemon route is named in the docs tree.", - "devtools.verify_docs_coverage", - use_when=( - "Catch doc drift in the other direction from doc-commands: a real public surface " - "(CLI command, MCP tool, config key, stable daemon route) shipped with zero doc-tree " - "mention. Fails naming the exact missing entry (polylogue-3tl.9). Pre-existing gaps " - "are tracked in docs/plans/docs-coverage-baseline.yaml as a ratchet, not an allowlist " - "to extend." - ), - examples=("devtools verify docs-coverage", "devtools verify docs-coverage --json"), - ), - CommandSpec( - "verify ci-workflows", - "verification", - "Verify CI workflow files reference locally-known devtools commands and existing paths.", - "devtools.verify_ci_workflows", - use_when=( - "Catch CI workflow files that reference unregistered devtools commands or " - "non-existent paths. Checks only locally verifiable facts — not remote CI state." - ), - examples=("devtools verify ci-workflows", "devtools verify ci-workflows --json"), - ), - CommandSpec( - "verify catalog-bypasses", - "verification", - "Reject direct devtools module or script execution outside sanctioned adapters.", - "devtools.verify_catalog_bypasses", - use_when=( - "Check workflow run blocks, repository hooks, and devtools process-launch sites for direct " - "python -m devtools.X or python devtools/X.py execution that bypasses the command catalog." - ), - examples=("devtools verify catalog-bypasses", "devtools verify catalog-bypasses --json"), - ), - CommandSpec( - "verify test-infra-currency", - "verification", - "Verify tests/infra/ helpers reference only tables that exist in the current SCHEMA_VERSION.", - "devtools.verify_test_infra_currency", - use_when=( - "Catch helpers that target renamed or removed tables (#1208). " - "When SCHEMA_VERSION bumps, helper SQL drifting away from the live " - "schema is invisible to testmon-selected runs until an unrelated change " - "invalidates the affected tests." - ), - examples=("devtools verify test-infra-currency", "devtools verify test-infra-currency --json"), - ), CommandSpec( "lab policy schema-versioning", "verification lab", @@ -1789,235 +1276,21 @@ class CatalogBypassSite: ), examples=("devtools lab policy schema-versioning", "devtools lab policy schema-versioning --json"), ), - CommandSpec( - "lab policy classifier-fingerprints", - "verification lab", - "Verify parser/classifier decision-boundary changes are declared as reparse-requiring or acknowledged.", - "devtools.verify_classifier_fingerprints", - use_when=( - "Catch the gap `lab policy schema-versioning` cannot see (polylogue-gucv): a parser/classifier " - "under polylogue/sources/ or polylogue/archive/artifact_taxonomy/ (looks_like*/classify_artifact* " - "functions) changes what it accepts for identical input bytes without any INDEX_SCHEMA_VERSION " - "bump at all, so already-indexed rows go silently stale with no signal a reparse was needed " - "(PR #3428 shipped exactly this, green, against the version-keyed gate)." - ), - examples=( - "devtools lab policy classifier-fingerprints", - "devtools lab policy classifier-fingerprints --json", - "devtools lab policy classifier-fingerprints --ack polylogue/sources/parsers/foo.py:looks_like_x " - "--reason 'only tightens a shape that never validly matched' --ref polylogue-abcd", - ), - ), - CommandSpec( - "lab policy raw-payload-hash-purity", - "verification lab", - "Verify no raw-capture write path splices a synthesized literal onto captured bytes before hashing.", - "devtools.verify_raw_payload_hash_purity", - use_when=( - "Prevent a regression of polylogue-u19l's confirmed bug: sources/live/batch.py used to prepend a " - "synthetic session_meta header onto every Codex append capture before hashing/storing it, so the " - "stored blob was never a literal byte-slice of the live file, permanently defeating live-source " - "byte-identity verification for ~59GB of raw rows. Statically forbids concatenating a synthesized " - "literal (bytes/str constant, f-string, json.dumps()/.encode() result) onto captured bytes anywhere " - "in the raw-capture write-path modules (WRITE_PATH_MODULES)." - ), - examples=("devtools lab policy raw-payload-hash-purity", "devtools lab policy raw-payload-hash-purity --json"), - ), - CommandSpec( - "lab policy table-exists-duplication", - "verification lab", - "Verify no module outside storage/introspection.py redefines table_exists/column_exists/index_exists.", - "devtools.verify_table_exists_duplication", - use_when=( - "Keep polylogue-48h's consolidation from silently regrowing: ~25 independently maintained " - "_table_exists/_column_exists/_index_exists copies (each trivially small and subtly different) " - "were merged into polylogue.storage.introspection. A grep-based tripwire forbidding a new " - "top-level def with one of the retired names outside that module." - ), - examples=( - "devtools lab policy table-exists-duplication", - "devtools lab policy table-exists-duplication --json", - ), - ), - CommandSpec( - "lab policy position-derived-identity", - "verification lab", - "Verify no parser mints cross-revision comparison identity from positional/index data.", - "devtools.verify_position_derived_identity", - use_when=( - "Prevent a regression of polylogue-hith/qkuq's already-fixed attachment-id bug (a synthetic id " - "seeded partly by array index was unstable across export vintages that reorder/insert entries, " - "manufacturing false divergence in revision-authority membership comparison) and catch new " - "occurrences of the same shape (polylogue-gysk3 found the identical hazard still live for " - "provider_message_id, the sole input to message_identity_hash). Statically scans " - "polylogue/sources/parsers/ for an identity-bearing field constructed from an f-string/format/" - "concatenation referencing a loop index/position variable, either inline or via a local variable." - ), - examples=( - "devtools lab policy position-derived-identity", - "devtools lab policy position-derived-identity --json", - "devtools lab policy position-derived-identity --ack " - "'polylogue/sources/parsers/foo.py:parse:provider_message_id' " - "--reason 'tracked in polylogue-xxxx, not fixed inline' --ref polylogue-xxxx", - ), - ), - CommandSpec( - "lab policy raw-authority-frontier-executability", - "verification lab", - "Verify every raw-authority frontier state has a reachable actuator.", - "devtools.verify_raw_authority_frontier_executability", - use_when=( - "polylogue-w32w / polylogue-lb39z (Phase 1, item 4): " - "RawAuthorityFrontierItem.__post_init__ raises if a CONSTRUCTED item pairs a " - "dispatched actuator (_APPLY_DISPATCHED_ACTUATORS) with a non-executable state " - "(_EXECUTABLE_STATES) -- but that only fires when a test or live classification " - "actually builds one; a new frontier state or re-paired actuator can ship an " - "unexercised branch that stays silent until it accumulates against real archive " - "data (the original defect: 4,174 blockers demanding an unreachable actuator, " - "undetected for weeks). This lint statically enumerates every literal " - "(state, actuator) construction site in polylogue/storage/raw_reconciler.py " - "(_item(...) and _StrategyOverride(...) calls) and re-checks the same invariant " - "at review time, independent of test coverage." - ), - examples=( - "devtools lab policy raw-authority-frontier-executability", - "devtools lab policy raw-authority-frontier-executability --json", - ), - ), - CommandSpec( - "lab policy acceptance-contracts", - "verification lab", - "Validate structured Beads acceptance contracts and the committed contract manifest.", - "devtools.beads_acceptance_contracts", - use_when=( - "Run the source-digest-bound acceptance-contract gate before shipping Beads state. " - "It validates typed contract fields, rendered criteria equality, planner-review " - "dispatch markers, safety and receipt clauses, and the committed 218-Bead manifest." - ), - examples=( - "devtools lab policy acceptance-contracts", - "devtools lab policy acceptance-contracts --json", - ), - ), - CommandSpec( - "lab policy acceptance-contract-reconcile", - "verification lab", - "Reconcile canonical acceptance contracts with a read-only live Beads export.", - "devtools.beads_acceptance_reconciliation", - use_when=( - "Run this file-level dry run before a coordinator applies acceptance contracts. " - "It reports authority differences, refuses source-digest mismatches, and emits " - "a targeted wave made from live rows with only acceptance_criteria and " - "metadata.acceptance_contract_v1 changed. It never invokes bd or mutates Dolt." - ), - examples=( - "devtools lab policy acceptance-contract-reconcile --repository .beads/issues.jsonl " - "--live /path/live.jsonl --wave /path/targeted.jsonl --report /path/report.json", - "devtools lab policy acceptance-contract-reconcile --verify-repository .beads/issues.jsonl " - "--verify-report /path/report.json --verify-before /path/before.jsonl " - "--verify-after /path/after.jsonl --verify-wave /path/targeted.jsonl --json", - ), - ), - CommandSpec( - "lab policy acceptance-contract-apply", - "verification lab", - "Apply an exact acceptance wave to a guarded JSONL file copy.", - "devtools.beads_acceptance_applier", - use_when=( - "Exercise the local guarded applier against an exact reconciliation report and wave. " - "This route never invokes bd or mutates the Beads database and accepts an identical " - "prior output idempotently." - ), - examples=( - "devtools lab policy acceptance-contract-apply --repository .beads/issues.jsonl " - "--before RUN/live-before.jsonl --wave RUN/targeted.jsonl --report RUN/reconciliation.json " - "--output RUN/applied.jsonl --json", - ), - ), - CommandSpec( - "lab policy backlog-hygiene", - "verification lab", - "Verify Beads backlog structure invariants (.beads/issues.jsonl).", - "devtools.verify_backlog_hygiene", - use_when=( - "Enforce the standing backlog-hygiene invariant lint (polylogue-8jg9.1): 20 checks " - "over the Beads export catching dangling dependency refs, blocks-cycles, missing " - "horizon/AC/design content on tech-tree beads, P0/P1 beads without acceptance " - "criteria, unlabeled non-epic beads, epics with no members or description, stale " - "'adopted' decisions left open, duplicate titles, bead ids named but never " - "created, an unclean/corrupt bd JSONL sync receipt (S1, consuming " - "polylogue-gxjh.1's monotonic sync contract), an active leaf that is itself an " - "epic (F1), an active leaf with a missing/dangling/mismatched program ref (F2), a " - "stale in_progress claim with no recent activity (F3, configurable window), and a " - "frontier_program=active program with no admitted active leaves (F4) -- catches " - "backlog structure drift before it needs an archaeology sweep to recover, instead " - "of only a manually-invoked script. Also reports a non-blocking active-set size " - "diagnostic (soft target/warn bands, never a hard cap or failure)." - ), - examples=( - "devtools lab policy backlog-hygiene", - "devtools lab policy backlog-hygiene --json", - "devtools lab policy backlog-hygiene --fresh", - "devtools lab policy backlog-hygiene --stale-claim-days 10", - ), - ), CommandSpec( "lab policy bead-graph", "verification lab", - "Bead-graph invariant lint and complete missing-AC census over live `bd` state.", + "Validate typed dependency endpoints, uniqueness, parent cardinality, and cycles in the Beads graph.", "devtools.verify_bead_graph", use_when=( - "Run right before shipping a bead-state delta (matches the sinex bead-graph-lint " - "convention). Checks LIVE `bd dep cycles` / `bd list --all --json` output rather than " - "the exported .beads/issues.jsonl snapshot, so it catches drift not yet re-exported. It fails " - "closed for real missing acceptance criteria and validates zero-or-one structured parent-child " - "parents. `--json` emits the complete deterministic missing-AC census, never a display page. " - "INTENTIONAL DIVERGENCE from sinex: only duplicate `wave:` labels are flagged " - "(polylogue's `lane:`/`delivery:`/`horizon:` taxonomy is local and not enforced here)." + "Run before shipping a bead-state delta. With no source option it checks live `bd` state; " + "`--export .beads/issues.jsonl` validates the branch snapshot without importing it into the " + "shared database. The gate reads dependency records only and does not make prose, labels, or " + "campaign-specific edge lists machine authority." ), - examples=("devtools lab policy bead-graph", "devtools lab policy bead-graph --json"), - ), - CommandSpec( - "lab policy demo-packet-registry", - "verification lab", - "Verify every registered 212 demo has a conforming Demo Finding Packet.", - "devtools.verify_demo_packet_registry", - use_when=( - "Enforce the 212 portfolio contract (polylogue-212.7): every demo prompt in " - ".agent/demos/registry.json must have a packet directory carrying PROMPT.md, " - "finding.yaml (five-part provenance stanza), report.md (fixed section order), " - "evidence.ndjson, queries.ndjson, checks.json, and run.log. Catches a missing " - "or malformed packet before it silently drops out of the demo shelf." - ), - examples=("devtools lab policy demo-packet-registry", "devtools lab policy demo-packet-registry --json"), - ), - CommandSpec( - "lab policy demo-tour-freshness", - "verification lab", - "Verify a freshly-run demo tour matches the committed docs/examples/demo-tour/ evidence artifacts.", - "devtools.verify_demo_tour_freshness", - use_when=( - "Catch drift between what `polylogue demo tour` actually emits at runtime (transcript, " - "report, per-step command output, recording tape) and the committed copies under " - "docs/examples/demo-tour/, modulo an explicit wall-clock-duration mask (polylogue-3tl.17). " - "Runs the real tour (~10s), so it lives in the lab tier rather than --quick." - ), - examples=("devtools lab policy demo-tour-freshness",), - ), - CommandSpec( - "lab policy docs-drift", - "verification lab", - "Verify checkable factual claims in the Reference-docs table against current source.", - "devtools.verify_docs_drift", - use_when=( - "Catch doc-vs-code drift in the hand-maintained Reference-docs table " - "(CLAUDE.md): a backtick-quoted file path that no longer exists, a " - "' schema version N' claim ahead of the tier's current constant, or " - "a watchlisted table name renamed to a different current name (e.g. " - "`artifact_observations` renamed to `raw_artifacts`) still asserted as " - "current (polylogue-9e5.13)." + examples=( + "devtools lab policy bead-graph", + "devtools lab policy bead-graph --export .beads/issues.jsonl --json", ), - examples=("devtools lab policy docs-drift", "devtools lab policy docs-drift --json"), ), CommandSpec( "lab policy timestamp-doctrine", @@ -2047,49 +1320,6 @@ class CatalogBypassSite: ), examples=("devtools lab policy insight-honesty", "devtools lab policy insight-honesty --json"), ), - CommandSpec( - "lab policy campaign-archive-boundaries", - "verification lab", - "Verify devtools synthetic benchmark/scale campaigns route through ArchiveLocation.", - "devtools.verify_campaign_archive_boundaries", - use_when=( - "Catch a regression of the phantom-benchmark.db bug (polylogue-ovme.3): a campaign " - "reintroducing a 'benchmark.db' sentinel, an ad hoc tier-path sibling derivation, or " - "an entry point (generate_archive/run_full_campaign/run_campaign._run) that no longer " - "routes through CampaignArchiveLocation. Scoped to the devtools campaign boundary only " - "-- the broader storage/diagnostics/daemon/maintenance/transitions boundary audit is " - "polylogue-ovme.2's migration surface." - ), - examples=( - "devtools lab policy campaign-archive-boundaries", - "devtools lab policy campaign-archive-boundaries --json", - ), - ), - CommandSpec( - "verify pytest-timeout-overrides", - "verification", - "Verify explicit pytest timeout overrides are positive, bounded, and justified.", - "devtools.verify_pytest_timeout_overrides", - use_when=( - "Check AST-parsed @pytest.mark.timeout decorators and managed pytest command literals. " - "Values above the pyproject default require an exact manifest rationale." - ), - examples=("devtools verify pytest-timeout-overrides", "devtools verify pytest-timeout-overrides --json"), - ), - CommandSpec( - "verify degrade-loudly", - "verification", - "Verify broad except-handlers in daemon/storage/insights/coordination log or signal on failure.", - "devtools.verify_degrade_loudly", - use_when=( - "Enforce the degrade-loudly doctrine (polylogue-cpf.4): a broad except-handler " - "(Exception/BaseException/*.Error) in derived-read, status, or probe code that " - "swallows the exception with no log call and no re-raise is indistinguishable from " - "'no data' to a reader. New silent sites must add a log call, or add a typed signal " - "plus a rationale entry in docs/plans/degrade-loudly-allowlist.yaml." - ), - examples=("devtools verify degrade-loudly", "devtools verify degrade-loudly --json"), - ), CommandSpec( "release verify-distribution", "release", @@ -2291,20 +1521,6 @@ class CatalogBypassSite: ), examples=("devtools verify layering", "devtools verify layering --json"), ), - CommandSpec( - "verify evidence", - "verification", - "Render the pytest-first evidence dashboard.", - "devtools.evidence_dashboard", - use_when=( - "Inspect pytest health, contract-evidence inventory, coverage, SLO " - "catalog, static-gate status, and campaign freshness." - ), - examples=( - "devtools verify evidence --json", - "devtools verify evidence --markdown", - ), - ), CommandSpec( "release build-package", "release", @@ -2316,25 +1532,12 @@ class CatalogBypassSite: CommandSpec( "bench mutation", "benchmarking", - "Run focused mutation campaigns and maintain their local index.", + "Run focused mutation campaigns with isolated execution and JSON artifacts.", "devtools.mutmut_campaign", use_when="Run or inspect focused mutation-testing work without shrinking the committed mutmut scope.", examples=("devtools bench mutation list", "devtools bench mutation run filters"), featured=True, ), - CommandSpec( - "bench campaign", - "benchmarking", - "Run or compare benchmark campaigns.", - "devtools.benchmark_campaign", - use_when="Record durable benchmark artifacts or compare a candidate run against a baseline artifact.", - examples=( - "devtools bench campaign list", - "devtools bench campaign run search-filters", - "devtools bench campaign compare baseline.json candidate.json", - ), - featured=True, - ), CommandSpec( "bench nightly-compare", "benchmarking", @@ -2357,21 +1560,6 @@ class CatalogBypassSite: "devtools bench synthetic --scale medium --campaign search-filters", ), ), - CommandSpec( - "workspace tasks", - "workspace", - "Record and query local agent task execution history.", - "devtools.task_history", - use_when="Log, view recent, or summarize agent task execution history during development sessions.", - examples=( - "devtools workspace tasks log --command 'devtools render all' --duration-ms 3200 --exit-code 0", - "devtools workspace tasks recent", - "devtools workspace tasks recent --count 20", - "devtools workspace tasks stats", - "devtools workspace tasks stats --json", - "devtools workspace tasks stats --resources", - ), - ), CommandSpec( "workspace failure-context", "workspace", @@ -2388,50 +1576,20 @@ class CatalogBypassSite: ), ), CommandSpec( - "workspace mandate-continuity-replay", + "workspace continuity-evidence", "workspace", - "Wire t8t continuity scenarios + work-evidence effects + discovery into one mandate artifact.", - "devtools.mandate_continuity_replay", + "Replay continuity scenarios and verify their query routes are discoverable.", + "devtools.continuity_evidence", use_when=( - "Run the polylogue-z9gh.7 terminal mandate gate: replay the polylogue-t8t continuity scenario " - "catalog over real MCP stdio JSON-RPC, reconcile this repository's own real git+Beads history " - "through the polylogue-1vpm.6.2 effect adapters, cross-check every query-tool route step against " - "the polylogue-z9gh.3 query-discovery catalog, and emit one JSON artifact with a mandate " - "acceptance-criteria matrix. Defaults to a fresh, privacy-safe synthetic archive; pass " - "--archive-root for an authorized live-scale replay." + "Replay the continuity scenario catalog over MCP stdio JSON-RPC and cross-check " + "its query routes against discovery. The default seeds the packaged synthetic corpus. " + "A supplied --archive-root must be paired with the exact --catalog that describes it; " + "the runner rejects an unrelated live archive rather than applying synthetic oracles." ), examples=( - "devtools workspace mandate-continuity-replay", - "devtools workspace mandate-continuity-replay --output .cache/mandate-continuity-replay.json", - "devtools workspace mandate-continuity-replay --archive-root /path/to/authorized/archive --keep-archive", - ), - ), - CommandSpec( - "workspace beads-state-report", - "workspace", - "Self-contained HTML state-of-the-backlog report over the whole bead population.", - "devtools.beads_state_report", - use_when=( - "Answer 'what shape is the backlog in?' across the whole bead population -- open " - "AND closed -- rather than the ready frontier. Computes status x priority x type, " - "epic/program trees with fill bars and per-epic trend sparklines, the blocks-graph " - "topology (ready/blocked/top blockers/cycles/densest cluster/parallel frontier), a " - "Pulse section reconstructing open/ready/blocked at now vs 7/14 days ago from " - "timestamps, creation/closure velocity over trailing windows, an age x priority " - "heatmap, graph-health review queues (dangling refs, id-vs-edge hierarchy " - "disagreement, open parents whose children all closed, stale in-progress claims, " - "duplicate titles), subsystem concentration by both area label and keyword, and the " - "dated hand-verified VERIFICATION(...) / RECONCILIATION marker subsets. Its findings " - "list is generated by conditional checks over the data, so regeneration cannot leave " - "stale claims. Answers a different question than `workspace bead-cluster` " - "(execution-frontier footprint clustering) and `workspace delivery-gate-status` " - "(per-gate progress): this is population shape and graph hygiene, not the next batch " - "to dispatch. Use --fresh, since bd mutations do not immediately re-export." - ), - examples=( - "devtools workspace beads-state-report --fresh", - "devtools workspace beads-state-report --out /tmp/beads-state.html", - "devtools workspace beads-state-report --json", + "devtools workspace continuity-evidence", + "devtools workspace continuity-evidence --output .cache/continuity-evidence.json", + "devtools workspace continuity-evidence --archive-root /path/to/archive --catalog /path/to/catalog.json", ), ), ) @@ -2439,121 +1597,6 @@ class CatalogBypassSite: COMMANDS: dict[str, CommandSpec] = {spec.name: spec for spec in COMMAND_SPECS} -WORKSPACE_COMMAND_DISPOSITIONS: tuple[WorkspaceCommandDisposition, ...] = ( - WorkspaceCommandDisposition( - "workspace index-fast-forward", - "retain", - "Recent production commits plus focused devtools/storage tests; emits a reusable proof receipt.", - "Keep the registered command as the index-tier actuator.", - ), - WorkspaceCommandDisposition( - "workspace archive-schema-fast-forward", - "remove", - "No current CommandSpec, implementation module, focused test, or history entry exists for this name.", - "Use workspace index-fast-forward for declared derived-index fast-forwards.", - "workspace index-fast-forward", - ), - WorkspaceCommandDisposition( - "workspace degraded-archive-proof", - "retain", - "Focused tests and deterministic self-healing proof artifacts cover the command.", - "Keep the registered command for archive repair evidence.", - ), - WorkspaceCommandDisposition( - "workspace frontier", - "retain", - "Operator-facing frontier report with documented workflow use and structured report output.", - "Keep the registered command for frontier batching and wait-ahead decisions.", - ), - WorkspaceCommandDisposition( - "workspace temporal-read-profile", - "retain", - "Focused tests cover the report and JSON timing output is reusable for read tuning.", - "Keep the registered command as the temporal read profiling entrypoint.", - ), - WorkspaceCommandDisposition( - "workspace temporal-devloop", - "retain", - "Focused tests cover structured and Markdown event sources; output is a reusable evidence window.", - "Keep the registered command as the devloop temporal evidence entrypoint.", - ), - WorkspaceCommandDisposition( - "workspace temporal-archive-aggregates", - "retain", - "Focused tests cover aggregate report construction and reusable archive artifacts.", - "Keep the registered command as the run-projection aggregate entrypoint.", - ), - WorkspaceCommandDisposition( - "workspace lineage-validation", - "retain", - "Focused tests cover lineage evidence and the command emits reusable count and composition proof.", - "Keep the registered command before publishing archive cardinality claims.", - ), - WorkspaceCommandDisposition( - "workspace cli-surface-audit", - "retain", - "Focused tests cover bounded output and stale-artifact pruning; the audit shelf is reusable.", - "Keep the registered command as the current CLI surface audit entrypoint.", - ), - WorkspaceCommandDisposition( - "demo real-slice-screen", - "retain", - "Focused privacy-screening tests cover redaction, PII review, and report generation.", - "Keep the registered command as the read-only real-archive screening entrypoint.", - ), -) - - -CATALOG_BYPASS_SITES: tuple[CatalogBypassSite, ...] = ( - CatalogBypassSite( - ".github/workflows/mutation-testing.yml", - "uv run devtools verify mutation-freshness", - "verify mutation-freshness", - "registered", - "CI invokes the catalog command so inventory and workflow validation see the freshness gate.", - ), - CatalogBypassSite( - ".github/workflows/nightly-scale.yml", - "uv run devtools bench nightly-compare", - "bench nightly-compare", - "registered", - "CI invokes the catalog command so the nightly comparison is discoverable and checked.", - ), - CatalogBypassSite( - "devtools/pre_push_gate.py", - 'control_plane_argv("lab policy backlog-hygiene")', - "lab policy backlog-hygiene", - "registered", - "The intentional Beads-only route remains narrow while using the registered policy command.", - ), - CatalogBypassSite( - "docs/test-economics.md", - "python -m devtools.test_economics_report", - "lab test-economics", - "sanctioned-bypass", - "The generated provenance header preserves the module that emitted the document; operators use the catalog command.", - ), - CatalogBypassSite( - ".githooks/pre-push", - "python -m devtools.pre_push_gate", - None, - "sanctioned-bypass", - "The hook adapter must receive Git's staged stdin update stream before dispatching its catalog-aware gate.", - occurrence_line=21, - expected_occurrences=1, - ), - CatalogBypassSite( - ".beads-hooks/pre-push", - "python -m devtools.pre_push_gate", - None, - "sanctioned-bypass", - "The Beads-augmented hook retains the same stdin adapter before its managed Beads section runs.", - occurrence_line=21, - expected_occurrences=1, - ), -) - - def command_name_from_tokens(tokens: Iterable[str], commands: Iterable[CommandSpec] = COMMAND_SPECS) -> str | None: """Resolve leading argv tokens to a registered command name.""" token_tuple = tuple(tokens) @@ -2586,8 +1629,7 @@ def featured_command_specs(commands: Iterable[CommandSpec] = COMMAND_SPECS) -> t def verification_lab_command_specs(commands: Iterable[CommandSpec] = COMMAND_SPECS) -> tuple[CommandSpec, ...]: - by_name = {spec.name: spec for spec in commands} - return tuple(by_name[name] for name in VERIFICATION_LAB_COMMAND_NAMES) + return tuple(spec for spec in commands if spec.category == "verification lab") def grouped_command_specs(commands: Iterable[CommandSpec] = COMMAND_SPECS) -> OrderedDict[str, list[CommandSpec]]: @@ -2604,17 +1646,13 @@ def grouped_command_specs(commands: Iterable[CommandSpec] = COMMAND_SPECS) -> Or "CATEGORY_ORDER", "COMMANDS", "COMMAND_SPECS", - "CATALOG_BYPASS_SITES", "CONTROL_PLANE", "CommandMain", "CommandSpec", - "WorkspaceCommandDisposition", "command_name_from_tokens", "control_plane_argv", "control_plane_command", "featured_command_specs", "grouped_command_specs", - "VERIFICATION_LAB_COMMAND_NAMES", - "WORKSPACE_COMMAND_DISPOSITIONS", "verification_lab_command_specs", ] diff --git a/devtools/continuity_evidence.py b/devtools/continuity_evidence.py new file mode 100644 index 0000000000..3b058d10a6 --- /dev/null +++ b/devtools/continuity_evidence.py @@ -0,0 +1,272 @@ +"""Replay continuity and query-discovery behavior together. + +The continuity scenario suite exercises seven workflows plus the +parallel-agent incident variant against a deterministic synthetic archive. +The executable query-discovery catalog describes the plans a cold client can +formulate. This module runs those two existing capabilities together without +reimplementing them. + +This module is that wiring, not a fourth reimplementation: + +- :func:`check_discovery_coverage` reuses the real + :data:`polylogue.archive.query.discovery.QUERY_DISCOVERY_EXAMPLES` catalog + to prove every ``query``-tool route step any continuity scenario executes + has a declared positive example of the same unit-source/route shape -- + i.e. a cold model relying on discovery alone could have found that plan + family, not just executed it once the runner already knew it. +- :func:`run_continuity_evidence` calls + :func:`devtools.continuity_replay.replay_archive` unmodified against either + a supplied archive paired with its explicit corpus/oracle catalog or a + freshly seeded synthetic corpus (the default, privacy-safe CI lane), then + combines both executable lanes into one JSON artifact. :func:`redact_report` + strips raw evidence prose from that + artifact (keeping refs/hashes/counts) for the live-archive lane; the + synthetic lane never touches private content so redaction there is a no-op + proof of the same mechanism, not a load-bearing privacy boundary. + +The report states whether it used a supplied live archive. It does not copy a +tracker item's acceptance prose into product output or infer tracker closure +from the three lane statuses. +""" + +from __future__ import annotations + +import argparse +import asyncio +import hashlib +import json +import sys +import time +from collections.abc import Sequence +from dataclasses import asdict, dataclass +from pathlib import Path +from tempfile import TemporaryDirectory +from typing import Literal, TextIO, cast + +if __package__ in {None, ""}: # pragma: no cover - exercised by the script entry point + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from devtools.continuity_replay import replay_archive +from polylogue.archive.query.discovery import QUERY_DISCOVERY_EXAMPLES +from polylogue.core.json import JSONDocument, JSONValue, require_json_document +from polylogue.product.continuity_scenarios import CONTINUITY_SCENARIOS, ContinuityScenarioSpec, continuity_scenario +from tests.infra.continuity import load_continuity_catalog, seed_continuity_archive + +# ── Discovery-coverage lane ─────────────────────────────────────────── + + +@dataclass(frozen=True, slots=True) +class DiscoveryCoverageGap: + """One continuity route step whose plan family has no discovery example.""" + + scenario_id: str + step_id: str + plan_atom: str + reason: str + + def to_dict(self) -> dict[str, str]: + return asdict(self) + + +@dataclass(frozen=True, slots=True) +class DiscoveryCoverageReport: + """Whether every continuity-scenario query plan is independently discoverable.""" + + checked_steps: int + covered_steps: int + gaps: tuple[DiscoveryCoverageGap, ...] + + @property + def status(self) -> Literal["pass", "fail"]: + return "pass" if not self.gaps else "fail" + + def to_dict(self) -> dict[str, object]: + return { + "status": self.status, + "checked_steps": self.checked_steps, + "covered_steps": self.covered_steps, + "gaps": [gap.to_dict() for gap in self.gaps], + } + + +def check_discovery_coverage( + scenarios: Sequence[ContinuityScenarioSpec], +) -> DiscoveryCoverageReport: + """Prove every ``query``-tool continuity route step is independently discoverable. + + A continuity scenario's route steps prove the runner *can execute* a + plan; they say nothing about whether a cold model, given only the + published query-discovery catalog (``archive/query/discovery.py``), could + have *formulated* that same plan family without hidden knowledge. This + cross-checks each ``query``-tool step's unit-source against + ``QUERY_DISCOVERY_EXAMPLES`` -- the same catalog z9gh.3 generates MCP + schemas/completions from -- so a shipped scenario whose plan family the + discovery catalog does not teach shows up as a named gap rather than a + silent success. + """ + + catalog_atoms = {f"query:{example.unit_source}" for example in QUERY_DISCOVERY_EXAMPLES if example.route == "query"} + gaps: list[DiscoveryCoverageGap] = [] + checked = 0 + for scenario in scenarios: + for step in scenario.route_steps: + if step.tool != "query": + continue + checked += 1 + atom = step.plan_atom + if atom not in catalog_atoms: + gaps.append( + DiscoveryCoverageGap( + scenario_id=scenario.scenario_id, + step_id=step.step_id, + plan_atom=atom, + reason=f"no declared query-discovery example teaches {atom!r}", + ) + ) + return DiscoveryCoverageReport(checked_steps=checked, covered_steps=checked - len(gaps), gaps=tuple(gaps)) + + +# ── Redaction ───────────────────────────────────────────────────────── + +_REDACTABLE_KEYS = frozenset({"label", "claim_text", "reason", "response_sha256"}) + + +def _redact_value(key: str, value: JSONValue) -> JSONValue: + if key in _REDACTABLE_KEYS and isinstance(value, str) and value: + return f"redacted:sha256:{hashlib.sha256(value.encode('utf-8')).hexdigest()}" + return value + + +def redact_report(document: JSONValue) -> JSONValue: + """Strip raw evidence prose from a mandate report, keeping refs/counts/hashes. + + Recursively walks the report replacing any string value stored under a + label/claim-text/reason-shaped key with a stable hash of itself. Refs, + ids, statuses, and counts (everything the mandate AC actually needs + cited) pass through unchanged -- only free-text prose that could carry + private archive content is hashed. + """ + + if isinstance(document, dict): + return {key: _redact_value(key, redact_report(value)) for key, value in document.items()} + if isinstance(document, list): + return [redact_report(item) for item in document] + return document + + +# ── Orchestration ────────────────────────────────────────────────────── + + +async def run_continuity_evidence( + *, + archive_root: Path | None = None, + catalog_path: Path | None = None, + scenario_names: Sequence[str] | None = None, + redact: bool = True, + keep_archive: bool = False, +) -> JSONDocument: + """Run the continuity and discovery lanes as one artifact. + + When ``archive_root`` is ``None`` (the default), a fresh, privacy-safe + synthetic continuity corpus is seeded and torn down automatically -- the + CI/deterministic lane. A supplied archive must also name the independent + corpus/oracle catalog that describes that archive. Reusing the planted + synthetic catalog against unrelated live data would test fixture identity, + not continuity behavior, so that ambiguous mode is rejected. + """ + + started_ns = time.perf_counter_ns() + live_archive = archive_root is not None + if live_archive and catalog_path is None: + raise ValueError("--archive-root requires --catalog for the selected archive") + catalog = load_continuity_catalog(catalog_path) + workdir: TemporaryDirectory[str] | None = None + + resolved_root: Path + if archive_root is None: + workdir = TemporaryDirectory(prefix="polylogue-continuity-evidence-") + resolved_root = Path(workdir.name) / "archive" + seed_continuity_archive(resolved_root, catalog=catalog) + else: + resolved_root = archive_root + + try: + continuity_report = await replay_archive(resolved_root, catalog, scenario_names=scenario_names) + finally: + if workdir is not None and not keep_archive: + workdir.cleanup() + + scenarios = ( + CONTINUITY_SCENARIOS if scenario_names is None else tuple(continuity_scenario(name) for name in scenario_names) + ) + discovery_report = check_discovery_coverage(scenarios) + overall_status: Literal["pass", "fail"] = ( + "pass" if continuity_report.get("status") == "pass" and discovery_report.status == "pass" else "fail" + ) + report: dict[str, object] = { + "schema_version": 3, + "live_archive": live_archive, + "catalog_sha256": hashlib.sha256( + json.dumps(catalog, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8") + ).hexdigest(), + "archive_root": str(resolved_root.resolve()) if keep_archive or live_archive else None, + "elapsed_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + "status": overall_status, + "continuity": continuity_report, + "discovery_coverage": discovery_report.to_dict(), + } + document = require_json_document(report, context="continuity evidence report") + return cast(JSONDocument, redact_report(document)) if redact else document + + +def _scenario_names(value: str) -> tuple[str, ...] | None: + if value == "all": + return None + return tuple(part.strip() for part in value.split(",") if part.strip()) + + +def main(argv: list[str] | None = None, *, stdout: TextIO | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--archive-root", + type=Path, + default=None, + help="Archive to replay; requires an exact matching --catalog.", + ) + parser.add_argument("--catalog", type=Path, help="Corpus/oracle catalog describing --archive-root") + parser.add_argument("--scenario", default="all", help="all or a comma-separated scenario id list") + parser.add_argument("--no-redact", action="store_true", help="Disable evidence redaction (CI/synthetic lane only)") + parser.add_argument("--keep-archive", action="store_true") + parser.add_argument("--output", type=Path) + args = parser.parse_args(argv) + + report = asyncio.run( + run_continuity_evidence( + archive_root=args.archive_root, + catalog_path=args.catalog, + scenario_names=_scenario_names(args.scenario), + redact=not args.no_redact, + keep_archive=args.keep_archive, + ) + ) + rendered = json.dumps(report, indent=2, sort_keys=True) + out = stdout or sys.stdout + if args.output is not None: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(rendered + "\n", encoding="utf-8") + print(rendered, file=out) + return 0 if report["status"] == "pass" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) + + +__all__ = [ + "DiscoveryCoverageGap", + "DiscoveryCoverageReport", + "check_discovery_coverage", + "main", + "redact_report", + "run_continuity_evidence", +] diff --git a/devtools/continuity_replay.py b/devtools/continuity_replay.py index 1a27c02762..0ecf66817c 100644 --- a/devtools/continuity_replay.py +++ b/devtools/continuity_replay.py @@ -13,21 +13,21 @@ import hashlib import inspect import json -import os import re import sys +import threading import time from collections.abc import Callable, Mapping, Sequence from contextlib import AsyncExitStack, suppress from dataclasses import dataclass, field from datetime import timedelta from pathlib import Path +from tempfile import TemporaryDirectory from typing import TYPE_CHECKING, Literal, Protocol, TypeAlias, TypeGuard, cast if __package__ in {None, ""}: # pragma: no cover - exercised by the script entry point sys.path.insert(0, str(Path(__file__).resolve().parents[1])) -from polylogue.archive.query.execution_control import DEFAULT_CAPACITY from polylogue.core.json import JSONDocument, JSONValue, require_json_document, require_json_value from polylogue.product.continuity_scenarios import ( CONTINUITY_SCENARIOS, @@ -51,13 +51,8 @@ ArgumentMutator: TypeAlias = Callable[[str, RouteArguments, int], RouteArguments] ResponseMutator: TypeAlias = Callable[[str, RouteArguments, int, str], str] DiscoveryMutator: TypeAlias = Callable[[JSONDocument], JSONDocument] -#: Concurrent copies of one real route step issued by the cancellation -#: exercise (see StdioMCPContinuityRoute.exercise_cancellation). Comfortably -#: above the production admission controller's default per-process capacity -#: so at least one copy is provably still queued (never touched SQLite) when -#: the cancellation notifications are sent, making the exercise deterministic -#: rather than a race against how fast any one query happens to run. -_CANCELLATION_PROBE_CONCURRENCY = DEFAULT_CAPACITY + 4 +_CONTINUITY_DAEMON_SINK_URL = "http://127.0.0.1:1" +_CONTINUITY_API_TOKEN_SENTINEL = "continuity-replay-call-log-disabled" _TEMPLATE_RE = re.compile(r"\{fixture:([A-Za-z0-9_.-]+)\}") _ATTEMPT_TOKEN_RE = re.compile(r"(?P[a-z_]+):(?P[A-Za-z0-9_-]+)") _ATTEMPT_GRADES: frozenset[str] = frozenset( @@ -72,6 +67,86 @@ ) +def _hermetic_stdio_environment(*, runtime_root: Path, archive_root: Path) -> dict[str, str]: + """Build the complete environment for an offline continuity subprocess. + + The route deliberately does not inherit the operator's environment. In + particular, daemon URLs/tokens, MCP capabilities, config locations, and + user data roots must not escape into an archive replay. The fixed daemon + sink and inert token make call-log isolation testable without granting any + real daemon authority. + """ + + roots = { + "HOME": runtime_root / "home", + "XDG_CONFIG_HOME": runtime_root / "config", + "XDG_DATA_HOME": runtime_root / "data", + "XDG_STATE_HOME": runtime_root / "state", + "XDG_CACHE_HOME": runtime_root / "cache", + "XDG_RUNTIME_DIR": runtime_root / "runtime", + "TMPDIR": runtime_root / "tmp", + } + for name, path in roots.items(): + path.mkdir(mode=0o700, parents=True, exist_ok=True) + if name == "XDG_RUNTIME_DIR": + path.chmod(0o700) + return { + **{name: str(path) for name, path in roots.items()}, + "POLYLOGUE_ARCHIVE_ROOT": str(archive_root), + "POLYLOGUE_CONFIG": str(runtime_root / "config" / "absent-polylogue.toml"), + "POLYLOGUE_SITE_CONFIG": "", + "POLYLOGUE_DAEMON": "off", + "POLYLOGUE_NO_DAEMON": "1", + "POLYLOGUE_DAEMON_URL": _CONTINUITY_DAEMON_SINK_URL, + "POLYLOGUE_API_AUTH_TOKEN": _CONTINUITY_API_TOKEN_SENTINEL, + "POLYLOGUE_MCP_WRITE_ENABLED": "0", + "POLYLOGUE_MCP_JUDGE_ENABLED": "0", + "POLYLOGUE_MCP_MAINTENANCE_ENABLED": "0", + "POLYLOGUE_FORCE_PLAIN": "1", + "PYTHONNOUSERSITE": "1", + "PYTHONUNBUFFERED": "1", + } + + +def _serve_saturated_cancellation_probe() -> None: + """Run the real MCP server with one production admission slot held. + + Fast fixture reads can finish before a cancellation notification reaches + the server. Guessing a sleep or cancelling a swarm of predicted request + ids races the MCP SDK's responder cleanup. This disposable subprocess + instead creates the ordinary production admission controller with one + slot, occupies that slot through its real ``admit_blocking`` path, and + then starts the unmodified Polylogue MCP server. The probe request is + therefore deterministically queued until its real MCP task is cancelled. + """ + + from polylogue.archive.query import execution_control + from polylogue.mcp.cli import main + + controller = execution_control.QueryAdmissionController(capacity=1, reserved_interactive=0) + holder = execution_control.QueryExecutionContext.create( + query_text="continuity-cancellation-holder", + timeout_s=None, + ) + held = threading.Event() + release = threading.Event() + + def hold_admission_slot() -> None: + with controller.admit_blocking(holder): + held.set() + release.wait() + + thread = threading.Thread(target=hold_admission_slot, daemon=True) + thread.start() + if not held.wait(timeout=5): + raise RuntimeError("continuity cancellation holder did not acquire admission") + with execution_control._default_controller_lock: + if execution_control._default_controller is not None: + raise RuntimeError("continuity cancellation server initialized admission too early") + execution_control._default_controller = controller + main() + + CancellationOutcome: TypeAlias = Literal[ "cancelled_confirmed", "completed_before_cancel", @@ -300,16 +375,20 @@ def __init__( response_mutator: ResponseMutator | None = None, discovery_mutator: DiscoveryMutator | None = None, read_timeout_seconds: float = 60.0, + saturated_cancellation_probe: bool = False, ) -> None: self.archive_root = archive_root.resolve() self.argument_mutator = argument_mutator self.response_mutator = response_mutator self.discovery_mutator = discovery_mutator self.read_timeout_seconds = read_timeout_seconds + self.saturated_cancellation_probe = saturated_cancellation_probe self._stack: AsyncExitStack | None = None self._session: ClientSession | None = None + self._runtime_workdir: TemporaryDirectory[str] | None = None self._discovery: JSONDocument = {} self._invocation_count = 0 + self._cancellation_receipts: dict[str, CancellationExerciseReceipt] = {} @property def transport_name(self) -> str: @@ -323,19 +402,18 @@ async def __aenter__(self) -> StdioMCPContinuityRoute: from mcp import ClientSession, StdioServerParameters from mcp.client.stdio import stdio_client - runtime_root = self.archive_root / ".continuity-runtime" - environment = dict(os.environ) - environment.update( - { - "POLYLOGUE_ARCHIVE_ROOT": str(self.archive_root), - "XDG_CONFIG_HOME": str(runtime_root / "config"), - "XDG_STATE_HOME": str(runtime_root / "state"), - "XDG_CACHE_HOME": str(runtime_root / "cache"), - } + runtime_workdir = TemporaryDirectory(prefix="polylogue-continuity-runtime-") + runtime_root = Path(runtime_workdir.name) + environment = _hermetic_stdio_environment(runtime_root=runtime_root, archive_root=self.archive_root) + server_expression = ( + "from devtools.continuity_replay import _serve_saturated_cancellation_probe; " + "_serve_saturated_cancellation_probe()" + if self.saturated_cancellation_probe + else "from polylogue.mcp.cli import main; main()" ) parameters = StdioServerParameters( command=sys.executable, - args=["-c", "from polylogue.mcp.cli import main; main()"], + args=["-c", server_expression], env=environment, cwd=str(Path.cwd()), ) @@ -360,6 +438,7 @@ async def __aenter__(self) -> StdioMCPContinuityRoute: break except Exception as exc: await stack.aclose() + runtime_workdir.cleanup() raise ContinuityReplayError( f"failed to initialize MCP stdio route: {exc}", kind="stdio_initialization_failed", @@ -368,6 +447,7 @@ async def __aenter__(self) -> StdioMCPContinuityRoute: self._stack = stack self._session = session + self._runtime_workdir = runtime_workdir self._discovery = _stdio_discovery(initialize, tools, self.transport_name) if self.discovery_mutator is not None: self._discovery = require_json_document( @@ -378,10 +458,16 @@ async def __aenter__(self) -> StdioMCPContinuityRoute: async def __aexit__(self, exc_type: object, exc: object, traceback: object) -> None: stack = self._stack + runtime_workdir = self._runtime_workdir self._stack = None self._session = None - if stack is not None: - await stack.aclose() + self._runtime_workdir = None + try: + if stack is not None: + await stack.aclose() + finally: + if runtime_workdir is not None: + runtime_workdir.cleanup() async def invoke(self, tool: str, arguments: Mapping[str, object]) -> str: session = self._session @@ -411,6 +497,35 @@ async def invoke(self, tool: str, arguments: Mapping[str, object]) -> str: async def exercise_cancellation( self, tool: str, arguments: Mapping[str, object], *, grace_ms: int + ) -> CancellationExerciseReceipt: + """Exercise cancellation once per interruptible production tool. + + Cancellation deliberately stresses request lifecycle and transport + teardown. A failed or partially supported cancellation must not + poison the session used for the graded continuity query that follows. + Keep the probe on the same production server route and archive, but + isolate its connection authority from the measured scenario. The + cancellation boundary is the registered tool route, not scenario + wording, so scenarios sharing one tool reuse its exact receipt instead + of launching equivalent MCP subprocesses. + """ + if cached := self._cancellation_receipts.get(tool): + return cached + async with StdioMCPContinuityRoute( + self.archive_root, + read_timeout_seconds=self.read_timeout_seconds, + saturated_cancellation_probe=True, + ) as probe: + receipt = await probe._exercise_cancellation_on_open_session( + tool, + arguments, + grace_ms=grace_ms, + ) + self._cancellation_receipts[tool] = receipt + return receipt + + async def _exercise_cancellation_on_open_session( + self, tool: str, arguments: Mapping[str, object], *, grace_ms: int ) -> CancellationExerciseReceipt: """Drive real MCP ``notifications/cancelled`` messages through the live stdio session and observe whether the production @@ -419,35 +534,11 @@ async def exercise_cancellation( admission queue's own cancellation check while a call is still waiting for a free slot) actually aborted an in-flight read. - This does not race a single call's wall-clock duration against a - settle delay: direct probing showed that approach is genuinely - flaky -- some scenarios' own real queries (a marker lookup with - ``limit=2``) complete in well under a millisecond end to end, so no - fixed settle window reliably wins, while heavier queries reliably - do; picking one or the other per scenario is not honest. Instead - this issues ``_CANCELLATION_PROBE_CONCURRENCY`` concurrent copies of - the SAME real (tool, arguments) call -- comfortably more than the - production admission controller's default capacity - (``DEFAULT_CAPACITY``, currently 4) -- and sends a cancellation - notification for every one of them. Regardless of how fast any - individual copy's own SQL work would complete, the ones that exceed - the shared admission ceiling are provably still queued (not yet - admitted, not yet touching SQLite) when the notifications are sent, - and the admission wait loop checks ``ctx.should_abort()`` on its own - short poll cadence -- making at least one confirmed cancellation - deterministic, not a race. Direct probing confirmed 100% success - (40/40 across 5 rounds) against the checked-in continuity fixture, - including under the same logging load that made the single-call - approach flake. - - This does not reuse :meth:`invoke`: it issues its own calls directly - against the session so the exercise never consumes a mutation - invocation slot or perturbs ``_BudgetState`` call/byte accounting -- - it is a probe, not a graded plan step. Request ids are predicted from - the session's own sequential counter read once before any of the - concurrent tasks are created; this is only valid because the SDK - assigns ids in task-creation order and the replay harness does not - interleave unrelated requests on this session while the probe runs. + The disposable server has its sole production admission slot occupied + before MCP startup, so this single real route call cannot complete + before cancellation. That removes both timing sleeps and request-id + swarms while still exercising FastMCP, the registered Polylogue tool, + QueryExecutionContext, and QueryAdmissionController. """ session = self._session if session is None: @@ -463,70 +554,69 @@ async def exercise_cancellation( call_arguments = dict(arguments) started = time.perf_counter() - base_request_id = session._request_id - probe_tasks: list[asyncio.Task[object]] = [ - asyncio.ensure_future(session.call_tool(tool, arguments=call_arguments)) - for _ in range(_CANCELLATION_PROBE_CONCURRENCY) - ] - await asyncio.sleep(0) # let every probe task reach its own request write - for offset in range(_CANCELLATION_PROBE_CONCURRENCY): - await session.send_notification( - ClientNotification( - CancelledNotification( - params=CancelledNotificationParams( - requestId=base_request_id + offset, - reason="continuity-replay-cancellation-exercise", - ) + request_id = session._request_id + probe_task = asyncio.create_task(session.call_tool(tool, arguments=call_arguments)) + for _ in range(3): + await asyncio.sleep(0) + if session._request_id == request_id + 1: + break + else: + raise ContinuityReplayError( + "MCP cancellation probe did not reserve its expected request id", + kind="cancellation_request_ids_unavailable", + failure_class="execution", + ) + await session.send_notification( + ClientNotification( + CancelledNotification( + params=CancelledNotificationParams( + requestId=request_id, + reason="continuity-replay-cancellation-exercise", ) ) ) + ) - confirmed_count = 0 - completed_count = 0 - failure_details: list[str] = [] - for probe_task in probe_tasks: - try: - await asyncio.wait_for(probe_task, timeout=max(1.0, grace_ms / 1000)) - except McpError as exc: - message = exc.error.message or "" - if exc.error.code == 0 and "cancel" in message.lower(): - confirmed_count += 1 - else: - failure_details.append(message or f"McpError code={exc.error.code}") - except TimeoutError: - probe_task.cancel() - with suppress(BaseException): - await probe_task - failure_details.append("probe call did not resolve within the grace budget") + confirmed = False + outcome: CancellationOutcome = "call_failed" + detail: str | None = None + try: + await asyncio.wait_for(probe_task, timeout=max(1.0, grace_ms / 1000)) + except McpError as exc: + message = exc.error.message or "" + if exc.error.code == 0 and "cancel" in message.lower(): + confirmed = True + outcome = "cancelled_confirmed" + detail = "queued production MCP read returned the SDK cancellation response" else: - completed_count += 1 + detail = message or f"McpError code={exc.error.code}" + except TimeoutError: + probe_task.cancel() + with suppress(BaseException): + await probe_task + outcome = "not_confirmed_within_grace" + detail = "queued production MCP read did not resolve within the cancellation grace budget" + else: + outcome = "completed_before_cancel" + detail = "saturated production admission unexpectedly allowed the probe to complete" + + # A cancellation response proves the handler stopped, but it does not + # prove the stdio server has consumed every preceding cancellation + # notification. Closing the disposable transport at that point can + # race the SDK's stdin reader: it reads one last notification after + # the server-side memory stream has closed and exits with + # BrokenResourceError. A protocol ping is an ordered transport + # barrier. Its response proves the server consumed all prior input + # and remains healthy before teardown; a broken session therefore + # remains an honest call failure instead of being hidden as a + # confirmed cancellation. + await session.send_ping() + elapsed_ms = (time.perf_counter() - started) * 1000 - detail = ( - f"{confirmed_count}/{_CANCELLATION_PROBE_CONCURRENCY} concurrent probe calls confirmed cancelled, " - f"{completed_count} completed normally" - ) - if failure_details: - detail += f"; {len(failure_details)} unexpected outcome(s): {failure_details[:3]}" - if confirmed_count > 0: - return CancellationExerciseReceipt( - attempted=True, - confirmed=True, - outcome="cancelled_confirmed", - elapsed_ms=elapsed_ms, - detail=detail, - ) - if completed_count == _CANCELLATION_PROBE_CONCURRENCY: - return CancellationExerciseReceipt( - attempted=True, - confirmed=False, - outcome="completed_before_cancel", - elapsed_ms=elapsed_ms, - detail=detail, - ) return CancellationExerciseReceipt( attempted=True, - confirmed=False, - outcome="call_failed", + confirmed=confirmed, + outcome=outcome, elapsed_ms=elapsed_ms, detail=detail, ) diff --git a/devtools/coordination_latency_probe.py b/devtools/coordination_latency_probe.py deleted file mode 100644 index 40ce01039e..0000000000 --- a/devtools/coordination_latency_probe.py +++ /dev/null @@ -1,94 +0,0 @@ -"""Measure compact coordination-envelope latency with stage attribution. - -Use ``devtools bench coordination-latency --samples 21 --out PATH`` to write a -portable raw artifact. Samples are intentionally taken through the production -envelope builder; each carries the complete per-stage timing map and the report -adds p50/p95 rather than hiding tail latency behind a single average. -""" - -from __future__ import annotations - -import argparse -import json -from datetime import UTC, datetime -from pathlib import Path -from time import perf_counter -from typing import Any - - -def _percentile(values: list[float], percentile: float) -> float: - ordered = sorted(values) - rank = max(1, min(len(ordered), round(percentile / 100 * len(ordered)))) - return ordered[rank - 1] - - -def measure(*, samples: int, cwd: Path | None = None) -> dict[str, Any]: - """Return raw compact samples and an honest distribution summary.""" - - from polylogue.coordination.envelope import build_coordination_envelope - from polylogue.paths import archive_root - - root = (cwd or Path.cwd()).resolve() - raw: list[dict[str, object]] = [] - latencies: list[float] = [] - for number in range(samples): - stages: dict[str, float] = {} - started = perf_counter() - payload = build_coordination_envelope(cwd=root, stage_timings_ms=stages) - latency_ms = round((perf_counter() - started) * 1_000, 3) - latencies.append(latency_ms) - raw.append( - { - "sample": number, - "latency_ms": latency_ms, - "stages_ms": stages, - "serialized_bytes": payload.projection.serialized_bytes, - } - ) - return { - "version": 1, - "generated_at": datetime.now(UTC).isoformat(), - "git_head": _git_head(root), - "cwd": str(root), - "archive_state": _archive_state(archive_root()), - "mode": "warm-in-process-core", - "samples": raw, - "distribution_ms": {"p50": _percentile(latencies, 50), "p95": _percentile(latencies, 95)}, - } - - -def _archive_state(root: Path) -> dict[str, object]: - """Record non-content archive facts needed to compare local samples.""" - - index = root / "index.db" - try: - index_bytes: int | None = index.stat().st_size - except OSError: - index_bytes = None - return {"root": str(root), "index_exists": index.exists(), "index_bytes": index_bytes} - - -def _git_head(cwd: Path) -> str | None: - import subprocess - - completed = subprocess.run(["git", "rev-parse", "HEAD"], cwd=cwd, text=True, capture_output=True, check=False) - return completed.stdout.strip() or None - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--samples", type=int, default=21) - parser.add_argument("--cwd", type=Path, default=None) - parser.add_argument("--out", type=Path, default=None, help="Write the full raw JSON artifact to this path.") - args = parser.parse_args(argv) - report = measure(samples=max(1, args.samples), cwd=args.cwd) - encoded = json.dumps(report, indent=2, sort_keys=True) - if args.out is not None: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(encoded + "\n", encoding="utf-8") - print(encoded) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/devtools/daemon_workload_probe.py b/devtools/daemon_workload_probe.py index c10680f6ff..cc322abd06 100644 --- a/devtools/daemon_workload_probe.py +++ b/devtools/daemon_workload_probe.py @@ -21,7 +21,6 @@ from pathlib import Path from typing import Any -from devtools.preflight_ledger import build_preflight_ledger from polylogue.config import Config from polylogue.paths import archive_root from polylogue.storage.archive_identity import resolve_active_index_path @@ -33,7 +32,7 @@ # Bumped when the JSON shape gains new top-level keys or changes a field type. # The compare path uses this to refuse incompatible inputs loudly. -REPORT_VERSION = 20 # v20 adds the opt-in read-only preflight ledger. +REPORT_VERSION = 19 UNKNOWN_TABLE_COUNT = -2 _EXPECTED_FTS_TRIGGERS: tuple[str, ...] = ("messages_fts_ai", "messages_fts_ad", "messages_fts_au") @@ -2356,7 +2355,6 @@ def probe( exact_derived_counts: bool = False, exact_table_counts: bool = False, blob_reference_debt: bool = False, - preflight: bool = False, ) -> dict[str, Any]: ops_db = db.with_name("ops.db") index_db = db.with_name("index.db") @@ -2420,11 +2418,6 @@ def probe( "convergence_debt": convergence_debt, "cursor_lag_baselines": _cursor_lag_baselines(conn, ops_db=ops_db), "query_plans": _query_plans(conn, db=db), - "preflight_ledger": ( - build_preflight_ledger(db.parent, limit=limit) - if preflight - else {"requested": False, "read_only": True, "state": "not_requested", "mutation_operations": []} - ), } finally: conn.close() @@ -2988,11 +2981,6 @@ def _parser() -> argparse.ArgumentParser: action="store_true", help="Count missing referenced blob files exactly (can stat many blob paths on large archives)", ) - parser.add_argument( - "--preflight", - action="store_true", - help="Add the strict read-only deployed-status preflight ledger (exact source/FTS/frontier reads)", - ) parser.add_argument( "--compare", nargs=2, @@ -3028,7 +3016,6 @@ def main(argv: list[str] | None = None) -> int: exact_derived_counts=args.exact_derived_counts, exact_table_counts=args.exact_table_counts, blob_reference_debt=args.blob_reference_debt, - preflight=args.preflight, ) if args.json: print(json.dumps(payload, indent=2, sort_keys=True)) @@ -3039,32 +3026,6 @@ def main(argv: list[str] | None = None) -> int: if not payload.get("ok"): print(f" error: {payload.get('error')}") return 1 - preflight = payload.get("preflight_ledger") or {} - if preflight.get("requested") is not False: - print( - " preflight: " - f"{preflight.get('state')} " - f"(blocking={len(preflight.get('blocking_checks') or [])}, " - f"warnings={len(preflight.get('warning_checks') or [])})" - ) - source = (preflight.get("checks") or {}).get("source") or {} - totals = source.get("totals") or {} - if totals: - print( - " source raw: " - f"{totals.get('raw_count', 0)} total, " - f"{totals.get('quarantined_count', 0)} quarantined, " - f"{totals.get('missing_census_count', 0)} missing census" - ) - for item in source.get("by_origin") or []: - quarantine = item.get("quarantine") or {} - census = item.get("census_coverage") or {} - print( - f" {item.get('origin')}: quarantine={quarantine.get('count', 0)} " - f"({(quarantine.get('size') or {}).get('display', '0.0GiB')}), " - f"census={census.get('missing_count', 0)} missing " - f"({(census.get('missing_size') or {}).get('display', '0.0GiB')})" - ) counts = payload["attempt_counts"] print(f" attempts: {counts['total']} total, {counts['running']} running, {counts['failed']} failed") print(f" stale cursor writes: {counts.get('stale_cursor_writes', 0)}") diff --git a/devtools/data/bead-lint-allow.txt b/devtools/data/bead-lint-allow.txt deleted file mode 100644 index 62814851c3..0000000000 --- a/devtools/data/bead-lint-allow.txt +++ /dev/null @@ -1,12 +0,0 @@ -# Backlog-hygiene allowlist for devtools.verify_backlog_hygiene. -# Format: CHECKBEAD_ID (one finding per line; keep the reason as a comment above it) -# -# D1 dangling-dep false positives: bd export emits an orphaned raw dependency -# row (depends_on_id without the "polylogue-" prefix) alongside the correct, -# re-added edge. bd's own query tools (bd show, bd dep list) already report -# only the single correct edge -- this is an export-layer artifact, not a -# real dangling reference. bd dep remove does not clear the underlying row -# (Dolt-backed store), so hand-editing the JSONL gets clobbered by the next -# pre-commit bd export. Revisit if bd's export/remove behavior changes. -D1 polylogue-jtwu -D1 polylogue-fko9 diff --git a/devtools/data/tracker-authority.json b/devtools/data/tracker-authority.json deleted file mode 100644 index a6920ad55b..0000000000 --- a/devtools/data/tracker-authority.json +++ /dev/null @@ -1,132 +0,0 @@ -{ - "version": 1, - "generated_at": "2026-07-26T19:05:00Z", - "relations": { - "gh_mirror": "The Bead and GitHub issue express the same outcome; scope and state must reconcile.", - "gh_public_parent": "GitHub owns the public outcome; this Bead owns the internal program or aggregate authority.", - "gh_implements": "This Bead is an executable child or proof slice under the named public GitHub outcome.", - "gh_supersedes_scope": "The Bead is the current implementation authority and deliberately replaces stale solution wording in the GitHub issue.", - "internal_only": "No public GitHub projection is required." - }, - "bindings": [ - { - "bead_id": "polylogue-fnm.4", - "github_issue": 1844, - "relation": "gh_mirror", - "external_ref": "gh-1844", - "note": "GitHub #1844 is the public and near-equivalent outcome for grammar-derived completion, fuzzy selection, and query-builder discovery." - }, - { - "bead_id": "polylogue-s8q", - "github_issue": 2308, - "relation": "gh_implements", - "external_ref": "gh-2308", - "note": "This Bead is the deployed-version/schema/origin attestation slice under the broader live archive closure gate in GitHub #2308; it is not a mirror of the whole issue." - }, - { - "bead_id": "polylogue-lkrc", - "github_issue": 2308, - "relation": "gh_implements", - "note": "Canonical raw-authority reconciler implementation under GitHub #2308." - }, - { - "bead_id": "polylogue-hjpx", - "github_issue": 2308, - "relation": "gh_implements", - "note": "Fixed-point replay-plan execution slice under GitHub #2308." - }, - { - "bead_id": "polylogue-yla8", - "github_issue": 2308, - "relation": "gh_implements", - "note": "Operator-authorized live replay and closure proof under GitHub #2308." - }, - { - "bead_id": "polylogue-jnj.9", - "github_issue": 2309, - "relation": "gh_implements", - "external_ref": "gh-2309", - "note": "This Bead owns effective-config inventory and provenance. GitHub #2309 remains open for residual governance, unsafe-combination rejection, and ownership simplification." - }, - { - "bead_id": "polylogue-5hf", - "github_issue": 2316, - "relation": "gh_mirror", - "external_ref": "gh-2316", - "note": "GitHub #2316 and this Bead jointly own honest provider usage, billing semantics, coverage, and reconciliation." - }, - { - "bead_id": "polylogue-rii", - "github_issue": 2384, - "relation": "gh_public_parent", - "external_ref": "gh-2384", - "note": "Public closure now requires a real emit -> query -> evidence-backed context consumption loop, not merely additional substrate architecture." - }, - { - "bead_id": "polylogue-rii.1", - "github_issue": 2384, - "relation": "gh_implements", - "external_ref": "gh-2384", - "note": "Internal live work-event write-leg owner. Former separate public issue #2459 was closed as duplicate scope; this Bead remains open until its acceptance criteria are satisfied." - }, - { - "bead_id": "polylogue-rii.2", - "github_issue": 2384, - "relation": "gh_implements", - "external_ref": "gh-2384", - "note": "Internal hook/OTLP evidence-projection owner. Former separate public issue #2461 was closed as duplicate scope; recognized hook raw materialization has advanced, while full projection/coverage and OTLP durability remain to prove." - }, - { - "bead_id": "polylogue-20d.6", - "github_issue": 2391, - "relation": "gh_mirror", - "external_ref": "gh-2391", - "note": "Remeasure current-head full-corpus ingest/catch-up performance before changing code; historical June timings are evidence, not current truth." - }, - { - "bead_id": "polylogue-fs1", - "github_issue": 2460, - "relation": "gh_supersedes_scope", - "external_ref": "gh-2460", - "title": "Bridge Hermes runtime evidence into queryable work graphs and forensic reports", - "description": "Ingest the strongest available Hermes runtime/state, observer/span, transcript, and repository evidence into a deterministic queryable work graph, then generate a citable run-forensics artifact with explicit coverage gaps. Parser-only snapshot enrichment is optional input work, not the authority model. Polylogue records evidence and projections; Hermes retains mutable runtime task authority.", - "acceptance_criteria": "At least one real Hermes corpus covers delegation, task lifecycle, retry/reroute or failure, skill/tool use, and continuation where available. Evidence from state/runtime/transcript sources is admitted with stable identity, timestamps, source provenance, and explicit caveats. Parent/child and continuation structure and task/tool/skill/run events are queryable without manufactured edges. Reingestion is deterministic and idempotent. A generated run-forensics artifact links every material claim to evidence refs and lets a fresh reviewer identify the main branch, failed or wasted branches, current state, and next justified action. Missing sources degrade visibly rather than producing a complete-looking graph.", - "note": "This Bead deliberately supersedes the original parse_hermes-only solution in GitHub #2460; the public issue was rewritten to match this broader authority model." - }, - { - "bead_id": "polylogue-4ts.5", - "github_issue": 2478, - "relation": "gh_mirror", - "external_ref": "gh-2478", - "note": "GitHub #2478 and this Bead jointly own precise compaction boundary ranges and effective-context derivation." - }, - { - "bead_id": "polylogue-z9gh.9", - "github_issue": 3283, - "relation": "gh_public_parent", - "external_ref": "gh-3283", - "note": "GitHub #3283 is the public outcome for bounded, cancellable, resumable archive reads across every transport." - }, - { - "bead_id": "polylogue-z9gh.9.1", - "github_issue": 3283, - "relation": "gh_implements", - "external_ref": "gh-3283", - "note": "Delivery owner for landing the shared query transaction across CLI, MCP, Python, daemon HTTP, and web." - }, - { - "bead_id": "polylogue-z9gh.1", - "github_issue": 3283, - "relation": "gh_implements", - "external_ref": "gh-3283", - "note": "Interruptibility, deadlines, cancellation, and resource-bounding repair slice under GitHub #3283." - }, - { - "bead_id": "polylogue-z9gh.2", - "github_issue": 3283, - "relation": "gh_implements", - "external_ref": "gh-3283", - "note": "Archive-wide action/delegation materialization repair slice under GitHub #3283." - } - ] -} diff --git a/devtools/delivery_gate_status.py b/devtools/delivery_gate_status.py deleted file mode 100644 index 391f26e47f..0000000000 --- a/devtools/delivery_gate_status.py +++ /dev/null @@ -1,231 +0,0 @@ -"""delivery-gate-status: per-release-gate progress board over .beads/issues.jsonl. - -The delivery overlay (2026-07-07, corpus escrowed at -.agent/scratch/corpus-gpt-pro-2026-07-07/) labels every active bead with a -delivery: gate and lane:. This computes, per gate: open / -in_progress / closed / blocked / ready counts, percent complete, and the -gate's exit criterion (embedded below — the corpus copy is ephemeral). - -Gate ORDER is the implementation sequence: a gate is "up next" when every -earlier gate is complete or explicitly waived. Exit criteria are prose — -they need human/agent judgment, so the board prints them as reminders, not -as computed booleans. - -Usage: devtools workspace delivery-gate-status [--json] [--fresh] - [--gate delivery:A-trust-floor] [path-to-issues.jsonl] ---fresh runs `bd export -o .beads/issues.jsonl` first (bd updates do NOT -immediately re-export; a stale file yields stale counts). -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -from collections import defaultdict -from pathlib import Path -from typing import Any - -from devtools import repo_root as _get_root - -IssueDict = dict[str, Any] - -# (gate-id, short name, exit criterion) in implementation order. -# Source: delivery overlay release_gates (adjudicated 2026-07-07); durable copy. -GATES: list[tuple[str, str, str]] = [ - ( - "R0-normalize", - "Backlog normalization", - "Every active bead has a delivery release, lane, readiness grade, proof lane, and either acceptance criteria or a deliberate horizon/spec status.", - ), - ( - "A-trust-floor", - "Trust floor", - "Full verification classified; security negative tests pass; missing bytes classified; numbers/time/prose-mined fields carry honest provenance; agent writes land as candidates.", - ), - ( - "B-storage-rebuild-bytes", - "Storage/rebuild/bytes", - "Blue-green derived rebuilds cannot show partial archives as ready; restore drill passes; blob refs resolve or carry classified missing state.", - ), - ( - "C-read-evidence-contract", - "Read + evidence contract", - "CLI, daemon, MCP, Python API, web, reports, and docs read through the same contract; content-hash citations expose drift states.", - ), - ( - "D-agent-context-coordination", - "Agent context/coordination", - "Hooks install; MCP roles/prompts discoverable; context scheduler emits ledgers; two-agent worktree proof exists.", - ), - ( - "E-variants-preferences", - "Variants + preferences", - "Variant refs/nodes/alignment/storage exist; reader/query/projection can show source, variant, and side-by-side views honestly.", - ), - ( - "F-lineage-compaction", - "Lineage + compaction", - "Shared content is stored/counted once, compaction loss is queryable, and regrounding packs pass through the context scheduler.", - ), - ( - "G-live-performance", - "Live performance", - "Named SLOs and regression gates exist; daemon push/live cache paths work; capture reliability is visible and tested.", - ), - ( - "H-web-cockpit", - "Web evidence cockpit", - "Evidence basket to citable export works; web UI shows stale/partial/degraded states instead of pretending readiness.", - ), - ( - "I-analytics-experiments", - "Analytics + experiments", - "Measures are registered, experiments are first-class, analytics render caveats and evidence tiers.", - ), - ( - "J-embeddings-retrieval", - "Embeddings + retrieval", - "FTS/vector/hybrid quality evals exist; local/cloud providers share one interface; large sessions bound vector work.", - ), - ( - "K-interop-origin-export", - "Interop + origin + export", - "Each origin has detector/parser/fixtures/fidelity docs; Polylogue export/import is content-hash idempotent; outbound citations preserve provenance.", - ), - ( - "L-external-legibility", - "External legibility", - "README first screen is clear; one-command demo passes; public claims ledger covers launch claims; cold-reader proof passes.", - ), - ( - "M-substrate-consolidation", - "Substrate consolidation", - "Storage twins collapse behind a clear boundary; public models are frozen; dead abstractions are deleted or adopted.", - ), - ( - "N-horizon", - "Horizon", - "Each item either has an implementation-grade spec or remains parked with a decision record.", - ), -] -GATE_IDS = [g[0] for g in GATES] - - -def load(path: Path) -> tuple[dict[str, IssueDict], list[tuple[str, str, str]]]: - issues: dict[str, IssueDict] = {} - deps: list[tuple[str, str, str]] = [] - for line in path.read_text().splitlines(): - if not line.strip(): - continue - d = json.loads(line) - if d.get("_type") == "issue": - issues[d["id"]] = d - for dep in d.get("dependencies") or []: - deps.append((d["id"], dep.get("depends_on_id"), dep.get("type", "blocks"))) - elif d.get("_type") == "dependency": - deps.append((d.get("issue_id"), d.get("depends_on_id"), d.get("type", "blocks"))) - return issues, deps - - -def main(argv: list[str] | None = None) -> int: - root = _get_root() - parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("path", nargs="?", default=None, help="path to issues.jsonl (default: .beads/issues.jsonl)") - parser.add_argument("--json", action="store_true") - parser.add_argument("--fresh", action="store_true", help="bd export first") - parser.add_argument("--gate", help="only this gate (accepts 'A-trust-floor' or 'delivery:A-trust-floor')") - args = parser.parse_args(argv) - - path = Path(args.path) if args.path else root / ".beads/issues.jsonl" - - if args.fresh: - subprocess.run(["bd", "export", "-o", str(path)], check=True, capture_output=True) - - issues, deps = load(path) - blockers: dict[str, list[str]] = defaultdict(list) - for src, dst, kind in deps: - if kind == "blocks" and dst in issues: - blockers[src].append(dst) - - def gate_of(d: IssueDict) -> str | None: - for lab in d.get("labels") or []: - lab_str = str(lab) - # delivery:ac-patched is an overlay marker, not a gate assignment - if lab_str.startswith("delivery:") and lab_str != "delivery:ac-patched": - return lab_str.removeprefix("delivery:") - return None - - by_gate: dict[str, list[IssueDict]] = defaultdict(list) - unlabeled_open = 0 - for d in issues.values(): - g = gate_of(d) - if g is None: - if d.get("status") in ("open", "in_progress"): - unlabeled_open += 1 - continue - by_gate[g].append(d) - - want = args.gate.removeprefix("delivery:") if args.gate else None - rows: list[dict[str, Any]] = [] - for gid, name, exit_crit in GATES: - if want and gid != want: - continue - beads = by_gate.get(gid, []) - closed = [b for b in beads if b["status"] == "closed"] - in_prog = [b for b in beads if b["status"] == "in_progress"] - open_ = [b for b in beads if b["status"] == "open"] - blocked = [b for b in open_ if any(issues[x]["status"] != "closed" for x in blockers.get(b["id"], []))] - ready = [b for b in open_ if b not in blocked] - rows.append( - { - "gate": gid, - "name": name, - "total": len(beads), - "closed": len(closed), - "in_progress": len(in_prog), - "ready": len(ready), - "blocked": len(blocked), - "pct": round(100 * len(closed) / len(beads)) if beads else None, - "exit": exit_crit, - "ready_ids": sorted(b["id"] for b in ready)[:12], - "in_progress_ids": sorted(b["id"] for b in in_prog), - } - ) - - unknown_gates = sorted(set(by_gate) - set(GATE_IDS)) - - if args.json: - print(json.dumps({"gates": rows, "unlabeled_open": unlabeled_open, "unknown_gates": unknown_gates}, indent=2)) - return 0 - - frontier_shown = False - for r in rows: - if r["total"] == 0: - bar = "(no beads)" - else: - done = int(round((r["pct"] or 0) / 10)) - bar = "#" * done + "." * (10 - done) + f" {r['pct']:>3}%" - marker = " " - if not frontier_shown and r["total"] and r["closed"] < r["total"]: - marker = ">" # first incomplete gate = active frontier - frontier_shown = True - print( - f"{marker} {r['gate']:<32} {bar} closed {r['closed']:>3} | wip {r['in_progress']:>2} | ready {r['ready']:>3} | blocked {r['blocked']:>3}" - ) - if marker == ">" or (want and rows): - print(f" exit: {r['exit']}") - if r["in_progress_ids"]: - print(f" wip: {', '.join(r['in_progress_ids'])}") - if r["ready_ids"]: - print(f" ready: {', '.join(r['ready_ids'])}") - if unlabeled_open: - print(f"\n {unlabeled_open} open/in_progress beads carry no delivery:* label") - if unknown_gates: - print(f" labels outside the gate registry: {', '.join(unknown_gates)}") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/devtools/demo_packet.py b/devtools/demo_packet.py deleted file mode 100644 index 5a33e41040..0000000000 --- a/devtools/demo_packet.py +++ /dev/null @@ -1,656 +0,0 @@ -"""Demo Packet v2 validation and registry policy (polylogue-212.12). - -The original Demo Finding Packet established a uniform directory shape. V2 -adds an epistemic contract: a demo is not conforming unless it predeclares one -primary construct and a receipt-backed claim, provides an independently -checkable oracle, baseline, negative and missing-evidence controls, explicit -falsifier, content-bound receipts, and bounded non-claims. - -The normative schema lives at ``docs/schemas/demo-packet-v2.schema.json``. -This module intentionally keeps the existing human packet files too: JSON is -the gate, Markdown/NDJSON are the inspectable publication surface. -""" - -from __future__ import annotations - -import hashlib -import json -from collections.abc import Iterable, Mapping -from dataclasses import dataclass, field -from pathlib import Path, PurePosixPath - -PROVENANCE_STANZA_FIELDS: tuple[str, ...] = ( - "archive_cursor", - "measure_version", - "commit_sha", - "sample_frame_predicate", - "run_date", -) - -REPORT_SECTION_ORDER: tuple[str, ...] = ( - "claim", - "corpus", - "method", - "findings", - "specimens", - "counterexamples", - "limits", - "non-claims", - "reproduce", -) - -PACKET_FILENAMES: tuple[str, ...] = ( - "PROMPT.md", - "packet.json", - "finding.yaml", - "report.md", - "evidence.ndjson", - "queries.ndjson", - "checks.json", - "NON-CLAIMS.md", - "run.log", -) -#: Optional -- present only when an external-agent annotation loop ran. -OPTIONAL_PACKET_FILENAMES: tuple[str, ...] = ("annotations.ndjson",) -DEFAULT_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "docs" / "schemas" / "demo-packet-v2.schema.json" -DEFAULT_DEMO_ROOT = Path(".agent/demos") - - -class DemoPacketValidationError(ValueError): - """Raised when a packet directory or registry entry fails the contract.""" - - -@dataclass(frozen=True) -class PacketValidationResult: - """Outcome of validating one packet directory.""" - - packet_dir: Path - ok: bool - missing_files: tuple[str, ...] = () - missing_stanza_fields: tuple[str, ...] = () - malformed_sections: tuple[str, ...] = () - schema_errors: tuple[str, ...] = () - receipt_errors: tuple[str, ...] = () - errors: tuple[str, ...] = field(default_factory=tuple) - - def to_dict(self) -> dict[str, object]: - return { - "packet_dir": str(self.packet_dir), - "ok": self.ok, - "missing_files": list(self.missing_files), - "missing_stanza_fields": list(self.missing_stanza_fields), - "malformed_sections": list(self.malformed_sections), - "schema_errors": list(self.schema_errors), - "receipt_errors": list(self.receipt_errors), - "errors": list(self.errors), - } - - -def _parse_minimal_yaml_mapping(text: str) -> dict[str, str]: - """Parse the packet's flat provenance stanza without a full YAML loader.""" - - result: dict[str, str] = {} - for raw_line in text.splitlines(): - line = raw_line.split("#", 1)[0].strip() - if not line or ":" not in line: - continue - key, _, value = line.partition(":") - result[key.strip()] = value.strip().strip("\"'") - return result - - -def _validate_ndjson(path: Path) -> str | None: - """Return an error string if *path* is not valid line-delimited JSON.""" - - for line_no, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1): - if not line.strip(): - continue - try: - json.loads(line) - except json.JSONDecodeError as exc: - return f"{path.name}:{line_no}: invalid JSON ({exc})" - return None - - -def _json_pointer(error: object) -> str: - absolute_path = getattr(error, "absolute_path", ()) - parts = [str(part).replace("~", "~0").replace("/", "~1") for part in absolute_path] - return "/" + "/".join(parts) if parts else "/" - - -def _validate_schema(payload: object, *, schema_path: Path) -> tuple[str, ...]: - try: - schema = json.loads(schema_path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError) as exc: - return (f"schema unavailable at {schema_path}: {exc}",) - - try: - from jsonschema import Draft202012Validator, FormatChecker - except ImportError: - return ("jsonschema is required for the Demo Packet v2 policy gate",) - - validator = Draft202012Validator(schema, format_checker=FormatChecker()) - errors = sorted(validator.iter_errors(payload), key=lambda item: (list(item.absolute_path), item.message)) - return tuple(f"packet.json{_json_pointer(error)}: {error.message}" for error in errors) - - -def _receipt_artifact_path(packet_dir: Path, raw_path: object) -> tuple[Path | None, str | None]: - if not isinstance(raw_path, str) or not raw_path.strip(): - return None, "receipt artifact_path must be a non-empty string" - posix = PurePosixPath(raw_path) - if posix.is_absolute() or ".." in posix.parts: - return None, f"receipt artifact_path escapes packet directory: {raw_path!r}" - candidate = (packet_dir / Path(*posix.parts)).resolve() - packet_root = packet_dir.resolve() - if not candidate.is_relative_to(packet_root): - return None, f"receipt artifact_path escapes packet directory: {raw_path!r}" - return candidate, None - - -def _iter_referenced_receipts(value: object, *, at_root: bool = True) -> Iterable[str]: - """Yield every receipt ref used by an epistemic field in *value*. - - The top-level ``receipts`` member declares the resolver table and is not a - use site. Nested members named ``receipts`` are citations and must resolve - to that table. The recursive form keeps future packet sections honest - without requiring a second list of allowed citation locations. - """ - - if isinstance(value, Mapping): - for key, child in value.items(): - if key == "receipts" and not at_root and isinstance(child, list): - yield from (ref for ref in child if isinstance(ref, str)) - continue - yield from _iter_referenced_receipts(child, at_root=False) - elif isinstance(value, list): - for child in value: - yield from _iter_referenced_receipts(child, at_root=False) - - -def _validate_receipts(packet_dir: Path, payload: object) -> tuple[str, ...]: - if not isinstance(payload, Mapping): - return () - raw_receipts = payload.get("receipts") - if not isinstance(raw_receipts, list): - return () # The schema reports this more precisely. - - errors: list[str] = [] - seen_refs: set[str] = set() - for index, raw_receipt in enumerate(raw_receipts): - if not isinstance(raw_receipt, Mapping): - continue - ref = raw_receipt.get("ref") - if isinstance(ref, str): - if ref in seen_refs: - errors.append(f"receipt[{index}] duplicates ref {ref!r}") - seen_refs.add(ref) - kind = raw_receipt.get("kind") - expected_kind = ref.partition(":")[0] - if isinstance(kind, str) and kind != expected_kind: - errors.append(f"receipt[{index}] kind {kind!r} does not match ref prefix {expected_kind!r}") - artifact, artifact_error = _receipt_artifact_path(packet_dir, raw_receipt.get("artifact_path")) - if artifact_error is not None: - errors.append(f"receipt[{index}]: {artifact_error}") - continue - assert artifact is not None - if not artifact.is_file(): - errors.append(f"receipt[{index}] artifact missing: {artifact.relative_to(packet_dir.resolve())}") - continue - try: - artifact_bytes = artifact.read_bytes() - content = artifact_bytes.decode("utf-8") - except (OSError, UnicodeDecodeError) as exc: - errors.append(f"receipt[{index}] artifact unreadable: {exc}") - continue - if isinstance(ref, str) and ref not in content: - errors.append( - f"receipt[{index}] ref {ref!r} is not present in {artifact.relative_to(packet_dir.resolve())}" - ) - expected_sha = raw_receipt.get("sha256") - if isinstance(expected_sha, str): - actual_sha = hashlib.sha256(artifact_bytes).hexdigest() - if actual_sha != expected_sha: - errors.append( - f"receipt[{index}] sha256 mismatch for {artifact.name}: expected {expected_sha}, got {actual_sha}" - ) - - undeclared = sorted(set(_iter_referenced_receipts(payload)) - seen_refs) - errors.extend(f"referenced receipt is not declared in packet.receipts: {ref!r}" for ref in undeclared) - return tuple(errors) - - -def _canonical_report_heading(section: str) -> str: - return f"## {section[0].upper()}{section[1:]}" - - -def _report_headings(report_text: str) -> tuple[str, ...]: - """Return Markdown headings outside fenced code, preserving exact bytes.""" - - headings: list[str] = [] - fence: str | None = None - for line in report_text.splitlines(): - stripped = line.lstrip() - marker = stripped[:3] - if marker in {"```", "~~~"}: - if fence is None: - fence = marker - elif fence == marker: - fence = None - continue - if fence is None and line.startswith("#"): - headings.append(line) - return tuple(headings) - - -def _validate_report_sections(report_text: str) -> tuple[str, ...]: - """Require each canonical level-two heading exactly once and in order.""" - - headings = _report_headings(report_text) - malformed: list[str] = [] - positions: list[int] = [] - for section in REPORT_SECTION_ORDER: - canonical = _canonical_report_heading(section) - matches = [index for index, heading in enumerate(headings) if heading == canonical] - if len(matches) != 1: - malformed.append(section) - else: - positions.append(matches[0]) - if len(positions) == len(REPORT_SECTION_ORDER) and positions != sorted(positions): - malformed.append("section-order") - return tuple(malformed) - - -def _duplicate_strings(values: Iterable[object]) -> tuple[str, ...]: - seen: set[str] = set() - duplicates: set[str] = set() - for value in values: - if not isinstance(value, str): - continue - if value in seen: - duplicates.add(value) - seen.add(value) - return tuple(sorted(duplicates)) - - -def _validate_semantic_consistency(payload: object) -> tuple[str, ...]: - """Check cross-field invariants that JSON Schema cannot state clearly.""" - - if not isinstance(payload, Mapping): - return () - errors: list[str] = [] - - falsifier = payload.get("falsifier") - if isinstance(falsifier, Mapping): - triggered = falsifier.get("triggered") - result = falsifier.get("result") - if triggered is True and result != "fail": - errors.append("falsifier.triggered=true requires falsifier.result='fail'") - if result == "fail" and triggered is not True: - errors.append("falsifier.result='fail' requires falsifier.triggered=true") - if result == "pass" and triggered is not False: - errors.append("falsifier.result='pass' requires falsifier.triggered=false") - - controls = payload.get("controls") - if isinstance(controls, Mapping): - control_ids: list[object] = [] - for lane in ("negative", "missing_evidence"): - raw_controls = controls.get(lane) - if isinstance(raw_controls, list): - control_ids.extend(control.get("id") for control in raw_controls if isinstance(control, Mapping)) - for duplicate in _duplicate_strings(control_ids): - errors.append(f"control id is duplicated across packet controls: {duplicate!r}") - - results = payload.get("results") - if isinstance(results, Mapping): - measurements = results.get("measurements") - if isinstance(measurements, list): - names = [measurement.get("name") for measurement in measurements if isinstance(measurement, Mapping)] - for duplicate in _duplicate_strings(names): - errors.append(f"measurement name is duplicated: {duplicate!r}") - - return tuple(errors) - - -def _logged_output( - log_text: str, - *, - section: str, - command: str, - exit_marker: str, -) -> tuple[str, int] | str: - """Read one structurally delimited command result from a packet log.""" - - lines = log_text.splitlines(keepends=True) - heading = f"=== {section} ===" - heading_indexes = [index for index, line in enumerate(lines) if line.rstrip("\r\n") == heading] - if len(heading_indexes) != 1: - return f"run.log must contain exactly one section heading {heading!r}" - command_index = heading_indexes[0] + 1 - expected_command = f"$ {command}" - if command_index >= len(lines) or lines[command_index].rstrip("\r\n") != expected_command: - return f"run.log section {section!r} does not record command {command!r}" - output_start = command_index + 1 - exit_indexes = [index for index in range(output_start, len(lines)) if lines[index].startswith(f"{exit_marker}=")] - if len(exit_indexes) != 1: - return f"run.log section {section!r} must contain exactly one {exit_marker} marker" - exit_index = exit_indexes[0] - raw_exit = lines[exit_index].rstrip("\r\n").partition("=")[2] - try: - exit_code = int(raw_exit) - except ValueError: - return f"run.log marker {exit_marker!r} has a non-integer exit code" - return "".join(lines[output_start:exit_index]), exit_code - - -def _validate_current_run(packet_dir: Path, payload: object) -> tuple[str, ...]: - """Bind an opt-in current run receipt to its structured log evidence. - - Older packets intentionally have only historical provenance. A packet that - claims a full current SHA must opt into this receipt so the claim cannot be - satisfied by editing a command transcript alone. - """ - - if not isinstance(payload, Mapping): - return () - provenance = payload.get("provenance") - if not isinstance(provenance, Mapping): - return () - commit_sha = provenance.get("commit_sha") - current_run = provenance.get("current_run") - if current_run is None: - if payload.get("packet_id") == "d4-behavioral-archaeology": - return ("provenance.current_run is required for the D4 current receipt",) - return () - if not isinstance(current_run, Mapping): - return ("provenance.current_run must be an object",) - - errors: list[str] = [] - if current_run.get("code_sha") != commit_sha: - errors.append("provenance.current_run.code_sha must equal provenance.commit_sha") - if current_run.get("route") != "polylogue": - errors.append("provenance.current_run.route must be 'polylogue'") - if current_run.get("private_data") is not False: - errors.append("provenance.current_run.private_data must be false") - - run_log = current_run.get("run_log") - log_path, path_error = _receipt_artifact_path(packet_dir, run_log) - if path_error is not None: - errors.append(f"provenance.current_run: {path_error}") - return tuple(errors) - assert log_path is not None - if not log_path.is_file(): - return (f"provenance.current_run log missing: {run_log}",) - try: - log_text = log_path.read_text(encoding="utf-8") - except OSError as exc: - return (f"provenance.current_run log unreadable: {exc}",) - - raw_commands = current_run.get("commands") - if not isinstance(raw_commands, list): - return tuple(errors) - command_ids = [item.get("id") for item in raw_commands if isinstance(item, Mapping)] - if command_ids != ["q1", "q2", "q3", "q4", "q5", "q6"]: - errors.append("provenance.current_run.commands must be ordered q1 through q6") - for item in raw_commands: - if not isinstance(item, Mapping): - continue - result = _logged_output( - log_text, - section=str(item.get("section", "")), - command=str(item.get("command", "")), - exit_marker=f"{str(item.get('id', '')).upper()}_EXIT_CODE", - ) - if isinstance(result, str): - errors.append(result) - continue - output, exit_code = result - if exit_code != item.get("exit_code"): - errors.append(f"run.log exit code does not match current_run command {item.get('id')!r}") - actual_sha = hashlib.sha256(output.encode("utf-8")).hexdigest() - if actual_sha != item.get("output_sha256"): - errors.append(f"run.log output digest does not match current_run command {item.get('id')!r}") - - explain = current_run.get("explain") - if isinstance(explain, Mapping): - result = _logged_output( - log_text, - section=str(explain.get("section", "")), - command=str(explain.get("command", "")), - exit_marker="EXPLAIN_EXIT_CODE", - ) - if isinstance(result, str): - errors.append(result) - else: - output, exit_code = result - if exit_code != explain.get("exit_code"): - errors.append("run.log explain exit code does not match provenance") - actual_sha = hashlib.sha256(output.encode("utf-8")).hexdigest() - if actual_sha != explain.get("output_sha256"): - errors.append("run.log explain output digest does not match provenance") - return tuple(errors) - - -def validate_packet(packet_dir: Path, *, schema_path: Path = DEFAULT_SCHEMA_PATH) -> PacketValidationResult: - """Validate *packet_dir* against both the human and v2 machine contracts. - - The function is read-only. Receipt resolution is intentionally bounded to - committed artifacts inside the packet directory; generation-time code may - additionally resolve the same refs against a live deterministic archive. - """ - - missing_files = tuple(name for name in PACKET_FILENAMES if not (packet_dir / name).exists()) - errors: list[str] = [] - missing_stanza_fields: tuple[str, ...] = () - malformed_sections: list[str] = [] - schema_errors: tuple[str, ...] = () - receipt_errors: tuple[str, ...] = () - packet_payload: object = None - - packet_path = packet_dir / "packet.json" - if packet_path.exists(): - try: - packet_payload = json.loads(packet_path.read_text(encoding="utf-8")) - except json.JSONDecodeError as exc: - errors.append(f"packet.json is not valid JSON: {exc}") - else: - schema_errors = _validate_schema(packet_payload, schema_path=schema_path) - receipt_errors = _validate_receipts(packet_dir, packet_payload) - errors.extend(_validate_semantic_consistency(packet_payload)) - errors.extend(_validate_current_run(packet_dir, packet_payload)) - - finding_path = packet_dir / "finding.yaml" - if finding_path.exists(): - stanza = _parse_minimal_yaml_mapping(finding_path.read_text(encoding="utf-8")) - missing_stanza_fields = tuple(name for name in PROVENANCE_STANZA_FIELDS if not stanza.get(name)) - if isinstance(packet_payload, Mapping): - provenance = packet_payload.get("provenance") - current_run = provenance.get("current_run") if isinstance(provenance, Mapping) else None - historical = provenance.get("historical_receipt") if isinstance(provenance, Mapping) else None - if isinstance(current_run, Mapping): - expected_finding_fields = { - "commit_sha": provenance.get("commit_sha") if isinstance(provenance, Mapping) else None, - "run_date": provenance.get("run_date") if isinstance(provenance, Mapping) else None, - "current_run_id": current_run.get("run_id"), - "current_run_timestamp": current_run.get("run_timestamp"), - "current_route": current_run.get("route"), - "historical_commit_sha": historical.get("commit_sha") if isinstance(historical, Mapping) else None, - "historical_run_date": historical.get("run_date") if isinstance(historical, Mapping) else None, - "historical_receipt_status": historical.get("status") if isinstance(historical, Mapping) else None, - } - for key, expected in expected_finding_fields.items(): - if stanza.get(key) != str(expected): - errors.append(f"finding.yaml {key} does not match packet provenance") - - report_path = packet_dir / "report.md" - if report_path.exists(): - report_text = report_path.read_text(encoding="utf-8") - malformed_sections.extend(_validate_report_sections(report_text)) - - nonclaims_path = packet_dir / "NON-CLAIMS.md" - if nonclaims_path.exists() and not nonclaims_path.read_text(encoding="utf-8").strip(): - errors.append("NON-CLAIMS.md must not be empty") - - checks_path = packet_dir / "checks.json" - if checks_path.exists(): - try: - checks_payload = json.loads(checks_path.read_text(encoding="utf-8")) - except json.JSONDecodeError as exc: - errors.append(f"checks.json is not valid JSON: {exc}") - else: - for required_key in ("pass", "unsupported_claims", "coverage_notes"): - if required_key not in checks_payload: - errors.append(f"checks.json missing required key: {required_key}") - - for ndjson_name in ("evidence.ndjson", "queries.ndjson"): - ndjson_path = packet_dir / ndjson_name - if ndjson_path.exists(): - error = _validate_ndjson(ndjson_path) - if error is not None: - errors.append(error) - - ok = not (missing_files or missing_stanza_fields or malformed_sections or schema_errors or receipt_errors or errors) - return PacketValidationResult( - packet_dir=packet_dir, - ok=ok, - missing_files=missing_files, - missing_stanza_fields=missing_stanza_fields, - malformed_sections=tuple(malformed_sections), - schema_errors=schema_errors, - receipt_errors=receipt_errors, - errors=tuple(errors), - ) - - -@dataclass(frozen=True) -class DemoRegistryEntry: - """One row in the demo registry manifest.""" - - slug: str - prompt_path: str - packet_dir: str - mode: str # public | private | fixture | anti-demo - required_primitives: tuple[str, ...] = () - - @staticmethod - def from_dict(payload: dict[str, object]) -> DemoRegistryEntry: - try: - raw_primitives = payload.get("required_primitives", ()) - primitives = raw_primitives if isinstance(raw_primitives, list | tuple) else () - return DemoRegistryEntry( - slug=str(payload["slug"]), - prompt_path=str(payload["prompt_path"]), - packet_dir=str(payload["packet_dir"]), - mode=str(payload["mode"]), - required_primitives=tuple(str(item) for item in primitives), - ) - except KeyError as exc: - raise DemoPacketValidationError(f"registry entry missing required key: {exc}") from None - - -def load_demo_registry(registry_path: Path) -> tuple[DemoRegistryEntry, ...]: - payload = json.loads(registry_path.read_text(encoding="utf-8")) - if not isinstance(payload, list): - raise DemoPacketValidationError(f"{registry_path}: expected a JSON array of registry entries") - return tuple(DemoRegistryEntry.from_dict(entry) for entry in payload) - - -@dataclass(frozen=True) -class RegistryLintResult: - """Outcome of linting every entry in a demo registry manifest.""" - - registry_path: Path - ok: bool - entry_results: tuple[tuple[str, PacketValidationResult | None], ...] - registry_errors: tuple[str, ...] = () - unregistered_packet_dirs: tuple[str, ...] = () - - def to_dict(self) -> dict[str, object]: - return { - "registry_path": str(self.registry_path), - "ok": self.ok, - "registry_errors": list(self.registry_errors), - "unregistered_packet_dirs": list(self.unregistered_packet_dirs), - "entries": [ - { - "slug": slug, - "packet_missing": result is None, - **({"validation": result.to_dict()} if result is not None else {}), - } - for slug, result in self.entry_results - ], - } - - -def _registered_packet_paths(entries: Iterable[DemoRegistryEntry], *, repo_root: Path) -> set[Path]: - return {(repo_root / entry.packet_dir).resolve() for entry in entries} - - -def _discover_v2_packet_dirs(*, repo_root: Path) -> set[Path]: - demo_root = repo_root / DEFAULT_DEMO_ROOT - if not demo_root.is_dir(): - return set() - return {path.parent.resolve() for path in demo_root.rglob("packet.json") if path.is_file()} - - -def lint_demo_registry( - registry_path: Path, - *, - repo_root: Path, - schema_path: Path = DEFAULT_SCHEMA_PATH, -) -> RegistryLintResult: - """Validate every registered packet and reject unregistered v2 packets.""" - - entries = load_demo_registry(registry_path) - results: list[tuple[str, PacketValidationResult | None]] = [] - registry_errors: list[str] = [] - slugs: set[str] = set() - packet_paths: set[Path] = set() - - for entry in entries: - if entry.slug in slugs: - registry_errors.append(f"duplicate registry slug: {entry.slug}") - slugs.add(entry.slug) - packet_dir = (repo_root / entry.packet_dir).resolve() - if packet_dir in packet_paths: - registry_errors.append(f"duplicate registry packet_dir: {entry.packet_dir}") - packet_paths.add(packet_dir) - prompt_path = (repo_root / entry.prompt_path).resolve() - if not prompt_path.is_file(): - registry_errors.append(f"{entry.slug}: prompt missing at {entry.prompt_path}") - if not packet_dir.is_dir(): - results.append((entry.slug, None)) - continue - results.append((entry.slug, validate_packet(packet_dir, schema_path=schema_path))) - - discovered = _discover_v2_packet_dirs(repo_root=repo_root) - registered = _registered_packet_paths(entries, repo_root=repo_root) - unregistered = tuple(sorted(str(path.relative_to(repo_root.resolve())) for path in discovered - registered)) - ok = not registry_errors and not unregistered and all(result is not None and result.ok for _, result in results) - return RegistryLintResult( - registry_path=registry_path, - ok=ok, - entry_results=tuple(results), - registry_errors=tuple(registry_errors), - unregistered_packet_dirs=unregistered, - ) - - -def iter_registry_slugs(entries: Iterable[DemoRegistryEntry]) -> tuple[str, ...]: - return tuple(entry.slug for entry in entries) - - -__all__ = [ - "DEFAULT_SCHEMA_PATH", - "OPTIONAL_PACKET_FILENAMES", - "PACKET_FILENAMES", - "PROVENANCE_STANZA_FIELDS", - "REPORT_SECTION_ORDER", - "DemoPacketValidationError", - "DemoRegistryEntry", - "PacketValidationResult", - "RegistryLintResult", - "iter_registry_slugs", - "lint_demo_registry", - "load_demo_registry", - "validate_packet", -] diff --git a/devtools/demo_shelf.py b/devtools/demo_shelf.py deleted file mode 100644 index 38dd2a79c1..0000000000 --- a/devtools/demo_shelf.py +++ /dev/null @@ -1,588 +0,0 @@ -"""Refresh or verify a curated current demo shelf.""" - -from __future__ import annotations - -import argparse -import json -import re -import subprocess -import sys -from dataclasses import dataclass -from pathlib import Path -from typing import Any - -DEFAULT_ROOT = Path(".agent/demos") -DEFAULT_MANIFEST = "MANIFEST.readable.json" -DEFAULT_SUMMARY_INDEX = "SUMMARY_INDEX.json" -DEFAULT_README = "README.md" -DEFAULT_CATALOG = "CURATED_CATALOG.md" -DEMO_SET_CONTRACT = "current-curated-demo-set" -READABLE_SUFFIXES = frozenset({".md", ".json", ".jsonl", ".csv", ".txt", ".yaml", ".yml"}) -SUMMARY_COVERAGE_FIELDS = frozenset({"claim", "non_claim", "proof_fields", "caveat_fields"}) -MAX_EXCLUSION_SAMPLES = 20 - - -@dataclass(frozen=True, slots=True) -class ShelfRender: - manifest_path: Path - summary_index_path: Path - readme_path: Path - catalog_path: Path - manifest_text: str - summary_index_text: str - readme_text: str - catalog_text: str - file_count: int - readable_count: int - summary_count: int - summary_coverage: dict[str, list[str]] - unsummarized_demos: list[str] - input_closure_mode: str - display_root: str - excluded_input_count: int - exclusion_reason_counts: dict[str, int] - exclusion_samples: list[dict[str, str]] - - -def _is_readable_artifact(path: Path) -> bool: - return path.suffix.lower() in READABLE_SUFFIXES - - -def _first_heading(path: Path) -> str: - try: - for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): - match = re.match(r"^#\s+(.+)$", line) - if match: - return match.group(1).strip() - except OSError: - return "" - return "" - - -def _nonempty_string(value: object) -> bool: - return isinstance(value, str) and bool(value.strip()) - - -def _summary_timestamp(payload: dict[str, Any]) -> object: - for key in ("updated_at", "generated_at", "created_at"): - value = payload.get(key) - if value is not None: - return value - return None - - -def _summary_record(root: Path, path: Path) -> dict[str, Any]: - rel = path.relative_to(root).as_posix() - demo = path.parent.relative_to(root).as_posix() - try: - payload = json.loads(path.read_text()) - except (OSError, json.JSONDecodeError) as exc: - return { - "demo": demo, - "summary_path": rel, - "parse_error": str(exc), - "coverage": { - "claim": False, - "non_claim": False, - "proof_fields": False, - "caveat_fields": False, - }, - } - if not isinstance(payload, dict): - return { - "demo": demo, - "summary_path": rel, - "parse_error": "summary root is not a JSON object", - "coverage": { - "claim": False, - "non_claim": False, - "proof_fields": False, - "caveat_fields": False, - }, - } - - proof_fields = sorted(key for key in payload if "proof" in key.lower()) - caveat_fields = sorted(key for key in payload if "caveat" in key.lower()) - claim = payload.get("claim") - non_claim = payload.get("non_claim") - record: dict[str, Any] = { - "demo": demo, - "summary_path": rel, - "artifact": payload.get("artifact", demo), - "claim": claim if _nonempty_string(claim) else None, - "non_claim": non_claim if _nonempty_string(non_claim) else None, - "proof_fields": proof_fields, - "caveat_fields": caveat_fields, - "archive_root": payload.get("archive_root"), - "index_schema_version": payload.get("index_schema_version", payload.get("index_schema")), - "timestamp": _summary_timestamp(payload), - "coverage": { - "claim": _nonempty_string(claim), - "non_claim": _nonempty_string(non_claim), - "proof_fields": bool(proof_fields), - "caveat_fields": bool(caveat_fields), - }, - } - return record - - -def _summary_records(root: Path, included_paths: set[Path]) -> list[dict[str, Any]]: - records: list[dict[str, Any]] = [] - for path in sorted(included_paths): - if path.name.endswith("summary.json"): - records.append(_summary_record(root, path)) - return records - - -def _summary_coverage(records: list[dict[str, Any]]) -> dict[str, list[str]]: - return { - "without_claim": [record["summary_path"] for record in records if not record["coverage"]["claim"]], - "without_non_claim": [record["summary_path"] for record in records if not record["coverage"]["non_claim"]], - "without_proof_fields": [ - record["summary_path"] for record in records if not record["coverage"]["proof_fields"] - ], - "without_caveat_fields": [ - record["summary_path"] for record in records if not record["coverage"]["caveat_fields"] - ], - } - - -def _schema_version_mismatches(records: list[dict[str, Any]], required: int | None) -> list[dict[str, Any]]: - if required is None: - return [] - mismatches: list[dict[str, Any]] = [] - for record in records: - observed = record.get("index_schema_version") - if observed is None: - continue - try: - observed_int = int(observed) - except (TypeError, ValueError): - observed_int = None - if observed_int != required: - mismatches.append( - { - "summary_path": record["summary_path"], - "observed": observed, - "required": required, - } - ) - return mismatches - - -def _unsummarized_demos(demo_records: list[dict[str, object]], summary_records: list[dict[str, Any]]) -> list[str]: - summarized_top_level = {str(record["demo"]).split("/", maxsplit=1)[0] for record in summary_records} - return [str(record["id"]) for record in demo_records if str(record["id"]) not in summarized_top_level] - - -def _build_summary_index( - root: Path, - summary_records: list[dict[str, Any]], - unsummarized_demos: list[str], -) -> tuple[str, int, dict[str, list[str]]]: - coverage = _summary_coverage(summary_records) - coverage = { - **coverage, - "unsummarized_demos": unsummarized_demos, - } - summary_index = { - "root": str(root), - "summary_count": len(summary_records), - "coverage": coverage, - "records": summary_records, - } - return json.dumps(summary_index, indent=2) + "\n", len(summary_records), coverage - - -def _demo_records( - root: Path, - files: list[dict[str, object]], - included_paths: set[Path], -) -> list[dict[str, object]]: - records: list[dict[str, object]] = [] - demo_ids = sorted( - { - str(record["path"]).split("/", maxsplit=1)[0] - for record in files - if "/" in str(record["path"]) and not str(record["path"]).startswith(".") - } - ) - for demo_id in demo_ids: - entry = root / demo_id - readme_candidates = (entry / "README.md", entry / "current" / "README.md") - analysis_candidates = (entry / "ANALYSIS.md", entry / "current" / "ANALYSIS.md") - readme = next((path for path in readme_candidates if path.resolve() in included_paths), None) - analysis = next((path for path in analysis_candidates if path.resolve() in included_paths), None) - title = ( - (_first_heading(readme) if readme is not None else "") - or (_first_heading(analysis) if analysis is not None else "") - or demo_id.replace("-", " ").title() - ) - entry_files = [record for record in files if str(record["path"]).startswith(f"{demo_id}/")] - readable_count = sum(1 for record in entry_files if record["readable"]) - records.append( - { - "id": demo_id, - "title": title, - "readme": readme.relative_to(root).as_posix() if readme is not None else None, - "analysis": analysis.relative_to(root).as_posix() if analysis is not None else None, - "file_count": len(entry_files), - "readable_count": readable_count, - } - ) - return records - - -def _build_readme(records: list[dict[str, object]]) -> str: - lines = [ - "# Polylogue Current Demo Shelf", - "", - "Generated by `devtools workspace demo-shelf`.", - "", - "This shelf contains the best current Polylogue demos for the active devloop.", - "It is not append-only. Replace, consolidate, or move stale demos out when a", - "better demo supersedes them.", - "", - "## Current Entries", - "", - ] - if records: - for record in records: - lines.append(f"- `{record['id']}` — {record['title']}") - if record["readme"]: - lines.append(f" - readme: `{record['readme']}`") - if record["analysis"]: - lines.append(f" - analysis: `{record['analysis']}`") - lines.append(f" - files: {record['file_count']} ({record['readable_count']} readable)") - else: - lines.append("No demo entries found.") - lines.extend(["", "Retired demo material belongs under `.agent/archive/retired-demos/`, not here.", ""]) - return "\n".join(lines) - - -def _build_catalog(records: list[dict[str, object]]) -> str: - lines = [ - "# Polylogue Demo Catalog", - "", - "Generated by `devtools workspace demo-shelf`.", - "", - ] - if records: - for record in records: - lines.append(f"## {record['title']}") - lines.append("") - lines.append(f"- id: `{record['id']}`") - if record["readme"]: - lines.append(f"- readme: `{record['readme']}`") - if record["analysis"]: - lines.append(f"- analysis: `{record['analysis']}`") - lines.append(f"- files: {record['file_count']} ({record['readable_count']} readable)") - lines.append("") - else: - lines.append("No demo entries found.") - lines.append("") - return "\n".join(lines) - - -def _parse_required_summary_coverage(value: str) -> set[str]: - if not value: - return set() - fields = {item.strip() for item in value.split(",") if item.strip()} - unknown = fields - SUMMARY_COVERAGE_FIELDS - if unknown: - expected = ", ".join(sorted(SUMMARY_COVERAGE_FIELDS)) - raise argparse.ArgumentTypeError( - f"unknown coverage field(s): {', '.join(sorted(unknown))}; expected {expected}" - ) - return fields - - -def _coverage_failures(summary_coverage: dict[str, list[str]], required: set[str]) -> dict[str, list[str]]: - return { - field: summary_coverage.get(f"without_{field}", []) - for field in sorted(required) - if summary_coverage.get(f"without_{field}", []) - } - - -def _repository_context(root: Path) -> tuple[Path, str] | None: - result = subprocess.run( - ["git", "-C", str(root), "rev-parse", "--show-toplevel"], - check=False, - capture_output=True, - text=True, - ) - if result.returncode != 0: - return None - repository_root = Path(result.stdout.strip()).resolve() - try: - relative_root = root.relative_to(repository_root).as_posix() - except ValueError: - return None - return repository_root, relative_root or "." - - -def _select_input_files( - root: Path, - *, - generated: set[Path], - output_root: Path, - private_projection: bool, -) -> tuple[str, str, list[Path], list[dict[str, str]]]: - repository_context = _repository_context(root) - relative_root = repository_context[1] if repository_context is not None else None - display_root = relative_root if relative_root is not None else str(root) - all_files = [ - path.resolve() - for path in sorted(root.rglob("*")) - if path.is_file() - and path.resolve() not in generated - and (output_root == root or not path.resolve().is_relative_to(output_root)) - ] - if private_projection or repository_context is None: - mode = "filesystem-private" if private_projection else "filesystem" - return mode, display_root, all_files, [] - - repository_root, relative_root = repository_context - result = subprocess.run( - ["git", "-C", str(repository_root), "ls-files", "-z", "--", relative_root], - check=True, - capture_output=True, - ) - tracked = { - (repository_root / raw.decode("utf-8", errors="surrogateescape")).resolve() - for raw in result.stdout.split(b"\0") - if raw - } - included = [path for path in all_files if path in tracked] - excluded = [ - {"path": path.relative_to(root).as_posix(), "reason": "not_git_tracked"} - for path in all_files - if path not in tracked - ] - return "git-tracked", display_root, included, excluded - - -def render_demo_shelf( - root: Path, - *, - manifest_name: str = DEFAULT_MANIFEST, - summary_index_name: str = DEFAULT_SUMMARY_INDEX, - readme_name: str = DEFAULT_README, - catalog_name: str = DEFAULT_CATALOG, - private_output: Path | None = None, -) -> ShelfRender: - """Build deterministic indexes from tracked inputs or an explicit private projection.""" - root = root.expanduser().resolve() - output_root = private_output.expanduser().resolve() if private_output is not None else root - if private_output is not None and output_root.is_relative_to(root): - raise ValueError("private output must be outside the committed shelf root") - manifest_path = output_root / manifest_name - summary_index_path = output_root / summary_index_name - readme_path = output_root / readme_name - catalog_path = output_root / catalog_name - generated = { - path.resolve() - for path in ( - manifest_path, - summary_index_path, - readme_path, - catalog_path, - root / manifest_name, - root / summary_index_name, - root / readme_name, - root / catalog_name, - ) - } - mode, display_root, input_files, excluded_inputs = _select_input_files( - root, - generated=generated, - output_root=output_root, - private_projection=private_output is not None, - ) - reason_counts: dict[str, int] = {} - for excluded in excluded_inputs: - reason = excluded["reason"] - reason_counts[reason] = reason_counts.get(reason, 0) + 1 - files = [ - { - "path": path.relative_to(root).as_posix(), - "bytes": path.stat().st_size, - "readable": _is_readable_artifact(path), - } - for path in input_files - ] - manifest = { - "contract": DEMO_SET_CONTRACT, - "root": display_root, - "input_closure": { - "mode": mode, - "included_count": len(files), - "excluded_count": len(excluded_inputs), - "exclusion_reason_counts": dict(sorted(reason_counts.items())), - "exclusion_samples": "bounded samples are emitted in the command JSON payload", - }, - "packaging": ( - "Use devtools workspace read-package for portable readable bundles; this helper only writes indexes." - ), - "curation_policy": ( - "This is not append-only. Keep the best current demos here; replace, consolidate, or move stale demos out." - ), - "file_count": len(files), - "readable_count": sum(1 for item in files if item["readable"]), - "files": files, - } - manifest_text = json.dumps(manifest, indent=2) + "\n" - included_paths = set(input_files) - summary_records = _summary_records(root, included_paths) - records = _demo_records(root, files, included_paths) - unsummarized_demos = _unsummarized_demos(records, summary_records) - summary_index_text, summary_count, summary_coverage = _build_summary_index( - Path(display_root), - summary_records, - unsummarized_demos, - ) - readme_text = _build_readme(records) - catalog_text = _build_catalog(records) - return ShelfRender( - manifest_path=manifest_path, - summary_index_path=summary_index_path, - readme_path=readme_path, - catalog_path=catalog_path, - manifest_text=manifest_text, - summary_index_text=summary_index_text, - readme_text=readme_text, - catalog_text=catalog_text, - file_count=len(files), - readable_count=sum(1 for item in files if item["readable"]), - summary_count=summary_count, - summary_coverage=summary_coverage, - unsummarized_demos=unsummarized_demos, - input_closure_mode=mode, - display_root=display_root, - excluded_input_count=len(excluded_inputs), - exclusion_reason_counts=dict(sorted(reason_counts.items())), - exclusion_samples=excluded_inputs[:MAX_EXCLUSION_SAMPLES], - ) - - -def _changed(path: Path, expected: str) -> bool: - try: - return path.read_text() != expected - except FileNotFoundError: - return True - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser( - prog="devtools workspace demo-shelf", - description="Refresh or verify curated current demo shelf indexes.", - ) - parser.add_argument("--root", type=Path, default=DEFAULT_ROOT, help=f"Demo shelf root (default: {DEFAULT_ROOT})") - parser.add_argument("--manifest", default=DEFAULT_MANIFEST, help=f"Manifest filename (default: {DEFAULT_MANIFEST})") - parser.add_argument( - "--summary-index", - default=DEFAULT_SUMMARY_INDEX, - help=f"Summary index filename (default: {DEFAULT_SUMMARY_INDEX})", - ) - parser.add_argument( - "--require-summary-coverage", - type=_parse_required_summary_coverage, - default=set(), - metavar="FIELDS", - help="In check mode, fail when summaries lack comma-separated fields: claim, non_claim, proof_fields, caveat_fields.", - ) - parser.add_argument( - "--require-index-schema-version", - type=int, - default=None, - metavar="VERSION", - help="In check mode, fail when a summary declares a different index_schema_version.", - ) - parser.add_argument( - "--private-output", - type=Path, - default=None, - help="Write a full-filesystem private projection to this separate, untracked directory.", - ) - parser.add_argument("--check", action="store_true", help="Verify files are current without writing.") - parser.add_argument("--json", action="store_true", help="Emit machine-readable JSON.") - args = parser.parse_args(argv) - rendered = render_demo_shelf( - args.root, - manifest_name=args.manifest, - summary_index_name=args.summary_index, - private_output=args.private_output, - ) - changed = { - "manifest": _changed(rendered.manifest_path, rendered.manifest_text), - "summary_index": _changed(rendered.summary_index_path, rendered.summary_index_text), - "readme": _changed(rendered.readme_path, rendered.readme_text), - "catalog": _changed(rendered.catalog_path, rendered.catalog_text), - } - coverage_failures = _coverage_failures(rendered.summary_coverage, args.require_summary_coverage) - summary_index_payload = json.loads(rendered.summary_index_text) - schema_mismatches = _schema_version_mismatches( - summary_index_payload["records"], - args.require_index_schema_version, - ) - undeclared_inputs = rendered.input_closure_mode == "git-tracked" and rendered.excluded_input_count > 0 - ok = not any(changed.values()) and not coverage_failures and not schema_mismatches and not undeclared_inputs - if not args.check and not undeclared_inputs: - rendered.manifest_path.parent.mkdir(parents=True, exist_ok=True) - rendered.manifest_path.write_text(rendered.manifest_text) - rendered.summary_index_path.write_text(rendered.summary_index_text) - rendered.readme_path.write_text(rendered.readme_text) - rendered.catalog_path.write_text(rendered.catalog_text) - ok = True - changed = {"manifest": False, "summary_index": False, "readme": False, "catalog": False} - coverage_failures = {} - schema_mismatches = [] - - payload = { - "ok": ok, - "root": str(args.root.expanduser().resolve()), - "manifest": str(rendered.manifest_path), - "summary_index": str(rendered.summary_index_path), - "file_count": rendered.file_count, - "readable_count": rendered.readable_count, - "summary_count": rendered.summary_count, - "unsummarized_demos": rendered.unsummarized_demos, - "required_summary_coverage": sorted(args.require_summary_coverage), - "summary_coverage_failures": coverage_failures, - "required_index_schema_version": args.require_index_schema_version, - "summary_schema_mismatches": schema_mismatches, - "input_closure": { - "mode": rendered.input_closure_mode, - "included_count": rendered.file_count, - "excluded_count": rendered.excluded_input_count, - "exclusion_reason_counts": rendered.exclusion_reason_counts, - }, - "undeclared_inputs": undeclared_inputs, - "exclusion_samples": rendered.exclusion_samples, - "changed": changed, - "mode": "check" if args.check else "write", - } - if args.json: - json.dump(payload, sys.stdout, indent=2) - sys.stdout.write("\n") - elif ok: - print( - f"demo shelf {'current' if args.check else 'refreshed'}: " - f"{rendered.file_count} files, {rendered.readable_count} readable, " - f"{rendered.summary_count} summaries, " - f"{len(rendered.unsummarized_demos)} unsummarized; portable bundles: read-package" - ) - else: - reasons = [name for name, value in changed.items() if value] - reasons.extend(f"summary coverage {field}" for field in coverage_failures) - if schema_mismatches: - reasons.append("summary schema version") - if undeclared_inputs: - reasons.append("undeclared shelf inputs (use --private-output for a private projection)") - print("demo shelf drift:", ", ".join(reasons), file=sys.stderr) - return 0 if ok else 1 - - -if __name__ == "__main__": # pragma: no cover - raise SystemExit(main()) diff --git a/devtools/devloop_temporal.py b/devtools/devloop_temporal.py deleted file mode 100644 index 626c8e6033..0000000000 --- a/devtools/devloop_temporal.py +++ /dev/null @@ -1,255 +0,0 @@ -"""Compose devloop-local event sources into a TemporalEvidenceWindow.""" - -from __future__ import annotations - -import argparse -import json -import re -import subprocess -import sys -from datetime import datetime -from pathlib import Path -from zoneinfo import ZoneInfo - -from polylogue.surfaces.temporal_evidence import TemporalEvidenceEvent, build_temporal_evidence_window - -ROOT = Path(__file__).resolve().parents[1] -DEFAULT_CONDUCTOR_ROOT = ROOT / ".agent/conductor-devloop" -DEFAULT_OPERATING_LOG = DEFAULT_CONDUCTOR_ROOT / "OPERATING-LOG.md" -DEFAULT_EVENT_LOG = DEFAULT_CONDUCTOR_ROOT / "EVENTS.jsonl" -WARSAW = ZoneInfo("Europe/Warsaw") -LOG_HEADING_RE = re.compile( - r"^## (?P\d{4}-\d{2}-\d{2}) (?P