diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index c760edbb98..5f55c0c41b 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -144,6 +144,7 @@ {"_type":"issue","id":"polylogue-tf2.1","title":"Rerun forensics on current archive; price origin_reported providers","description":"Rerun scripts/agent_forensics.py against the current archive (v23+); price origin_reported providers via the vendored LiteLLM catalog (match last path segment); all-provider headline or explicitly-labeled per-provenance figures that cannot be misread; record deltas vs 06-27; verify chart SVGs render. Cache-inclusion must be disambiguated (Codex input INCLUDES cached ~96%; see bd memories). Also blocked on logical-session token attribution — the headline must not be double-counted.","notes":"Correction to close_reason monetary values: stored/provider-priced subset was $239,453.14; catalog API-equivalent was $318,650.88; origin_reported catalog estimate was $79,197.74. The original close_reason text lost dollar-prefixed digits due shell expansion, not measurement drift.","status":"closed","priority":0,"issue_type":"task","assignee":"Sinity","owner":"ezo.dev@gmail.com","created_at":"2026-07-03T04:31:33Z","created_by":"Sinity","updated_at":"2026-07-31T22:35:43Z","started_at":"2026-07-03T09:28:10Z","closed_at":"2026-07-03T09:59:02Z","close_reason":"Completed with blocker caveat preserved: scripts/agent_forensics.py now prices origin_reported rows through the shared vendored LiteLLM pricing catalog while preserving stored provenance; report separates stored/provider-priced cost from catalog API-equivalent estimates and carries logical-session/cache caveats instead of claiming final billing reconciliation. Regenerated current artifact at .agent/demos/agent-forensics against /home/sinity/.local/share/polylogue schema v23: 16,498 physical sessions, 4,142,175 messages, 356.5B tokens, ,453.14 stored/provider-priced subset, ,650.88 catalog API-equivalent, and ,197.74 origin_reported catalog estimate. SVG parse check passed for 9 charts; devtools test tests/unit/scripts/test_agent_forensics.py passed; devtools verify --quick passed run 20260703T095718Z-quick-753466-96559776; devloop-review clean. Remaining final-reconciliation blocker stays open as polylogue-4ts.2.","labels":["area:usage","campaign"],"dependencies":[{"issue_id":"polylogue-tf2.1","depends_on_id":"polylogue-4ts.2","type":"blocks","created_at":"2026-07-03T06:32:45Z","created_by":"Sinity","metadata":"{}"},{"issue_id":"polylogue-tf2.1","depends_on_id":"polylogue-sru.7","type":"blocks","created_at":"2026-07-03T06:31:33Z","created_by":"Sinity","metadata":"{}"},{"issue_id":"polylogue-tf2.1","depends_on_id":"polylogue-tf2","type":"parent-child","created_at":"2026-07-03T06:31:33Z","created_by":"Sinity","metadata":"{}"}],"dependency_count":2,"dependent_count":2,"comment_count":0} {"_type":"issue","id":"polylogue-tf2","title":"Campaign: agent-forensics regeneration + all-provider repricing","description":"Regenerate the agent-forensics packet on the current archive with an honest all-provider headline. The 2026-06-27 report (546.6B tokens, $89,368 API-list equivalent, 216x cache amplification) is the most stranger-legible artifact on any shelf, but its numbers are pre-dedup stale and the headline prices only the priced-provenance subset (Claude Code cost_usd rows); Codex/ChatGPT/Gemini are origin_reported token counts with no dollar value (operator estimate ~$150K all-provider). Sequenced after claim-vs-evidence per operator direction 2026-07-02.","design":"Current slice design: turn the existing agent-forensics/cost headline into a product-backed all-provider repricing artifact. First inspect devtools/scripts and polylogue analyze surfaces for agent_forensics/cost code. Use active archive usage headline (detail=headline) for authoritative physical_session and logical_session_model_high_water token totals. Keep priced-provenance dollars and origin-reported token estimates separate: do not multiply every token by one blended price without a labeled lane. Add or reuse a shared pricing/projection helper so the demo artifact is regenerated from Polylogue product code, not ad hoc SQL. Acceptance for this slice: the generated agent-forensics artifact names archive root/schema, includes physical vs logical token grain, separates priced subset from origin-reported estimate lanes, gives reproduction commands, and has focused tests for any new repricing helper/surface.","acceptance_criteria":"Terminal state: regenerated forensics packet on the current archive with an honest all-provider headline (priced subset AND origin-reported estimate lanes separated), agent_forensics.py folded into polylogue analyze (tf2.2), artifact on the demo shelf with reproduction commands, cold-reader gate passed. Epic closes only when that artifact is recorded.","status":"closed","priority":0,"issue_type":"epic","assignee":"Sinity","owner":"ezo.dev@gmail.com","created_at":"2026-07-03T04:31:32Z","created_by":"Sinity","updated_at":"2026-07-31T22:35:43Z","started_at":"2026-07-03T18:47:23Z","closed_at":"2026-07-03T19:06:44Z","close_reason":"Completed: provider usage headline now exposes product-backed pricing lanes in polylogue analyze usage --detail headline, separating stored/provider-priced cost from catalog API-equivalent estimates for origin_reported rows. Regenerated the current .agent/demos/agent-forensics artifact against /home/sinity/.local/share/polylogue schema v23: physical-session tokens 395,320,980,423; logical high-water tokens 288,741,229,728; stored/provider-priced USD 243,392.189328; catalog API-equivalent USD 337,565.031618; priced lane 13,889 rows / 12,331 sessions / 12,650 matched rows; origin_reported lane 2,308 rows / 2,270 sessions / 2,302 matched rows. Verification: live polylogue --plain analyze usage --detail headline --format json --limit 0 wrote /realm/tmp/polylogue-usage-headline-pricing-current.json; devtools test tests/unit/storage/test_provider_usage_report.py tests/unit/cli/test_diagnostics.py passed 23 tests; devtools verify --quick passed run 20260703T190553Z-quick-2226137-d91d4e8f; devtools workspace demo-shelf --json reported ok. Non-claim preserved: this is not final billing reconciliation and physical/logical token grains stay explicitly separated.","labels":["area:usage","campaign","size:M","spine"],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-sru","title":"Campaign: claim-vs-evidence report to finding-grade","description":"Terminal state: an externally publishable finding ('how often do coding agents proceed past failed tool calls, by model/tool') with stated sample frame, calibrated markers, benign/consequential split, seeded stranger-runnable reproduction, and a passed cold-reader gate. Slice closure is NOT campaign closure; this epic stays top-of-frame until its terminal state is recorded.\\n\\nState as of 2026-07-03 after calibrated active-archive regeneration: archive root /home/sinity/.local/share/polylogue, index schema v23, 41,886 structured failures total, 5,000 origin-stratified failures inspected (3,746 claude-code-session, 1,247 codex-session, 7 claude-ai-export), 100 unpaired structured failures. Marker vocabulary was tightened to avoid broad issue/fix/block/gitignored false positives. Immediate next-turn totals: acknowledged=420, silent_proceed=1,205, ambiguous=3,375 (2,624 wordless tool continuations; 751 prose without marker). Lower-bound silent rate is 24.1%; among classified immediate next turns, silent rate is 74.2%. Next-3 sensitivity window, stopping before the next user message, finds 302 acknowledgments that appear only after the next turn; window3 silent lower bound is 37.0%. Calibration: 50 hand-labeled immediate-next-turn rows, acknowledged-marker precision=1.0, recall=0.8421052631578947, invalid rows=0. Artifact: .agent/demos/claim-vs-evidence/claim-vs-evidence.report.json.","notes":"2026-07-03 update: methodology package is now cold-read gated. .agent/demos/claim-vs-evidence contains aggregate live evidence, public-summary.json, PUBLIC_REPRODUCTION.md, COLD_READER_GATE.md, and COLD_READ_RESULT.md. Seeded reproduction is meaningful, not empty: 4 structured failures, 2 acknowledged follow-ups, 2 silent-proceed follow-ups, 0 unpaired. Cold-reader subagent PASS recovered claim/non-claim, sample frame, rates, calibration, caveats, and reproduction commands from the artifact directory only. Remaining campaign child: polylogue-sru.1 productizes action-unit outcome/followup_class capability.","status":"closed","priority":0,"issue_type":"epic","owner":"ezo.dev@gmail.com","created_at":"2026-07-03T04:31:26Z","created_by":"Sinity","updated_at":"2026-07-31T22:35:43Z","closed_at":"2026-07-03T09:28:09Z","close_reason":"Completed: all seven campaign children are closed. The claim-vs-evidence finding now has bounded sample-frame reporting, calibrated marker precision/recall, handler-class and next-3 sensitivity splits, meaningful seeded reproduction, cold-reader PASS, and productized action-unit followup_class/followup_message_ref query capability. Current artifact lives under .agent/demos/claim-vs-evidence and was regenerated against /home/sinity/.local/share/polylogue schema v23.","labels":["area:substrate","campaign"],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"polylogue-z7sv3","title":"Make PR scope CI authority immutable at the base revision","description":"The first structured PR-scope carrier landed a green CI gate, but the validator still executes from the PR checkout and CircleCI can lack CIRCLE_PULL_REQUEST. Harden the process boundary so a pull request cannot weaken the validator it is being judged by.","design":"Modify devtools/pr_scope.py to resolve repository and PR metadata through GitHub REST, validate exact checkout/head identity, fetch the base revision validator, and run it in an isolated subprocess. Extend CircleCI quick-gate to call check-ci with CIRCLE_SHA1 and repo metadata. Make merge receipts bind scope_digest, beads_digest, and assigned IDs, and pass --match-head-commit to gh pr merge. Add unit coverage for no-PR-URL resolution, base-validator authority, schema rejection, receipt drift, and stale-head refusal. Remove the natural-language PR state guard and update lane/CI documentation.","acceptance_criteria":"1. CircleCI validates the exact checkout head and resolves the unique open PR when CIRCLE_PULL_REQUEST is absent. 2. When the base revision already contains pr_scope.py, CI executes that base validator rather than the PR-modified validator. 3. The carrier schema rejects unknown fields and merge receipts bind scope digest, Bead digest, and assigned Bead IDs. 4. The prose-parsing PR state guard is removed or replaced by structured validation. 5. Focused tests and devtools verify --quick pass.","status":"open","priority":1,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-06T06:44:24Z","created_by":"Sinity","updated_at":"2026-08-06T06:44:34Z","dependency_count":0,"dependent_count":1,"comment_count":0} {"_type":"issue","id":"polylogue-taj0o","title":"Unify Claude Code eager and streaming parsers into one incremental multi-way merge","design":"Root architectural cause behind polylogue-4987i (session_events ordering\ninstability), which was fixed tactically in PR #3669 via a reconciliation\npass, not fixed structurally.\n\nCurrent design (path dependence, not principled):\n- Eager (parse_payload -\u003e dispatch.py grouping -\u003e _parse_code_records):\n materializes the ENTIRE raw JSONL payload into memory, groups ALL records\n by sessionId across the whole file (a complete partition, independent of\n file order), THEN feeds each session's full record list to\n _parse_code_records as one coherent single pass. Correct by construction\n because the parser never sees interleaving -- an earlier full-materialize\n step already removed it.\n- Streaming (parse_stream_payload -\u003e _claude_code_stream_sessions,\n dispatch.py:824): exists because raw JSONL ingest can be multi-GiB and\n can't be buffered wholesale. Instead of a true incremental multi-way\n merge, it takes a shortcut: detect CONTIGUOUS runs of the same sessionId\n and treat each run as an independent mini-file, reusing the exact same\n per-session \"I see the whole session at once\" parser\n (_parse_code_records via parse_code_stream) UNMODIFIED on each run. Chunks\n are then concatenated (merge_parsed_session_chunks) and reconciled after\n the fact (reconcile_code_session_chunks) to approximate what eager would\n have produced.\n\nWhy this is the wrong shape: reconcile_code_session_chunks has to\nre-implement, after the fact, every piece of session-wide accumulation\n_parse_code_records already does in its main loop (background-completion\ndedup, delegation-progress tick summation, coverage count summation,\nsession-wide event ordering) -- and every time a NEW session-wide summary\nevent type is added to the eager parser's main loop (which has happened\nseveral times, per its own comments: polylogue-pbuh AC5's coverage event,\ndelegation-progress events, session_kind), reconcile_code_session_chunks has\nto be remembered and updated to fold it too, or the same class of\neager-vs-streaming divergence bug recurs for the new event type. #3669 fixed\nthe THREE known cases; nothing prevents a fourth from being added without\nanyone updating reconcile. This is a structural bug-factory, not a one-off.\n\nProposed fix: replace both _claude_code_stream_sessions' contiguous-run\nchunking AND the eager grouping-then-parse call in dispatch.py's non-stream\npath with ONE incremental multi-way merge:\n- Walk the record stream exactly once, in file order (regardless of size).\n- Maintain a dict of open per-session accumulator state, keyed by session\n id -- the SAME state _parse_code_records currently builds up locally\n during its single-session main loop (messages, session_events-in-progress,\n delegation_progress dict, coverage counters, etc.), but keyed per session\n instead of assumed-singular.\n- Fold each record into its session's accumulator as it streams past\n (exactly the same per-record logic _parse_code_records already has, just\n addressed by session id instead of implicit \"the one session\").\n- Finalize (emit ParsedSession, apply order_session_events, run the\n post-loop coverage/background/delegation appends) a session's accumulator\n only when the stream ends (or, for true bounded-memory operation on\n extremely long-lived files, on an explicit flush signal -- out of scope\n for a first cut, current per-run memory is already \"proportional to\n unique record identifiers\" per _claude_code_stream_sessions' own\n docstring, i.e. already bounded well below full-file materialization).\n- Eager's dispatch.py grouping call and streaming's chunk-and-glue\n machinery both become this ONE function. reconcile_code_session_chunks,\n merge_parsed_session_chunks' claude-code-specific glue, and the\n eager/streaming duality in general are deleted, not deprecated\n (automagic-invariants doctrine: no break-glass tier once one path proven\n to correctly subsume the other).\n\nKnown hazards to preserve (read before touching):\n- Identity/carryover resolution (bd polylogue-jc4q, dispatch.py:848-863):\n contiguous-run-based primary/carryover detection for resume/fork/quirk\n boundaries. A multi-way merge needs the equivalent notion (which record\n run is THIS file's own primary content vs an ancestor's carryover\n prefix) re-derived under session-keyed accumulation, not run-keyed.\n Get this wrong and the fix reintroduces the exact bug this session's\n polylogue-slshy/polylogue-2hwl active-leaf-by-position lineage fixed.\n- Tool-result sidecar streaming join (polylogue-wjgf): currently teed\n through ToolResultIndexAccumulator per contiguous run, joined once a\n run's iterator is exhausted. Needs to become per-session-accumulator\n scoped instead of per-run scoped.\n- is_agent / agent-* fallback id special-casing (dispatch.py:878-882).\n- Sidecar join for the eager path (join_tool_result_sidecars, needs the\n full tool_use_id index) currently assumes full materialization; the\n merged design should reuse the SAME per-session-scoped join the\n streaming path already does, not the eager whole-file index -- one\n fewer thing that can diverge.\n\nScope note: this is a genuine parser-core rewrite of the hottest path in\nthe codebase (every Claude Code session, live and reindexed, goes through\nit). Do NOT attempt as a quick patch; needs its own dedicated session with\nfull regression coverage of tests/unit/sources/test_claude_code_normalization_laws.py,\ntest_claude_code_sidecar_evidence.py, test_parsers_claude_code_artifacts.py,\ntest_delegation_provider_fixtures.py, and a live-archive parity spot-check\n(parse every real multi-chunk/subagent-interleaved session in the archive\nboth ways, old vs new, before/after, diff zero).\n","notes":"ADDITIONAL FINDING (2026-08-03): this is a THREE-way duplication, not two. dispatch.py's eager grouping (_claude_code_grouped_record_specs, line 671) defines \"primary group\" as the group with the MOST records: `primary_group_id = max(groups, key=lambda group_id: len(groups[group_id]))`. Streaming's chunking (_claude_code_stream_sessions, line 948) defines primary as the group whose session_id equals the caller-supplied fallback_id: `is_primary_group = group_session_id == fallback_id`. These are NOT provably equivalent -- a file where the fallback_id-matching session has fewer records than another interleaved session in the same file would resolve differently under eager vs streaming.\n\nLive-archive check (read-only, source.db mode=ro): sampled 400 claude-code-session raw_sessions rows sized 200KB-5MB, zero contained \u003e1 distinct sessionId (i.e. zero genuinely session-interleaved files in that sample). Separately checked all 12 blob_hash values shared across \u003e1 distinct native_id in the whole archive -- these turned out to be a DIFFERENT, already-known phenomenon (polylogue-omsw's file-history-snapshot/artifact classification duplication, not sessionId-based session interleaving; the shared blobs contain zero \"sessionId\" fields at all). So: no live confirmed case of the eager/streaming primary-definition mismatch actually diverging on this archive today, but the code-level divergence is real and provable by inspection, not hypothetical -- it just hasn't been hit yet, or the two algorithms happen to agree in every case seen so far (files where the fallback_id-matching session also happens to have the most records, which is the common/expected shape).\n\nThis changes the design target for the unification: it's not just \"collapse eager-loop-state vs streaming-chunk-state into one accumulator\" (the code_parser.py duality already scoped), it ALSO needs ONE canonical \"which interleaved session is this file's own primary content\" algorithm shared by both paths, replacing both dispatch.py:671's max-by-count and dispatch.py:948's fallback_id-match (need to decide which definition, or a new one, is actually correct -- likely fallback_id-match, since that's grounded in the caller's own knowledge of which file this is, whereas max-by-count is a heuristic that could pick the WRONG group for a small main session with a huge subagent transcript in the same file).\n\nDecision: scoped as a dedicated lane dispatch (agent-executed, worktree-isolated, execution-grade design already documented above + this note) rather than attempted serially inline, per this repo's own orchestration doctrine and operator's earlier explicit correction this session (\"why are you not orchestrating anymore\"). Not a deferral -- dispatching now, in parallel with continued campaign work.\nStage 1 merged 2026-08-03 (PR #3680): _SessionAccumulator dataclass extraction from _parse_code_records, mechanical, zero behavior change (verified: rebased onto post-4987i master, full named regression suite 373/373 passed, mypy --strict clean). Stage 2 (the actual multi-way merge: key by session id, resolve the eager-vs-streaming primary-definition conflict, delete reconcile_code_session_chunks/merge_parsed_session_chunks's Claude-Code branch/_claude_code_stream_sessions/_claude_code_grouped_record_specs) remains open -- Stage 1 sets up the exact accumulator shape Stage 2 needs but does not itself unify eager/streaming. Bead stays open.","status":"closed","priority":1,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T18:41:03Z","created_by":"Sinity","updated_at":"2026-08-05T07:26:34Z","closed_at":"2026-08-05T07:26:34Z","close_reason":"Stage 2 already merged as 25434d0f0 (#3691): one incremental multi-way Claude Code accumulator with canonical fallback-id primary selection replaces eager/streaming duality. The named parity suites passed (480) and quick verification is recorded.","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-gbs02","title":"Acquire-only degraded mode: index-tier mismatch must not park raw acquisition","description":"Found closing mhx95 (2026-08-03). After deploy+source-migration, the daemon (current master) parks ALL 16 loops INCLUDING the live watcher on 'index.db:46!=57' — but acquisition writes only source.db, which is current (v24). Consequence: zero ingestion until 818fy runs, for no durable-tier reason; browser-capture/hook spools accumulate unprocessed and time-sensitive captures wait on a derived-tier rebuild. Fix shape: schema preflight distinguishes durable-tier mismatch (park everything — correct) from derived-only mismatch (run acquisition + spool drains + source-tier loops; park parse/materialize/index-writing loops). The parked-loop log line already enumerates loops, so the split is a classification over the existing registry. AC: with index.db deliberately at an old version and source.db current, the daemon acquires new raws (source.db row appears; spool drains) while materialization stays parked and health still reports the index mismatch. Falsification: revert the classification and the acquire test freezes again. Ref mhx95 evidence trail.","notes":"\n2026-08-03 design investigation (no code change -- this needs careful per-loop classification before touching live daemon startup sequencing, not a quick patch):\n\nCurrent structure (polylogue/daemon/cli.py ~2148-2191, health.py:341 _check_schema_version_fast):\n- _check_schema_version_fast() computes ONE aggregate severity across ALL tiers (source/index/embeddings/user/ops) with no per-tier durability distinction in its return value (HealthAlert has no detail/tier-breakdown field) -- it correctly reports CRITICAL whenever ANY tier's user_version mismatches, and the AC explicitly wants this UNCHANGED (\"health still reports the index mismatch\").\n- watcher_blocked = enable_watch and schema_alert.severity == CRITICAL currently gates BOTH the live watcher AND all 14 named loops in _SCHEMA_BLOCKED_MAINTENANCE_LOOP_NAMES via one shared `if not watcher_blocked:` block (cli.py ~2253+) -- they start together or not at all.\n\nThe fix needs TWO independent gates, not a narrowed version of the existing one:\n1. A new, SEPARATE durable-tier-only check (source.db + user.db, durability in {\"irreplaceable\",\"human\"} per ARCHIVE_TIER_SPECS) -- call it durable_mismatch. Only THIS should gate the live watcher + acquisition/spool-drain loops (the AC's \"source-tier loops\"). Add as a new function alongside _check_schema_version_fast, not a modification to it (that function's HealthAlert-typed return is consumed elsewhere for periodic health reporting and must keep reporting the FULL aggregate severity, per the AC).\n2. The EXISTING aggregate check (any tier, i.e. current behavior) must keep gating every loop that writes a derived tier (index.db/embeddings.db) -- raw materialization convergence, session insight convergence, convergence debt retry, embedding backlog catch-up, embedding orphan reconcile, fts merge, fts identity drift recompute, fts orphan audit, db optimize (likely index-tier VACUUM/ANALYZE) -- these must NOT start on a stale index.db even once gate 1 is relaxed.\n\nPer-loop classification still needed (NOT done this session -- each of the 14 names in _SCHEMA_BLOCKED_MAINTENANCE_LOOP_NAMES needs its actual write-tier confirmed by reading its implementation, not guessed from its name):\n- Likely index/embeddings-tier (must stay gated on ANY mismatch): raw materialization convergence, session insight convergence, convergence debt retry, embedding backlog catch-up, embedding orphan reconcile, fts merge, fts identity drift recompute, fts orphan audit, db optimize, judgment automation sweep (uses embeddings for judgment scoring, verify).\n- Likely source-tier-only or tier-agnostic (candidates to move to gate 1, i.e. safe to run on derived-only mismatch): wal checkpoint (verify which db(s) it checkpoints), heartbeat, status snapshot refresh (verify what it snapshots), blob gc check (blob store is source-tier), secret scan sweep (likely scans raw content = source-tier).\n- drive source catch-up (_SCHEMA_BLOCKED_OPTIONAL_DRIVE_CATCHUP_LOOP_NAME): acquisition-adjacent, likely gate-1 candidate.\n\nRisk if this is done wrong: a loop incorrectly reclassified as \"safe\" that actually writes index.db against a stale schema could silently corrupt the live production index during exactly the highest-stakes window (mid-reindex-campaign). This needs the per-loop write-tier confirmed by reading each loop's actual body, then a real test proving the split (per this bead's own AC: index.db old + source.db current -\u003e watcher runs + source.db row appears, materialization loops provably don't start), not inferred from loop names. Left for a dedicated implementation pass with that verification, not attempted blind in this session.","status":"closed","priority":1,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T16:09:27Z","created_by":"Sinity","updated_at":"2026-08-03T17:00:10Z","closed_at":"2026-08-03T17:00:10Z","close_reason":"Implemented acquire-only degraded mode: DegradedReason.derived_only flag + is_fully_degraded() (core/degraded.py), durable_tier_schema_mismatch() narrow check (daemon/health.py), two-gate split in daemon/cli.py (watcher_blocked for maintenance loops, watcher_creation_blocked for the watcher itself), acquire-then-skip-parse in both batch.py and append_ingest.py (the primary tailed-file path, which had no degraded check at all before). Regression test proves the exact AC (raw acquired, parsed_at_ms NULL, no parse_error). Commit cb60c02a3, devtools test 2211 passed (2 pre-existing load-flaky failures unrelated).","dependencies":[{"issue_id":"polylogue-gbs02","depends_on_id":"polylogue-9qnzy","type":"relates-to","created_at":"2026-08-03T18:23:53Z","created_by":"Sinity","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-0v4tn","title":"blob_refs GC oracle broken: 73,427 raw_payload + 1,336 attachment refs orphaned (hook-deinflation residue)","description":"Baseline census 2026-08-03 (invariant I3): 73,427 of 116,149 raw_payload blob_refs and ALL 1,336 attachment blob_refs have ref_ids that no longer resolve in their referent tables — overwhelmingly the hook-deinflation residue (64,896 raw_sessions rows deleted 2026-07-22 without pruning their blob_refs; same class as closed i3zo for raw_authority_plans). Consequences: (a) blob GC's snapshot-reference safety check treats ~73K blobs as referenced forever — GC can never collect them; (b) any 'blobstore pristine / no weirdness' claim (r9xsj) is false while the reference substrate lies. Blob FILES are fine (300/300 + 100/100 presence samples pass); this is bookkeeping-tier. Fix shape: set-based orphan identification (LEFT JOIN refs to referents) + prune in one guarded pass with a receipt, mirroring i3zo/PR #3530's pattern; then re-run I3 to 0. Attachment refs need their own referent-table check first — determine what ref_id should point at (attachment_refs moved tiers historically) before deleting anything. Baseline artifact: .agent/scratch/reindex-baseline-2026-08-03.md.","status":"in_progress","priority":1,"issue_type":"bug","assignee":"Sinity","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T15:39:48Z","created_by":"Sinity","updated_at":"2026-08-05T05:25:03Z","started_at":"2026-08-05T05:25:03Z","lease_expires_at":"2026-08-05T05:30:03Z","heartbeat_at":"2026-08-05T05:25:03Z","dependency_count":0,"dependent_count":1,"comment_count":0} @@ -652,7 +653,7 @@ {"_type":"issue","id":"polylogue-22ldr","title":"Map Gemini/AI-Studio codeExecutionResult.outcome to tool_result_is_error","description":"P-class (reindex-gate-hunt task #1, adjudicated 2026-08-03). polylogue/sources/providers/gemini_message.py:113-124 _tool_result_content_block constructs every TOOL_RESULT block from Gemini/AI-Studio codeExecutionResult (call sites :303, :312) and never sets ContentBlock.is_error/.exit_code — blocks.tool_result_is_error lands NULL unconditionally for this provider family (write.py:97-98 reads it straight from the parsed block).\n\nProof the signal exists and is dropped (not absent upstream): the provider schema fingerprint (schemas/providers/gemini/versions/v2/elements/session_document.schema.json.gz) records $.chunkedPrompt.chunks[*].codeExecutionResult.outcome field-length stats avg=11.3 min=10 max=14 — exactly bracketing OUTCOME_OK (10) and OUTCOME_FAILED (14), so the live corpus contains BOTH outcomes and real tool failures are being read past.\n\nContrast: the gemini-cli sibling path (local_agent.py:584-620, different dispatch route) DOES set is_error/exit_code from its envelope — this parser simply lags its sibling. Related: cuxz.4 AC#3 asks for exactly this kind of gap to be enumerated; this bead is one named instance. is_error is populated at parse/write time only (no independent backfill), so unfixed it survives any rebuild as NULL — board the 818fy batch.\n","acceptance_criteria":"gemini_message.py _tool_result_content_block sets is_error from outcome: OUTCOME_OK -\u003e False; OUTCOME_FAILED / OUTCOME_DEADLINE_EXCEEDED -\u003e True; OUTCOME_UNSPECIFIED/absent -\u003e None. exit_code stays None (no equivalent in payload). Fixture test covers both OK and FAILED shapes. Boards the 818fy reparse batch so existing NULL rows repopulate.","status":"open","priority":2,"issue_type":"bug","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T12:19:08Z","created_by":"Sinity","updated_at":"2026-08-03T12:19:08Z","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-iltbx","title":"Widen block content-hash payload with tool_result outcome fields (is_error/exit_code/outcome_unknown_reason)","description":"P-class, efficiency-critical timing (reindex-gate-hunt tasks #5+#10, adjudicated 2026-08-03). Persisted-but-unhashed parsed fields form acquire-time skip-drift channels: a re-acquired raw whose only delta is an unhashed field produces an identical content hash, so hash-skip idempotency leaves the index row stale until a full rebuild.\n\nEnumeration verdict (task #10): the ONLY high-realism channel is the tool_result outcome trio — code_parser.py _task_output_outcome (lines ~900-925) documents Claude Code polling a background task twice with the real verdict only on the second poll; that corrected outcome is currently hash-invisible. Write path already treats is_error/exit_code as content-bearing (block-level content_hash in write.py). Explicit NO-ACTION set: signature (deliberately excluded — providers re-sign every replay; including it would break fork-prefix/citation matching, vf9x); position/branch_index/is_active_path (documented array-order-exclusion design); message-level usage/model/stop_reason (low-realism under append-only acquisition; unresolved items stay unresolved, not folded in).\n\nTIMING (the actual reason this bead exists now): xselt lowering_fingerprint hashes pipeline/ids.py identity/hash function source — one global value; ANY later widening forces a full-archive differential reparse by construction. Landing inside the 818fy batch makes that reparse free (already happening); landing after costs a dedicated full pass.\n","acceptance_criteria":"pipeline/ids.py _content_block_payload includes tool_result_is_error, tool_result_exit_code, tool_result_outcome_unknown_reason (None-sentinel normalized). signature remains EXCLUDED (deliberate, vf9x re-sign-on-replay). Test: two parses differing only in is_error produce different session content hashes; re-acquisition with a corrected outcome triggers re-write. Landed inside the 818fy semantic-reparse batch (before or with the rebuild), never after xselt stamps without a fingerprint plan.","status":"open","priority":2,"issue_type":"bug","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T12:19:08Z","created_by":"Sinity","updated_at":"2026-08-03T12:19:08Z","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-m73wk","title":"Verify v50 thinking/reasoning recovery post-rebuild (zn1k-shaped successor for the unmet 8b10 closure gate)","description":"V-class (reindex-gate-hunt task #6, adjudicated 2026-08-03): gates 818fy ACCEPTANCE completeness, not its start. Beads r39b/mctu/8b10 (v50: recover Claude Code thinking blocks dropped by an if-text guard in base_support.py content_blocks_from_segments; materialize standalone Codex reasoning records) were closed 2026-08-02T22:09Z although 8b10's own notes state its closure gate — post-rebuild sum(thinking_count) non-zero — was never met (the rebuild has not run). Grep-verified: zero coverage of thinking_count/reasoning-block checks in 818fy, f1vg, r9xsj, or t0m73. The structurally identical v48 case (zn1k) is correctly open with depends_on:818fy — this bead restores the same pattern for v50. 818fy AC #5 amendment recorded in 818fy notes (same beading batch).\n\nThird instance of the recurring closure pattern (forward-fix closes, verification/repair AC deferred without successor) — see the closure-discipline process bead from this batch.\n","acceptance_criteria":"After the 818fy rebuild promotes: per-origin sum(thinking_count) \u003e 0 for claude-code-session and codex-session populations known to carry thinking/reasoning in raw; the specific pre-v50 dropped shapes (empty-text+signature-only THINKING segments; standalone Codex reasoning records) demonstrably materialize as blocks. Result recorded on this bead; if zero, the v50 parser fix is re-opened as failed.","status":"open","priority":2,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T12:19:08Z","created_by":"Sinity","updated_at":"2026-08-03T12:19:08Z","dependencies":[{"issue_id":"polylogue-m73wk","depends_on_id":"polylogue-818fy","type":"blocks","created_at":"2026-08-03T14:19:07Z","created_by":"Sinity","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} -{"_type":"issue","id":"polylogue-1cbeh","title":"Wire merge-gate verdict as a GitHub status check to enable safe auto-merge","description":"Coordinator manually cycled checkout-\u003etest-\u003emerge-gate-\u003emerge 18 times this session (2026-08-03), one PR at a time -- the single biggest process bottleneck found this session. Every piece needed for safe auto-merge already exists except one: merge-gate check verdict is a local CLI result, not a GitHub-visible check/status, so `gh pr merge --auto` and branch protection have nothing to wait on.\n\nFix: after a lane records its own merge-gate receipt (per the standing lane.md contract added today) and merge-gate check passes (including its CodeRabbit grace-period poll and unacked-actionable-comment block), post a GitHub commit status (`gh api repos/OWNER/REPO/statuses/SHA -f state=success -f context=merge-gate ...`) reflecting that verdict. Add merge-gate to required status checks in branch protection. Lanes then call `gh pr merge --auto --squash --delete-branch` as their actual final step instead of leaving the PR open for a human/coordinator to manually cycle through.\n\nSafety already covered by existing merge-gate semantics: real CodeRabbit findings keep the status pending/failed (forcing the same triage done by hand today for PRs #3613/#3631); real git conflicts just fail to merge, GitHub handles that natively; per-PR CI already required (lint+mypy). No new risk surface, just making an existing local verdict visible to GitHubs own gating mechanism.","status":"open","priority":2,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T10:53:10Z","created_by":"Sinity","updated_at":"2026-08-03T10:53:10Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"_type":"issue","id":"polylogue-1cbeh","title":"Wire merge-gate verdict as a GitHub status check to enable safe auto-merge","description":"Coordinator manually cycled checkout-\u003etest-\u003emerge-gate-\u003emerge 18 times this session (2026-08-03), one PR at a time -- the single biggest process bottleneck found this session. Every piece needed for safe auto-merge already exists except one: merge-gate check verdict is a local CLI result, not a GitHub-visible check/status, so `gh pr merge --auto` and branch protection have nothing to wait on.\n\nFix: after a lane records its own merge-gate receipt (per the standing lane.md contract added today) and merge-gate check passes (including its CodeRabbit grace-period poll and unacked-actionable-comment block), post a GitHub commit status (`gh api repos/OWNER/REPO/statuses/SHA -f state=success -f context=merge-gate ...`) reflecting that verdict. Add merge-gate to required status checks in branch protection. Lanes then call `gh pr merge --auto --squash --delete-branch` as their actual final step instead of leaving the PR open for a human/coordinator to manually cycle through.\n\nSafety already covered by existing merge-gate semantics: real CodeRabbit findings keep the status pending/failed (forcing the same triage done by hand today for PRs #3613/#3631); real git conflicts just fail to merge, GitHub handles that natively; per-PR CI already required (lint+mypy). No new risk surface, just making an existing local verdict visible to GitHubs own gating mechanism.","status":"open","priority":2,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T10:53:10Z","created_by":"Sinity","updated_at":"2026-08-03T10:53:10Z","dependencies":[{"issue_id":"polylogue-1cbeh","depends_on_id":"polylogue-z7sv3","type":"blocks","created_at":"2026-08-06T08:44:53Z","created_by":"Sinity","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-e2uns","title":"Audit devtools tooling self-sprawl: dual invocation conventions, closed-loop checks, category density","design":"Findings from a 2026-08-03 cruft audit (subagent \"audit-devtools-sprawl\"), read-only, not yet acted on:\n\n1. Dual invocation conventions: 7 real, tested/used devtools modules bypass the `devtools \u003ccommand\u003e` CommandSpec dispatch entirely, invoked as raw `python -m devtools.X` / `python devtools/X.py` scripts instead: benchmark_compare_nightly.py (invoked from .github/workflows/nightly-scale.yml:73), verify_mutation_freshness.py (invoked from .github/workflows/mutation-testing.yml:52), pre_push_gate.py (invoked from .githooks/pre-push), raw_append_chain_backfill_apply.py (documented only in a bead note, polylogue-818fy), reconcile_tracker_authority.py (documented in docs/tracker-authority.md:59-60), resume_ranking_eval.py (has tests, no CommandSpec), render_semantic_card_registry.py / render_semantic_card_fixtures.py. None of these show up in `devtools status`/`--help`. Consider either wiring them into command_catalog.py for discoverability, or documenting the split convention explicitly (e.g. \"CI-only scripts\" vs \"operator commands\").\n\n2. render_semantic_card_registry.py / render_semantic_card_fixtures.py generate docs/generated/semantic-card-tool-map.md but are NOT wired into render_all.py's surface list or command_catalog — nothing re-runs or verifies them at `render all --check` time, so this doc can go silently stale, the exact failure mode the render-all freshness gate exists to prevent everywhere else.\n\n3. Category density: of 138 total CommandSpecs, `workspace` alone is 52 (38%). Not inherently a problem, but worth a pass to check for near-duplicate workspace commands.\n\n4. Closed-loop checks: 15 of ~27 commands wired into `verify --quick`/`--all` (verify manifests, verify ci-workflows, verify test-infra-currency, verify pytest-timeout-overrides, verify degrade-loudly, lab policy demo-tour-freshness/raw-payload-hash-purity/position-derived-identity/raw-authority-frontier-executability/backlog-hygiene/timestamp-doctrine/insight-honesty/docs-drift/campaign-archive-boundaries, bench slo) have zero human-facing documentation anywhere outside auto-generated reference docs — pure closed loop, only the gate itself would notice if one were subtly wrong or miscalibrated. (Note: most of these were separately audited 2026-08-03 and found to be legitimate behavioral checks, not fossilized-diff bureaucracy — this finding is about documentation/discoverability, not correctness.)\n\n5. `devtools workspace lane-init` shows only 2 commit-message mentions in git history despite CLAUDE.md billing it as \"load-bearing, use every time\" for lane worktree provisioning — worth checking whether that mandate is actually being followed in practice, or whether the doc oversells actual usage.\n\n6. A 6-way \"is docs in sync with code\" cluster exists (verify doc-commands, verify docs-coverage, lab policy docs-drift, verify manifests, verify closure-matrix, render docs-surface) — each has genuinely distinct scope/mechanism (not byte-identical duplicates) but real overlap at the category-intent level. Worth a pass to check whether any two could merge without losing coverage.\n\nNone of these are urgent; this bead exists to make the findings durable and trackable rather than let them evaporate at end of session.","notes":"Dissection 2026-08-03 category measurement (feed for this audit): devtools = 82,285 lines / 202 files; verify* 14.6K/34 files · probe/proof 12.5K/20 · render* 5.9K/21 · *report* 5.8K/14 · beads tools 3.8K/5 · workspace/lane/merge 3.5K/11 · command_catalog 2.3K. tests/unit/devtools = 34,733 (second-order verification). Verdicts proposed (operator ratifies): raw-authority proofs (scale/restart/daemon-health, 2.7K) die with the acquire-time-authority root fix; process-analytics trio (trajectory_report self-describes as 'the missing third view' beside beads_state_report + backlog-calibration) -\u003e keep backlog-calibration only (~3-4K out); claim_vs_evidence 1.7K + affordance_usage 1.4K are product questions — promote to insights/ or close as one-shots; render family audited by actual readership. Report: /realm/data/derived/reports/polylogue-structural-dissection-2026-08-03.html.\nPER-FILE DISPOSITION LIST (iteration 6, 2026-08-03 — execution is now a checklist; delete = rm file + its tests + command-catalog entry + render refs): DELETE-WITH-R1 (premise dies with acquire-time authority / drain): raw_authority_scale_proof.py 1170, raw_authority_restart_proof.py 1005, raw_authority_daemon_health_proof.py 519, verify_raw_authority_frontier_executability.py 261. DELETE-SPENT (version-pinned one-time actuators for past generations; current index far beyond their range): index_fast_forward.py 1085 ('v32-\u003ev35' in its own docstring), archive_schema_fast_forward.py 985 ('accepts only the observed v35 file set'); confirm no lifecycle.py import ties first (the runtime fast-forward in storage/sqlite/lifecycle.py is SEPARATE and stays). DELETE-SPENT: codex_exec_child_census.py 308 (one-shot census comparing pre/post child-projection parsers; projection shipped). DELETE-NOW: trajectory_report.py 1173 (third dev-process view); beads_state_report.py 2542 AFTER folding its graph-health checks into workspace backlog-calibration (d63uz overlaps — coordinate). PROMOTE-OR-DELETE after their one-shot runs: claim_vs_evidence.py 1712 (67ac owns the experiment), affordance_usage.py 1416 (product analytics question — insights/ or gone). PENDING DISSECT-10 audit: continuity_replay.py 1935 + mandate_continuity_replay + render_product_workflows (the product/workflows closed-loop family — disposition follows the executable-route audit). KEEP: dev_loop.py (live dev preflight), deployment_smoke.py (deployed-surface probe), daemon_workload_probe.py (operator diagnostic; slim candidate later), verify.py/verify_runs.py/command_catalog.py (harness core), workspace/lane/merge family (load-bearing fanout tooling). Everything not named: unexamined, do not sweep blind.","status":"open","priority":2,"issue_type":"chore","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T10:50:18Z","created_by":"Sinity","updated_at":"2026-08-03T14:25:12Z","dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-m6tjl","title":"Origins x capabilities matrix: parser claims vs live census, every cell verdicted","description":"9 origins x (parent links, titles, timestamp confidence, attachments, tool pairing, thread structure, events, costs): parser-code claims cross-checked against mode=ro census; every empty cell is declared-impossible (provider does not ship it) or a finding. ksgg found one hole; the matrix denominator is exact.","acceptance_criteria":"1. Rendered matrix committed (doc or generated). 2. Every empty cell annotated. 3. New findings beaded with discovered-from this bead.","notes":"Dissection 2026-08-03 L12 retriage: run ONCE as an audit; the standing-matrix half belongs to OriginSpec (2qx) which declares capabilities as data — a hand-maintained matrix beside it would be a second register of the same facts.","status":"open","priority":2,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T07:40:55Z","created_by":"Sinity","updated_at":"2026-08-03T13:16:17Z","labels":["area:sources"],"dependencies":[{"issue_id":"polylogue-m6tjl","depends_on_id":"polylogue-ksgg","type":"relates-to","created_at":"2026-08-03T09:40:54Z","created_by":"Sinity","metadata":"{}"},{"issue_id":"polylogue-m6tjl","depends_on_id":"polylogue-wwph1","type":"relates-to","created_at":"2026-08-03T09:40:54Z","created_by":"Sinity","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"_type":"issue","id":"polylogue-ky67h","title":"lab check: evidence-of-execution — automation with zero lifetime runs","description":"Inventory every daemon loop/convergence stage/maintenance path from code; join against live ops.db daemon events. Believed-running-never-fired (zoek0 class; 5xxmc's frozen 12 loops) becomes a standing check — the automagic-invariants doctrine given teeth.","acceptance_criteria":"1. Stage inventory is code-derived, not hand-listed. 2. Zero-execution automations reported with last-run timestamps for the rest. 3. Dedupe vs t0m73 resolved.","status":"open","priority":2,"issue_type":"task","owner":"ezo.dev@gmail.com","created_at":"2026-08-03T07:40:51Z","created_by":"Sinity","updated_at":"2026-08-03T07:40:51Z","labels":["area:daemon"],"dependencies":[{"issue_id":"polylogue-ky67h","depends_on_id":"polylogue-t0m73","type":"relates-to","created_at":"2026-08-03T09:40:51Z","created_by":"Sinity","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} diff --git a/.circleci/config.yml b/.circleci/config.yml index d7b9a9a5bf..e38e7371bf 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -70,18 +70,24 @@ jobs: - mypy-v1- - run: name: Public claims gate - command: ~/.local/bin/uv run devtools verify public-claims --json + command: ~/.local/bin/uv run devtools verify public-claims --json | tee /tmp/polylogue-public-claims.json - run: name: Structured PR scope carrier command: | - if [ -z "${CIRCLE_PULL_REQUEST:-}" ]; then - if [ "${CIRCLE_BRANCH:-}" = "master" ]; then - exit 0 - fi - echo "quick-gate requires CIRCLE_PULL_REQUEST for every non-master build" >&2 - exit 1 + set -o pipefail + if [ "${CIRCLE_BRANCH:-}" = "master" ]; then + exit 0 + fi + if [ -n "${CIRCLE_PULL_REQUEST:-}" ]; then + ~/.local/bin/uv run devtools workspace pr-scope check-ci \ + --pr "${CIRCLE_PULL_REQUEST##*/}" \ + --repo "Sinity/polylogue" \ + --expected-head-sha "${CIRCLE_SHA1}" 2>&1 | tee /tmp/polylogue-pr-scope.log + else + ~/.local/bin/uv run devtools workspace pr-scope check-ci \ + --repo "Sinity/polylogue" \ + --expected-head-sha "${CIRCLE_SHA1}" 2>&1 | tee /tmp/polylogue-pr-scope.log fi - ~/.local/bin/uv run devtools workspace pr-scope check --pr "${CIRCLE_PULL_REQUEST##*/}" - run: name: devtools verify --quick command: ~/.local/bin/uv run devtools verify --quick @@ -89,6 +95,14 @@ jobs: key: mypy-v1-{{ .Branch }}-{{ epoch }} paths: - .mypy_cache + - store_artifacts: + path: /tmp/polylogue-public-claims.json + destination: diagnostics + when: always + - store_artifacts: + path: /tmp/polylogue-pr-scope.log + destination: diagnostics + when: always lab-policies: docker: diff --git a/.claude/agents/lane.md b/.claude/agents/lane.md index af0d26373f..7d1b29ce1f 100644 --- a/.claude/agents/lane.md +++ b/.claude/agents/lane.md @@ -109,6 +109,15 @@ Open a PR (branch off `master`, conventional commit-style subject, rejected if there was a real fork. - **Verification** — the exact commands you ran and the output line that matters, not "tests pass". +- **Bead disposition matrix** — one whole-Bead disposition per assigned ID, + typed evidence refs, and an existing open successor for every residual + outcome. + +Before publishing the PR as non-draft, render the versioned embedded carrier +with `devtools workspace pr-scope render --input `, put that exact +comment beside the human matrix, and validate the published PR with +`devtools workspace pr-scope check --pr `. Never infer a disposition from +Bead acceptance prose or invent a successor ID. Reference any bead with neutral wording only (`Ref polylogue-xxxx` / `Ref #N`). **Never use GitHub resolver keywords** (closes/fixes/resolves) diff --git a/.codex/agents/narrow-worker.toml b/.codex/agents/narrow-worker.toml index da1bda7b90..39cc3a0434 100644 --- a/.codex/agents/narrow-worker.toml +++ b/.codex/agents/narrow-worker.toml @@ -27,10 +27,10 @@ write to Beads, do not merge or push, do not touch /realm/db/polylogue. Commit coherent checkpoints. Return exact changed files, commands run and their output, an acceptance-criteria match table, and residual uncertainty. -For a lane that publishes a PR, emit the versioned PR-scope carrier before the -non-draft PR is opened: render it from assigned Bead IDs, typed whole-Bead -dispositions, evidence refs, and open successors for residual scope using -`devtools workspace pr-scope render`. Embed the result in the PR body and run -`devtools workspace pr-scope check --pr `. Never derive a disposition by -parsing acceptance prose or fabricate a successor Bead ID. +Provide the coordinator with a JSON scope input containing assigned Bead IDs, +typed whole-Bead dispositions, evidence refs, and open successors for residual +scope. The coordinator owns rendering the versioned carrier, embedding it in +the PR body, validating the published non-draft PR, and opening or updating the +PR. Never derive a disposition by parsing acceptance prose or fabricate a +successor Bead ID. """ diff --git a/.codex/agents/worker.toml b/.codex/agents/worker.toml index 1414d3bec7..d84b7442a9 100644 --- a/.codex/agents/worker.toml +++ b/.codex/agents/worker.toml @@ -35,12 +35,11 @@ changed files; the exact commands you ran and their output; an acceptance-criteria match table against the bead(s) you were assigned; residual uncertainty; and the commit hash(es). -Before publishing a non-draft PR, create a JSON scope input that names every -assigned Bead, one whole-Bead disposition per ID, typed evidence refs, and an -existing open successor for every partial/deferred/superseded outcome. Render -the embedded carrier with `devtools workspace pr-scope render`, put that exact -comment in the PR body beside the human disposition matrix, and validate the -published PR with `devtools workspace pr-scope check --pr `. Do not ask a -machine to infer Bead acceptance from prose and do not invent missing Bead IDs; -report missing IDs to the coordinator. +Provide the coordinator with a JSON scope input that names every assigned Bead, +one whole-Bead disposition per ID, typed evidence refs, and an existing open +successor for every partial/deferred/superseded outcome. The coordinator owns +rendering the embedded carrier, putting it in the PR body, validating the +published non-draft PR, and opening or updating the PR. Do not ask a machine to +infer Bead acceptance from prose and do not invent missing Bead IDs; report +missing IDs to the coordinator. """ diff --git a/.github/workflows/pr-state-guard.yml b/.github/workflows/pr-state-guard.yml deleted file mode 100644 index 62e7ee28d9..0000000000 --- a/.github/workflows/pr-state-guard.yml +++ /dev/null @@ -1,71 +0,0 @@ -name: PR State Guard - -on: - pull_request: - branches: [master] - types: - - opened - - edited - - reopened - - synchronize - - ready_for_review - -permissions: - # Read repository metadata for this checkout-free PR workflow. - contents: read - # Read pull request metadata exposed through the event payload. - pull-requests: read - -concurrency: - group: ${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true - -jobs: - issue-state-keywords: - name: reject issue state keywords - timeout-minutes: 3 - runs-on: ubuntu-latest - steps: - - name: Check PR title and body - run: | - python3 - <<'PY' - import json - import os - import re - import sys - - event_path = os.environ.get("GITHUB_EVENT_PATH") - if not event_path: - print("Error: GITHUB_EVENT_PATH is not set") - sys.exit(1) - try: - with open(event_path, encoding="utf-8") as handle: - event = json.load(handle) - except FileNotFoundError: - print(f"Error: GitHub event payload not found: {event_path}") - sys.exit(1) - except json.JSONDecodeError as exc: - print(f"Error: GitHub event payload is not valid JSON: {exc}") - sys.exit(1) - - pr = event.get("pull_request") or {} - fields = { - "title": pr.get("title") or "", - "body": pr.get("body") or "", - } - pattern = re.compile( - r"\b(?:close(?:s|d)?|fix(?:es|ed)?|resolve(?:s|d)?)\s*:?\s+" - r"(?:(?:[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+)?#\d+\b|" - r"https://github\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+/issues/\d+\b)", - re.IGNORECASE, - ) - hits = [(name, match.group(0)) for name, text in fields.items() for match in pattern.finditer(text)] - if hits: - print("PR title/body contains GitHub issue-state keywords next to issue refs.") - print("Use neutral references such as 'Ref #NNN' and describe remaining scope explicitly.") - for field, hit in hits: - print(f"- {field}: {hit!r}") - sys.exit(1) - - print("PR state guard: no issue-state keywords found") - PY diff --git a/CLAUDE.md b/CLAUDE.md index 80e1f5affc..8ec1019b7d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -383,7 +383,13 @@ workflow, not optional conveniences — use them at the point named, every time: it in the PR body beside the human whole-Bead disposition matrix, then run `devtools workspace pr-scope check --pr `. The carrier binds the exact head SHA, canonical Bead records, typed dispositions, evidence refs, and - open successors for residual work; it never parses acceptance prose. + open successors for residual work; it never parses acceptance prose. After + the final commit is created, regenerate the carrier for that exact SHA and + update the PR body before pushing; CircleCI does not rerun for a body-only + edit. + CircleCI uses `pr-scope check-ci`, resolves PR metadata through public GitHub + REST when `CIRCLE_PULL_REQUEST` is absent, and executes the validator from + the PR base revision so a PR cannot weaken its own scope gate. - **Immediately after spawning a worktree-isolated lane, not after it reports back**: `devtools workspace verify-worktree --expect-branch ` — confirms the worktree is real and isolated before the lane has diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index cb97011317..4698fb1520 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -219,7 +219,7 @@ The repository should stay aligned with the workflow above: - protect `master` against direct pushes - require pull requests for normal changes -- require the `CI`, `Nix`, and `PR State Guard` checks before merge +- require the authoritative CI checks before merge - keep squash merge enabled and leave merge-commit and rebase-merge disabled - enable automatic deletion of head branches after merge - allow Update branch for stale PRs diff --git a/devtools/command_catalog.py b/devtools/command_catalog.py index f1990df59e..b4f162939b 100644 --- a/devtools/command_catalog.py +++ b/devtools/command_catalog.py @@ -782,6 +782,7 @@ class CatalogBypassSite: examples=( "devtools workspace pr-scope render --input .agent/pr-scope.json > /tmp/pr-scope.md", "devtools workspace pr-scope check --pr 3517", + "devtools workspace pr-scope check-ci --pr 3517 --repo Sinity/polylogue --expected-head-sha $(git rev-parse HEAD)", "devtools workspace pr-scope check --body-file pr-body.md --head-sha $(git rev-parse HEAD)", ), ), @@ -829,8 +830,7 @@ class CatalogBypassSite: "Replace a bare `gh pr merge --squash` with this at the actual merge boundary " "(polylogue-ct3r2 / polylogue-t6iga: duplicate filings of the same finding -- " "`merge-gate record/check` and the one-full-verify-per-train rule both existed but " - "fired only if a coordinator remembered to invoke them). `merge ` auto-records a " - "validates the non-draft PR's structured scope carrier and auto-records a merge-gate receipt if none is fresh for the current head sha (running `--command`, " + "fired only if a coordinator remembered to invoke them). `merge ` validates the non-draft PR's structured scope carrier and auto-records a merge-gate receipt if none is fresh for the current head sha (running `--command`, " 'default "devtools verify"), runs `merge-gate check` and refuses to merge on any ' "BLOCK, strips a doubled `(#N) (#N)` squash-subject suffix (the 2026-07-12/13 " "incident), then runs the actual `gh pr merge --squash`. `--dry-run` runs every check " diff --git a/devtools/merge_boundary.py b/devtools/merge_boundary.py index ab3479e514..3b0e05890e 100644 --- a/devtools/merge_boundary.py +++ b/devtools/merge_boundary.py @@ -197,7 +197,17 @@ def cmd_merge( return 0 merge_result = subprocess.run( - ["gh", "pr", "merge", str(pr), "--squash", "--subject", clean_title], + [ + "gh", + "pr", + "merge", + str(pr), + "--squash", + "--match-head-commit", + head_sha, + "--subject", + clean_title, + ], capture_output=True, text=True, timeout=120, diff --git a/devtools/merge_gate.py b/devtools/merge_gate.py index b55ce69fb3..b728d54c3e 100644 --- a/devtools/merge_gate.py +++ b/devtools/merge_gate.py @@ -222,17 +222,6 @@ def cmd_record(pr: int, command: str) -> int: info = _gh_json(["pr", "view", str(pr), "--json", "headRefOid,headRefName,body,isDraft"]) head_sha = info["headRefOid"] - scope = pr_scope.validate_pr_body( - info.get("body") or "", - head_sha=head_sha, - is_draft=bool(info.get("isDraft")), - ) - if not scope.ok: - print(f"REFUSING to record: PR #{pr} has an invalid structured pr-scope carrier:", file=sys.stderr) - for reason in scope.reasons: - print(f" - {reason}", file=sys.stderr) - return 2 - local_head = _git_head_sha() if local_head != head_sha: print( @@ -251,6 +240,17 @@ def cmd_record(pr: int, command: str) -> int: ) return 2 + scope = pr_scope.validate_pr_body( + info.get("body") or "", + head_sha=head_sha, + is_draft=bool(info.get("isDraft")), + ) + if not scope.ok: + print(f"REFUSING to record: PR #{pr} has an invalid structured pr-scope carrier:", file=sys.stderr) + for reason in scope.reasons: + print(f" - {reason}", file=sys.stderr) + return 2 + argv = shlex.split(command) if not argv: print("REFUSING to record: --command is empty after shell splitting.", file=sys.stderr) @@ -461,6 +461,16 @@ def cmd_check( verdict.reasons.append( "receipt pr_scope_digest does not match the current carrier -- re-record after scope changes" ) + if receipt.get("pr_scope_beads_digest") != scope.beads_digest: + verdict.ok = False + verdict.reasons.append( + "receipt pr_scope_beads_digest does not match the current canonical Bead records -- re-record" + ) + if receipt.get("pr_scope_assigned_beads") != scope.assigned_beads: + verdict.ok = False + verdict.reasons.append( + "receipt pr_scope_assigned_beads does not match the current carrier -- re-record" + ) age_s = time.time() - receipt.get("recorded_at", 0) if age_s > max_age_s: verdict.ok = False diff --git a/devtools/pr_scope.py b/devtools/pr_scope.py index a30993d9b2..8dab3e4907 100644 --- a/devtools/pr_scope.py +++ b/devtools/pr_scope.py @@ -16,19 +16,47 @@ import re import subprocess import sys +import tempfile from dataclasses import asdict, dataclass, field +from enum import StrEnum from pathlib import Path from typing import Any -from urllib import request +from urllib import error, parse, request _CARRIER_PREFIX = "polylogue-pr-scope:v1" _CARRIER_START = f"" _VERSION = 1 _BEADS_PATH = Path(".beads/issues.jsonl") -_DISPOSITIONS = frozenset({"satisfied", "partial", "deferred", "superseded"}) -_RESIDUAL_DISPOSITIONS = frozenset({"partial", "deferred", "superseded"}) -_EVIDENCE_KINDS = frozenset({"command", "commit", "diff", "receipt", "review", "test"}) +_GITHUB_API_URL = "https://api.github.com" +_REPOSITORY_PATTERN = re.compile(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+") +_CARRIER_KEYS = frozenset({"version", "head_sha", "assigned_beads", "beads_digest", "dispositions", "scope_digest"}) +_DISPOSITION_KEYS = frozenset({"bead_id", "disposition", "evidence", "successors"}) +_EVIDENCE_KEYS = frozenset({"kind", "ref"}) + + +class ScopeDisposition(StrEnum): + SATISFIED = "satisfied" + PARTIAL = "partial" + DEFERRED = "deferred" + SUPERSEDED = "superseded" + + +class EvidenceKind(StrEnum): + COMMAND = "command" + COMMIT = "commit" + DIFF = "diff" + RECEIPT = "receipt" + REVIEW = "review" + TEST = "test" + + +_DISPOSITIONS = frozenset(item.value for item in ScopeDisposition) +_RESIDUAL_DISPOSITIONS = frozenset( + {ScopeDisposition.PARTIAL.value, ScopeDisposition.DEFERRED.value, ScopeDisposition.SUPERSEDED.value} +) +_EVIDENCE_KINDS = frozenset(item.value for item in EvidenceKind) +_SUCCESSOR_LINK_TYPES = frozenset({"blocks", "discovered-from", "relates-to", "supersedes"}) @dataclass(frozen=True, slots=True) @@ -40,6 +68,18 @@ class ScopeVerdict: assigned_beads: list[str] = field(default_factory=list) +@dataclass(frozen=True, slots=True) +class PullRequestMetadata: + body: str + head_sha: str + base_sha: str + is_draft: bool + + +class NoOpenPullRequestError(ValueError): + """CI checked a commit that has no open pull request to validate.""" + + def _canonical_json(value: object) -> str: return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")) @@ -77,6 +117,21 @@ def canonical_beads_digest(records: dict[str, dict[str, Any]], bead_ids: list[st return _digest({"version": _VERSION, "records": [records[bead_id] for bead_id in sorted(bead_ids)]}) +def _successor_is_linked(source_id: str, successor_id: str, records: dict[str, dict[str, Any]]) -> bool: + """Require a durable Beads relationship between a source and its successor.""" + for record_id, target_id in ((source_id, successor_id), (successor_id, source_id)): + record = records.get(record_id) + dependencies = record.get("dependencies") if record is not None else None + if not isinstance(dependencies, list): + continue + for dependency in dependencies: + if not isinstance(dependency, dict): + continue + if dependency.get("depends_on_id") == target_id and dependency.get("type") in _SUCCESSOR_LINK_TYPES: + return True + return False + + def carrier_digest(carrier: dict[str, Any]) -> str: """Digest a carrier excluding its self-referential digest field.""" payload = dict(carrier) @@ -104,6 +159,22 @@ def extract_carrier(body: str) -> tuple[dict[str, Any] | None, list[str]]: return carrier, [] +def _validate_keys( + value: dict[str, Any], + *, + label: str, + allowed: frozenset[str], + required: frozenset[str], + reasons: list[str], +) -> None: + unknown = sorted(set(value) - allowed) + missing = sorted(required - set(value)) + if unknown: + reasons.append(f"{label} has unknown field(s): {', '.join(unknown)}") + if missing: + reasons.append(f"{label} is missing required field(s): {', '.join(missing)}") + + def validate_carrier( carrier: dict[str, Any], *, @@ -112,6 +183,13 @@ def validate_carrier( beads_path: Path = _BEADS_PATH, ) -> ScopeVerdict: reasons: list[str] = [] + _validate_keys( + carrier, + label="carrier", + allowed=_CARRIER_KEYS, + required=_CARRIER_KEYS, + reasons=reasons, + ) assigned = carrier.get("assigned_beads") if not isinstance(assigned, list) or not assigned or not all(isinstance(item, str) and item for item in assigned): reasons.append("assigned_beads must be a non-empty list of Bead IDs") @@ -155,6 +233,13 @@ def validate_carrier( if not isinstance(entry, dict) or not isinstance(entry.get("bead_id"), str): reasons.append("each disposition must be an object with a bead_id") continue + _validate_keys( + entry, + label=f"disposition for {entry['bead_id']}", + allowed=_DISPOSITION_KEYS, + required=_DISPOSITION_KEYS, + reasons=reasons, + ) bead_id = entry["bead_id"] if bead_id in by_bead: reasons.append(f"duplicate disposition for assigned Bead {bead_id}") @@ -175,6 +260,14 @@ def validate_carrier( reasons.append(f"{bead_id}: disposition needs at least one typed evidence reference") else: for ref in evidence: + if isinstance(ref, dict): + _validate_keys( + ref, + label=f"evidence for {bead_id}", + allowed=_EVIDENCE_KEYS, + required=_EVIDENCE_KEYS, + reasons=reasons, + ) if ( not isinstance(ref, dict) or ref.get("kind") not in _EVIDENCE_KINDS @@ -199,6 +292,8 @@ def validate_carrier( reasons.append(f"{bead_id}: successor {successor} is unknown") elif record.get("status") == "closed": reasons.append(f"{bead_id}: successor {successor} is closed") + elif not _successor_is_linked(bead_id, successor, records): + reasons.append(f"{bead_id}: successor {successor} has no durable Beads relationship") if successor == bead_id: reasons.append(f"{bead_id}: cannot name itself as a successor") @@ -239,6 +334,9 @@ def build_carrier(input_payload: dict[str, Any], *, head_sha: str, beads_path: P raise ValueError("input assigned_beads must be a list of Bead IDs") carrier["beads_digest"] = canonical_beads_digest(load_bead_records(beads_path), assigned) carrier["scope_digest"] = carrier_digest(carrier) + verdict = validate_carrier(carrier, head_sha=head_sha, is_draft=False, beads_path=beads_path) + if not verdict.ok: + raise ValueError("invalid scope input: " + "; ".join(verdict.reasons)) return carrier @@ -247,40 +345,221 @@ def _git_head_sha() -> str: return result.stdout.strip() -def _github_repo_slug() -> str: +def _repository_from_remote(remote: str) -> str | None: + value = remote.strip() + if value.startswith("git@github.com:"): + value = value.removeprefix("git@github.com:") + elif "github.com/" in value: + value = value.split("github.com/", 1)[1] + else: + return None + value = value.removesuffix(".git").strip("/") + return value if _REPOSITORY_PATTERN.fullmatch(value) else None + + +def resolve_repository(explicit: str | None = None) -> str: + candidates = [ + explicit, + os.environ.get("GITHUB_REPOSITORY"), + ( + f"{os.environ['CIRCLE_PROJECT_USERNAME']}/{os.environ['CIRCLE_PROJECT_REPONAME']}" + if os.environ.get("CIRCLE_PROJECT_USERNAME") and os.environ.get("CIRCLE_PROJECT_REPONAME") + else None + ), + ] + for candidate in candidates: + if candidate and _REPOSITORY_PATTERN.fullmatch(candidate): + return candidate remote = subprocess.run( - ["git", "remote", "get-url", "origin"], capture_output=True, text=True, check=True - ).stdout.strip() - match = re.search(r"github\.com[:/]([^/]+)/([^/]+?)(?:\.git)?$", remote) - if match is None: - raise ValueError(f"cannot derive a GitHub repository from origin remote {remote!r}") - return f"{match.group(1)}/{match.group(2)}" + ["git", "remote", "get-url", "origin"], + capture_output=True, + text=True, + check=False, + ) + if remote.returncode == 0: + repository = _repository_from_remote(remote.stdout) + if repository: + return repository + raise ValueError("cannot resolve GitHub repository; pass --repo OWNER/REPO") -def _pr_body_from_github_api(pr: int) -> tuple[str, str, bool]: - headers = {"Accept": "application/vnd.github+json"} - token = os.environ.get("GH_TOKEN") or os.environ.get("GITHUB_TOKEN") +def _github_request_bytes( + path: str, + *, + accept: str = "application/vnd.github+json", + missing_ok: bool = False, +) -> bytes | None: + api_url = os.environ.get("GITHUB_API_URL", _GITHUB_API_URL).rstrip("/") + url = f"{api_url}/{path.lstrip('/')}" + headers = { + "Accept": accept, + "User-Agent": "polylogue-pr-scope", + "X-GitHub-Api-Version": "2022-11-28", + } + token = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN") if token: headers["Authorization"] = f"Bearer {token}" - api_request = request.Request(f"https://api.github.com/repos/{_github_repo_slug()}/pulls/{pr}", headers=headers) - with request.urlopen(api_request, timeout=30) as response: - payload = json.loads(response.read()) - return payload.get("body") or "", payload["head"]["sha"], bool(payload.get("draft")) + api_request = request.Request(url, headers=headers) + try: + with request.urlopen(api_request, timeout=30) as response: + payload = response.read() + if not isinstance(payload, bytes): + raise RuntimeError(f"GitHub API returned non-bytes content for {path}") + return payload + except error.HTTPError as exc: + if missing_ok and exc.code == 404: + return None + detail = exc.read().decode(errors="replace")[:300] + raise RuntimeError(f"GitHub API returned HTTP {exc.code} for {path}: {detail}") from exc + except error.URLError as exc: + raise RuntimeError(f"GitHub API request failed for {path}: {exc.reason}") from exc + + +def _pr_metadata_from_payload(payload: object) -> PullRequestMetadata: + if not isinstance(payload, dict): + raise ValueError("GitHub PR response must be an object") + head = payload.get("head") + base = payload.get("base") + if not isinstance(head, dict) or not isinstance(head.get("sha"), str): + raise ValueError("GitHub PR response is missing head.sha") + if not isinstance(base, dict) or not isinstance(base.get("sha"), str): + raise ValueError("GitHub PR response is missing base.sha") + return PullRequestMetadata( + body=payload.get("body") or "", + head_sha=head["sha"], + base_sha=base["sha"], + is_draft=bool(payload.get("draft")), + ) -def _pr_body(pr: int) -> tuple[str, str, bool]: - """Read the published PR through ``gh``, or GitHub's public API in CI.""" - try: +def fetch_pr_metadata(pr: int, *, repository: str) -> PullRequestMetadata: + raw = _github_request_bytes(f"repos/{repository}/pulls/{pr}") + if raw is None: + raise RuntimeError(f"GitHub API returned no metadata for PR #{pr}") + return _pr_metadata_from_payload(json.loads(raw)) + + +def fetch_pr_for_head(*, repository: str, head_sha: str) -> tuple[int, PullRequestMetadata]: + raw = _github_request_bytes(f"repos/{repository}/commits/{head_sha}/pulls") + if raw is None: + raise RuntimeError(f"GitHub API returned no PR metadata for head {head_sha[:8]}") + payload = json.loads(raw) + if not isinstance(payload, list): + raise ValueError("GitHub commit-pulls response must be a list") + candidates: list[tuple[int, PullRequestMetadata]] = [] + for item in payload: + if not isinstance(item, dict) or item.get("state") != "open" or not isinstance(item.get("number"), int): + continue + metadata = _pr_metadata_from_payload(item) + if metadata.head_sha == head_sha: + candidates.append((item["number"], metadata)) + if not candidates: + raise NoOpenPullRequestError(f"no open PR found for head {head_sha[:8]}") + if len(candidates) != 1: + raise ValueError(f"expected one open PR for head {head_sha[:8]}, found {len(candidates)}") + return candidates[0] + + +def fetch_base_validator_source(*, repository: str, base_sha: str) -> bytes | None: + local = subprocess.run( + ["git", "show", f"{base_sha}:devtools/pr_scope.py"], + capture_output=True, + check=False, + ) + if local.returncode == 0 and local.stdout: + return local.stdout + path = f"repos/{repository}/contents/devtools/pr_scope.py?ref={parse.quote(base_sha, safe='')}" + return _github_request_bytes(path, accept="application/vnd.github.raw+json", missing_ok=True) + + +def _emit_verdict(verdict: ScopeVerdict, *, head_sha: str, as_json: bool) -> int: + if as_json: + print(json.dumps(asdict(verdict), indent=2)) + elif verdict.ok: + print(f"pr-scope OK @ {head_sha[:8]}: {', '.join(verdict.assigned_beads)}") + else: + print(f"pr-scope BLOCK @ {head_sha[:8]}:") + for reason in verdict.reasons: + print(f" - {reason}") + return 0 if verdict.ok else 1 + + +def _run_validator_source( + source: bytes, + *, + metadata: PullRequestMetadata, + beads_path: Path, +) -> int: + with tempfile.TemporaryDirectory(prefix="polylogue-pr-scope-base-") as temporary: + root = Path(temporary) + validator_path = root / "pr_scope.py" + body_path = root / "pr-body.md" + validator_path.write_bytes(source) + body_path.write_text(metadata.body, encoding="utf-8") result = subprocess.run( - ["gh", "pr", "view", str(pr), "--json", "body,headRefOid,isDraft"], + [ + sys.executable, + str(validator_path), + "check", + "--body-file", + str(body_path), + "--head-sha", + metadata.head_sha, + "--beads-path", + str(beads_path.resolve()), + ], capture_output=True, text=True, - check=True, + check=False, ) - except (FileNotFoundError, subprocess.CalledProcessError): - return _pr_body_from_github_api(pr) - payload = json.loads(result.stdout) - return payload.get("body") or "", payload["headRefOid"], bool(payload.get("isDraft")) + sys.stdout.write(result.stdout) + sys.stderr.write(result.stderr) + return result.returncode + + +def check_ci_metadata( + metadata: PullRequestMetadata, + *, + repository: str, + beads_path: Path, + checkout_head_sha: str, + expected_head_sha: str | None, +) -> int: + if checkout_head_sha != metadata.head_sha: + print( + f"REFUSING CI pr-scope check: checkout HEAD {checkout_head_sha[:8]} does not match " + f"PR head {metadata.head_sha[:8]}", + file=sys.stderr, + ) + return 2 + if expected_head_sha and expected_head_sha != metadata.head_sha: + print( + f"REFUSING CI pr-scope check: CI head {expected_head_sha[:8]} does not match " + f"PR head {metadata.head_sha[:8]}", + file=sys.stderr, + ) + return 2 + if metadata.is_draft: + print("REFUSING CI pr-scope check: PR is draft; publish it before validation", file=sys.stderr) + return 2 + + base_source = fetch_base_validator_source(repository=repository, base_sha=metadata.base_sha) + if base_source is not None: + print(f"pr-scope CI authority: base revision {metadata.base_sha[:8]}") + return _run_validator_source(base_source, metadata=metadata, beads_path=beads_path) + + print( + f"pr-scope CI bootstrap: base revision {metadata.base_sha[:8]} has no validator; " + "using the checked-out validator for this first landing", + file=sys.stderr, + ) + verdict = validate_pr_body( + metadata.body, + head_sha=metadata.head_sha, + is_draft=metadata.is_draft, + beads_path=beads_path, + ) + return _emit_verdict(verdict, head_sha=metadata.head_sha, as_json=False) def main(argv: list[str] | None = None) -> int: @@ -296,10 +575,17 @@ def main(argv: list[str] | None = None) -> int: check_source = check.add_mutually_exclusive_group(required=True) check_source.add_argument("--pr", type=int, help="GitHub PR number to inspect") check_source.add_argument("--body-file", type=Path, help="PR body file for local validation") + check.add_argument("--repo", help="GitHub OWNER/REPO (default: CI metadata or origin remote)") check.add_argument("--head-sha", help="required with --body-file") check.add_argument("--beads-path", type=Path, default=_BEADS_PATH) check.add_argument("--json", action="store_true", dest="as_json") + check_ci = sub.add_parser("check-ci", help="validate with the PR base revision's authoritative checker") + check_ci.add_argument("--pr", type=int, help="GitHub PR number (default: resolve from --expected-head-sha)") + check_ci.add_argument("--repo", required=True, help="GitHub OWNER/REPO from CI metadata") + check_ci.add_argument("--expected-head-sha", default=os.environ.get("CIRCLE_SHA1")) + check_ci.add_argument("--beads-path", type=Path, default=_BEADS_PATH) + args = parser.parse_args(argv) if args.action == "render": try: @@ -313,9 +599,39 @@ def main(argv: list[str] | None = None) -> int: print(render_carrier(carrier)) return 0 + if args.action == "check-ci": + try: + repository = resolve_repository(args.repo) + if args.pr is not None: + pr_number = args.pr + metadata = fetch_pr_metadata(args.pr, repository=repository) + else: + if not args.expected_head_sha: + raise ValueError("--pr or --expected-head-sha is required") + pr_number, metadata = fetch_pr_for_head(repository=repository, head_sha=args.expected_head_sha) + print(f"pr-scope CI metadata: resolved PR #{pr_number} from head {args.expected_head_sha[:8]}") + checkout_head_sha = _git_head_sha() + return check_ci_metadata( + metadata, + repository=repository, + beads_path=args.beads_path, + checkout_head_sha=checkout_head_sha, + expected_head_sha=args.expected_head_sha, + ) + except NoOpenPullRequestError as exc: + print(f"pr-scope CI skip: {exc}", file=sys.stderr) + return 0 + except (OSError, ValueError, json.JSONDecodeError, RuntimeError, subprocess.SubprocessError) as exc: + print(f"REFUSING CI pr-scope check: {exc}", file=sys.stderr) + return 2 + try: if args.pr is not None: - body, head_sha, is_draft = _pr_body(args.pr) + repository = resolve_repository(args.repo) + metadata = fetch_pr_metadata(args.pr, repository=repository) + body = metadata.body + head_sha = metadata.head_sha + is_draft = metadata.is_draft else: if not args.head_sha: raise ValueError("--head-sha is required with --body-file") @@ -327,15 +643,7 @@ def main(argv: list[str] | None = None) -> int: print(f"REFUSING to check pr-scope carrier: {exc}", file=sys.stderr) return 2 - if args.as_json: - print(json.dumps(asdict(verdict), indent=2)) - elif verdict.ok: - print(f"pr-scope OK @ {head_sha[:8]}: {', '.join(verdict.assigned_beads)}") - else: - print(f"pr-scope BLOCK @ {head_sha[:8]}:") - for reason in verdict.reasons: - print(f" - {reason}") - return 0 if verdict.ok else 1 + return _emit_verdict(verdict, head_sha=head_sha, as_json=args.as_json) if __name__ == "__main__": diff --git a/tests/unit/devtools/test_merge_boundary.py b/tests/unit/devtools/test_merge_boundary.py index bb6c8c075e..7fa06d6c87 100644 --- a/tests/unit/devtools/test_merge_boundary.py +++ b/tests/unit/devtools/test_merge_boundary.py @@ -165,6 +165,8 @@ def _run(cmd: list[str], **kwargs: Any) -> MagicMock: assert exit_code == 0 subject_index = captured["cmd"].index("--subject") + 1 assert captured["cmd"][subject_index] == "fix: thing (#42)" + match_index = captured["cmd"].index("--match-head-commit") + 1 + assert captured["cmd"][match_index] == "abc123" def test_merge_refuses_when_pr_not_open(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: diff --git a/tests/unit/devtools/test_merge_gate.py b/tests/unit/devtools/test_merge_gate.py index 7adc6bca34..6d26285f94 100644 --- a/tests/unit/devtools/test_merge_gate.py +++ b/tests/unit/devtools/test_merge_gate.py @@ -198,6 +198,37 @@ def test_check_ok_when_receipt_fresh_and_matches_head_with_no_late_comments( assert exit_code == 0 +@pytest.mark.parametrize( + ("receipt_field", "mutated_value", "reason"), + [ + ("pr_scope_digest", "changed-body-digest", "pr_scope_digest"), + ("pr_scope_beads_digest", "stale", "pr_scope_beads_digest"), + ("pr_scope_assigned_beads", ["polylogue-other"], "pr_scope_assigned_beads"), + ], +) +def test_check_blocks_when_receipt_scope_components_are_mutated( + receipt_field: str, + mutated_value: object, + reason: str, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + capsys: pytest.CaptureFixture[str], +) -> None: + monkeypatch.chdir(tmp_path) + pr_view = _base_pr_view() + _record(monkeypatch, pr_view) + receipt_path = merge_gate._receipt_path(42) + receipt = json.loads(receipt_path.read_text()) + receipt[receipt_field] = mutated_value + receipt_path.write_text(json.dumps(receipt)) + + monkeypatch.setattr(subprocess, "run", _fake_run(pr_view, [])) + exit_code = merge_gate.cmd_check(42, max_age_s=3600, poll_rounds=1, poll_interval_s=0, as_json=False) + + assert exit_code == 1 + assert reason in capsys.readouterr().out + + def test_check_blocks_when_receipt_is_for_a_stale_sha(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: monkeypatch.chdir(tmp_path) _record(monkeypatch, _base_pr_view(head_sha="abc123")) diff --git a/tests/unit/devtools/test_pr_scope.py b/tests/unit/devtools/test_pr_scope.py index 5f8ecce6e9..d636a915d4 100644 --- a/tests/unit/devtools/test_pr_scope.py +++ b/tests/unit/devtools/test_pr_scope.py @@ -2,8 +2,8 @@ import json import subprocess -from io import BytesIO from pathlib import Path +from unittest.mock import MagicMock from urllib import request import pytest @@ -52,7 +52,13 @@ def _input(disposition: str = "satisfied", successors: list[str] | None = None) def _body(input_payload: dict[str, object], beads_path: Path, *, head_sha: str = HEAD_SHA) -> str: - carrier = pr_scope.build_carrier(input_payload, head_sha=head_sha, beads_path=beads_path) + carrier = dict(input_payload) + carrier["version"] = 1 + carrier["head_sha"] = head_sha + assigned = carrier["assigned_beads"] + assert isinstance(assigned, list) + carrier["beads_digest"] = pr_scope.canonical_beads_digest(pr_scope.load_bead_records(beads_path), assigned) + carrier["scope_digest"] = pr_scope.carrier_digest(carrier) return f"## Summary\n\nStructured scope test.\n\n{pr_scope.render_carrier(carrier)}\n" @@ -68,6 +74,20 @@ def _check( return f"{exit_code}\n{output}" +class _FakeHttpResponse: + def __init__(self, payload: bytes) -> None: + self.payload = payload + + def __enter__(self) -> _FakeHttpResponse: + return self + + def __exit__(self, *_args: object) -> None: + return None + + def read(self) -> bytes: + return self.payload + + def test_rendered_carrier_passes_the_production_check_command( beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] ) -> None: @@ -83,6 +103,226 @@ def test_rendered_carrier_passes_the_production_check_command( assert _check(rendered, beads_path, tmp_path, capsys).startswith("0\npr-scope OK") +def test_pr_check_uses_public_github_rest_without_cli_auth( + monkeypatch: pytest.MonkeyPatch, + beads_path: Path, + capsys: pytest.CaptureFixture[str], +) -> None: + body = _body(_input(), beads_path) + requests: list[request.Request] = [] + + def _urlopen(api_request: request.Request, *, timeout: int) -> _FakeHttpResponse: + assert timeout == 30 + requests.append(api_request) + payload = { + "body": body, + "draft": False, + "head": {"sha": HEAD_SHA}, + "base": {"sha": "b" * 40}, + } + return _FakeHttpResponse(json.dumps(payload).encode()) + + monkeypatch.delenv("GITHUB_TOKEN", raising=False) + monkeypatch.delenv("GH_TOKEN", raising=False) + monkeypatch.setattr(request, "urlopen", _urlopen) + + exit_code = pr_scope.main(["check", "--pr", "42", "--repo", "Sinity/polylogue", "--beads-path", str(beads_path)]) + + assert exit_code == 0 + assert capsys.readouterr().out.startswith("pr-scope OK") + assert len(requests) == 1 + assert requests[0].full_url == "https://api.github.com/repos/Sinity/polylogue/pulls/42" + assert requests[0].get_header("Authorization") is None + + +def test_ci_resolves_pr_from_exact_head_when_circle_pr_url_is_absent( + monkeypatch: pytest.MonkeyPatch, +) -> None: + requests: list[request.Request] = [] + + def _urlopen(api_request: request.Request, *, timeout: int) -> _FakeHttpResponse: + assert timeout == 30 + requests.append(api_request) + payload = [ + { + "number": 3845, + "state": "open", + "body": "carrier", + "draft": False, + "head": {"sha": HEAD_SHA}, + "base": {"sha": "b" * 40}, + }, + { + "number": 3800, + "state": "closed", + "body": "old carrier", + "draft": False, + "head": {"sha": HEAD_SHA}, + "base": {"sha": "c" * 40}, + }, + ] + return _FakeHttpResponse(json.dumps(payload).encode()) + + monkeypatch.setattr(request, "urlopen", _urlopen) + + pr_number, metadata = pr_scope.fetch_pr_for_head(repository="Sinity/polylogue", head_sha=HEAD_SHA) + + assert pr_number == 3845 + assert metadata.head_sha == HEAD_SHA + assert requests[0].full_url == f"https://api.github.com/repos/Sinity/polylogue/commits/{HEAD_SHA}/pulls" + + +def test_fetch_pr_for_head_reports_when_no_open_pr_matches( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(request, "urlopen", lambda *_args, **_kwargs: _FakeHttpResponse(b"[]")) + + with pytest.raises(pr_scope.NoOpenPullRequestError, match="no open PR"): + pr_scope.fetch_pr_for_head(repository="Sinity/polylogue", head_sha=HEAD_SHA) + + +def test_ci_skips_when_commit_has_no_open_pr( + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], +) -> None: + monkeypatch.setattr(pr_scope, "resolve_repository", lambda _repo: "Sinity/polylogue") + monkeypatch.setattr(pr_scope, "_git_head_sha", lambda: HEAD_SHA) + + def _no_pr(**_kwargs: object) -> tuple[int, pr_scope.PullRequestMetadata]: + raise pr_scope.NoOpenPullRequestError("no open PR found for head aaaaaaaa") + + monkeypatch.setattr(pr_scope, "fetch_pr_for_head", _no_pr) + + assert pr_scope.main(["check-ci", "--repo", "Sinity/polylogue", "--expected-head-sha", HEAD_SHA]) == 0 + assert "pr-scope CI skip" in capsys.readouterr().err + + +@pytest.mark.parametrize( + ("checkout_head_sha", "expected_head_sha"), + [("c" * 40, HEAD_SHA), (HEAD_SHA, "d" * 40)], +) +def test_ci_check_refuses_head_mismatch_before_fetching_base( + checkout_head_sha: str, + expected_head_sha: str, + monkeypatch: pytest.MonkeyPatch, + beads_path: Path, +) -> None: + metadata = pr_scope.PullRequestMetadata( + body=_body(_input(), beads_path), + head_sha=HEAD_SHA, + base_sha="b" * 40, + is_draft=False, + ) + fetch_base = MagicMock() + monkeypatch.setattr(pr_scope, "fetch_base_validator_source", fetch_base) + + assert ( + pr_scope.check_ci_metadata( + metadata, + repository="Sinity/polylogue", + beads_path=beads_path, + checkout_head_sha=checkout_head_sha, + expected_head_sha=expected_head_sha, + ) + == 2 + ) + fetch_base.assert_not_called() + + +def test_ci_check_refuses_draft_before_fetching_base( + monkeypatch: pytest.MonkeyPatch, + beads_path: Path, +) -> None: + metadata = pr_scope.PullRequestMetadata( + body=_body(_input(), beads_path), + head_sha=HEAD_SHA, + base_sha="b" * 40, + is_draft=True, + ) + fetch_base = MagicMock() + monkeypatch.setattr(pr_scope, "fetch_base_validator_source", fetch_base) + + assert ( + pr_scope.check_ci_metadata( + metadata, + repository="Sinity/polylogue", + beads_path=beads_path, + checkout_head_sha=HEAD_SHA, + expected_head_sha=HEAD_SHA, + ) + == 2 + ) + fetch_base.assert_not_called() + + +def test_ci_check_executes_base_revision_validator( + monkeypatch: pytest.MonkeyPatch, + beads_path: Path, +) -> None: + metadata = pr_scope.PullRequestMetadata( + body="## Summary\n\nA PR-modified validator would accept this body.", + head_sha=HEAD_SHA, + base_sha="b" * 40, + is_draft=False, + ) + base_source = Path(pr_scope.__file__).read_bytes() + current_validator = MagicMock(return_value=pr_scope.ScopeVerdict(ok=True)) + monkeypatch.setattr(pr_scope, "fetch_base_validator_source", lambda **_kwargs: base_source) + monkeypatch.setattr(pr_scope, "validate_pr_body", current_validator) + + exit_code = pr_scope.check_ci_metadata( + metadata, + repository="Sinity/polylogue", + beads_path=beads_path, + checkout_head_sha=HEAD_SHA, + expected_head_sha=HEAD_SHA, + ) + + assert exit_code == 1 + current_validator.assert_not_called() + + +def test_ci_check_bootstraps_once_when_base_has_no_validator( + monkeypatch: pytest.MonkeyPatch, + beads_path: Path, +) -> None: + metadata = pr_scope.PullRequestMetadata( + body=_body(_input(), beads_path), + head_sha=HEAD_SHA, + base_sha="b" * 40, + is_draft=False, + ) + monkeypatch.setattr(pr_scope, "fetch_base_validator_source", lambda **_kwargs: None) + + assert ( + pr_scope.check_ci_metadata( + metadata, + repository="Sinity/polylogue", + beads_path=beads_path, + checkout_head_sha=HEAD_SHA, + expected_head_sha=HEAD_SHA, + ) + == 0 + ) + + +def test_fetch_base_validator_prefers_local_base_object( + monkeypatch: pytest.MonkeyPatch, +) -> None: + github_fetch = MagicMock() + monkeypatch.setattr(pr_scope, "_github_request_bytes", github_fetch) + monkeypatch.setattr( + subprocess, + "run", + lambda *_args, **_kwargs: subprocess.CompletedProcess( + args=["git", "show"], returncode=0, stdout=b"base validator", stderr=b"" + ), + ) + + assert pr_scope.fetch_base_validator_source(repository="Sinity/polylogue", base_sha="b" * 40) == b"base validator" + github_fetch.assert_not_called() + + def test_check_rejects_carrier_bound_to_a_different_head_sha( beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] ) -> None: @@ -116,6 +356,33 @@ def test_check_rejects_closed_or_unknown_residual_successor( assert reason in result +def test_check_rejects_unlinked_residual_successor( + beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + records = [_record(ASSIGNED), _record(OPEN_SUCCESSOR)] + records[1]["dependencies"] = [ + {"issue_id": OPEN_SUCCESSOR, "depends_on_id": "polylogue-unrelated", "type": "relates-to"} + ] + beads_path.write_text("\n".join(json.dumps(record) for record in records) + "\n") + + result = _check(_body(_input("partial", [OPEN_SUCCESSOR]), beads_path), beads_path, tmp_path, capsys) + + assert result.startswith("1\n") + assert "has no durable Beads relationship" in result + + +def test_check_accepts_linked_residual_successor( + beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + records = [_record(ASSIGNED), _record(OPEN_SUCCESSOR)] + records[1]["dependencies"] = [{"issue_id": OPEN_SUCCESSOR, "depends_on_id": ASSIGNED, "type": "discovered-from"}] + beads_path.write_text("\n".join(json.dumps(record) for record in records) + "\n") + + result = _check(_body(_input("partial", [OPEN_SUCCESSOR]), beads_path), beads_path, tmp_path, capsys) + + assert result.startswith("0\n") + + def test_check_rejects_stale_canonical_beads_digest( beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] ) -> None: @@ -139,38 +406,37 @@ def test_check_rejects_partial_disposition_without_successor( assert "partial disposition requires a named successor" in result -def test_check_rejects_pr_body_without_carrier( +def test_check_rejects_unknown_schema_fields( beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] ) -> None: - result = _check("## Summary\n\nNo structured carrier.\n", beads_path, tmp_path, capsys) + carrier = pr_scope.build_carrier(_input(), head_sha=HEAD_SHA, beads_path=beads_path) + carrier["acceptance_summary"] = "silently extending v1 would make the schema ambiguous" + carrier["scope_digest"] = pr_scope.carrier_digest(carrier) - assert result.startswith("1\n") - assert "missing the structured pr-scope carrier" in result + result = _check(pr_scope.render_carrier(carrier), beads_path, tmp_path, capsys) + assert result.startswith("1\n") + assert "unknown field(s): acceptance_summary" in result -def test_pr_lookup_falls_back_to_github_api_when_gh_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None: - class Response(BytesIO): - def __enter__(self) -> Response: - return self - def __exit__(self, *args: object) -> None: - self.close() +def test_render_refuses_invalid_partial_scope_input( + beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + scope_input = tmp_path / "scope.json" + scope_input.write_text(json.dumps(_input("partial"))) - def missing_gh(*args: object, **kwargs: object) -> object: - command = args[0] - if isinstance(command, list) and command[:1] == ["gh"]: - raise FileNotFoundError("gh") - return type("Completed", (), {"stdout": "git@github.com:Sinity/polylogue.git\n"})() + exit_code = pr_scope.main( + ["render", "--input", str(scope_input), "--head-sha", HEAD_SHA, "--beads-path", str(beads_path)] + ) - seen_urls: list[str] = [] + assert exit_code == 2 + assert "partial disposition requires a named successor" in capsys.readouterr().err - def github_response(api_request: object, *, timeout: int) -> Response: - assert isinstance(api_request, request.Request) - seen_urls.append(api_request.full_url) - return Response(json.dumps({"body": "carrier", "draft": False, "head": {"sha": HEAD_SHA}}).encode()) - monkeypatch.setattr(subprocess, "run", missing_gh) - monkeypatch.setattr(request, "urlopen", github_response) +def test_check_rejects_pr_body_without_carrier( + beads_path: Path, tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + result = _check("## Summary\n\nNo structured carrier.\n", beads_path, tmp_path, capsys) - assert pr_scope._pr_body(3845) == ("carrier", HEAD_SHA, False) - assert seen_urls == ["https://api.github.com/repos/Sinity/polylogue/pulls/3845"] + assert result.startswith("1\n") + assert "missing the structured pr-scope carrier" in result