diff --git a/.github/dispatch-successor.sh b/.github/dispatch-successor.sh new file mode 100755 index 00000000..1047c3c0 --- /dev/null +++ b/.github/dispatch-successor.sh @@ -0,0 +1,101 @@ +#!/usr/bin/env bash +# +# Hand the live-refresh loop off to a fresh run of itself, and prove it landed. +# +# The loop yields at ~5.5h to stay under GitHub's 6h per-job ceiling, and used +# to depend on the `*/15` cron having left a queued successor in the shared +# concurrency group during that window. On 2026-08-27 it had not: run #1019 +# yielded at 20:01Z announcing "the queued successor takes over" and nothing +# took over. GitHub delivered ZERO scheduled fires for the workflow between +# 14:30Z and 23:17Z, when the loop was restarted by hand — 3h16m with no live +# updates during round 2 of Worlds, and 5h38m earlier the same day (08:52Z -> +# 14:30Z). Corroborating the scheduler as the culprit rather than the queue: +# refresh.yml's `0 11 * * *` cron fired that day at 20:57Z, ~10 hours late. +# +# So the hand-off no longer waits for a cron fire that may never come — the +# yielding run dispatches its own successor. `workflow_dispatch` (with +# `repository_dispatch`) is the documented exception to the rule that events +# triggered by GITHUB_TOKEN do not create a new workflow run, so this needs no +# PAT; it needs `actions: write`, which live-refresh.yml now requests. +# +# The successor is dispatched BEFORE the current run exits, so it lands in the +# concurrency group as the pending run and starts the instant the slot frees — +# the same shape the cron was supposed to produce, minus the hoping. +# +# Verification is the point, not a nicety. On 2026-08-26 an Actions incident +# left dispatches returning 204 and then silently dropping: accepted, never +# enqueued. A hand-off that reports success without a run to show for it is +# exactly the failure this script exists to prevent, so it polls for the new +# run, re-dispatches once, and only then gives up — loudly, non-zero. +# +# Usage: dispatch-successor.sh [key=value ...] +# key=value pairs are passed through as workflow inputs (`gh -f key=value`). +# +# Env: +# GH_TOKEN required; the workflow's github.token is enough +# GITHUB_REPOSITORY owner/repo (set by Actions; falls back to gh's remote) +# SUCCESSOR_WORKFLOW workflow to dispatch (default live-refresh.yml) +# SUCCESSOR_REF ref to dispatch it on (default main) +# SUCCESSOR_ATTEMPTS dispatch attempts (default 2) +# SUCCESSOR_POLL_TRIES polls per attempt (default 10) +# SUCCESSOR_POLL_SECS seconds between polls (default 3) +# The four tunables exist so the test suite can drive this without sleeping. +# +# Deliberately not `set -e`: every failure here is handled and reported, and +# the caller decides what a failed hand-off means for its own exit status. +set -uo pipefail + +workflow="${SUCCESSOR_WORKFLOW:-live-refresh.yml}" +ref="${SUCCESSOR_REF:-main}" +attempts="${SUCCESSOR_ATTEMPTS:-2}" +poll_tries="${SUCCESSOR_POLL_TRIES:-10}" +poll_secs="${SUCCESSOR_POLL_SECS:-3}" + +repo="${GH_REPO:-${GITHUB_REPOSITORY:-}}" +repo_args=() +[ -n "$repo" ] && repo_args=(--repo "$repo") + +input_args=() +for pair in "$@"; do + input_args+=(-f "$pair") +done + +# Newest run id for this workflow. Run ids increase, and the current run is +# already in the list, so "the newest id changed" is a sufficient and cheap +# test for "a new run exists" without needing to identify it precisely. +newest_run_id() { + gh run list "${repo_args[@]}" --workflow "$workflow" --limit 1 \ + --json databaseId --jq '.[0].databaseId' 2>/dev/null +} + +before="$(newest_run_id)" +if [ -z "$before" ]; then + # Not fatal on its own: we can still dispatch, we just cannot confirm by + # comparison. Treat an unreadable baseline as "nothing seen yet" so any id + # appearing below counts as the successor. + echo "warning: could not read the current newest run id for $workflow" >&2 + before="" +fi + +for attempt in $(seq 1 "$attempts"); do + if gh workflow run "$workflow" "${repo_args[@]}" --ref "$ref" "${input_args[@]}"; then + echo "dispatch attempt $attempt: accepted" + else + echo "dispatch attempt $attempt: gh workflow run failed" >&2 + continue + fi + + # A 204 is not a run. Poll until one actually shows up. + for _ in $(seq 1 "$poll_tries"); do + sleep "$poll_secs" + now="$(newest_run_id)" + if [ -n "$now" ] && [ "$now" != "$before" ]; then + echo "successor queued: run $now (dispatched $workflow on $ref)" + exit 0 + fi + done + echo "dispatch attempt $attempt: accepted but no new run appeared" >&2 +done + +echo "::error::Could not hand off to a successor run of $workflow after $attempts attempts — live updates stop here until the schedule restarts the loop." +exit 1 diff --git a/.github/invariant-alert.sh b/.github/invariant-alert.sh new file mode 100755 index 00000000..9453cf7c --- /dev/null +++ b/.github/invariant-alert.sh @@ -0,0 +1,71 @@ +#!/usr/bin/env bash +# +# Decide whether the current invariant violations are worth alerting on. +# +# dgpt/invariants.py rewrites data/cache/invariant_violations.txt on every +# refresh (empty = clean). Its lines are stable keys with no scores in them, +# precisely so this comparison is possible: the same bad player-row seen six +# minutes later produces a byte-identical line. +# +# Exit codes, so the caller can order its publish/commit/alert steps around it: +# 0 nothing to alert — clean, or this exact violation set was already alerted +# 3 NEW violations; the caller should go red (and hand off before it dies) +# 1 the script itself could not do its job +# +# Why the alerted marker is NOT kept in data/cache/ (where it used to live): +# that directory is gitignored and carried between runs only by actions/cache, +# whose save runs in a post step — and a job that ends non-zero skips it. The +# alerting run ALWAYS ends non-zero, so the marker written by the run that +# alerts is the one guaranteed never to be saved. Observed 2026-08-29 on runs +# #1027 (MPO Sander Bahnerth cur=+981) and #1029 (FPO Samantha Zaborowski +# cur=+849): both wrote the marker, both showed "Post Restore results cache: +# skipped", and neither left anything behind for the next run to compare +# against. The de-dupe the workflow comment promised could not have worked. +# +# So the marker lives at data/invariant_alerted.txt, tracked in git alongside +# the other cross-run state the pipeline reads back (data/live_signature.txt, +# data/current_ratings.json). The caller must run this BEFORE its state commit +# so the updated marker rides along with it. +# +# A clean run clears the marker. Alerting is once per contiguous episode, not +# once per season: if a bad row disappears and later comes back, that is news +# again. +set -uo pipefail + +viol="${1:-data/cache/invariant_violations.txt}" +alerted="${2:-data/invariant_alerted.txt}" + +if ! mkdir -p "$(dirname "$alerted")" 2>/dev/null; then + echo "invariant-alert: cannot create $(dirname "$alerted")" >&2 + exit 1 +fi + +# No marker file at all means the refresh never got as far as writing one. +# That is not "clean" — it is "unknown" — but it is also not a violation to +# alert on, and the refresh failing is reported by its own path. +if [ ! -f "$viol" ]; then + echo "invariant-alert: no violations file at $viol — nothing to compare" + exit 0 +fi + +if [ ! -s "$viol" ]; then + if [ -s "$alerted" ]; then + : > "$alerted" || exit 1 + echo "invariant-alert: violations cleared; marker reset" + else + echo "invariant-alert: clean" + fi + exit 0 +fi + +if cmp -s "$viol" "$alerted"; then + echo "invariant-alert: same violations already alerted; staying quiet" + exit 0 +fi + +if ! cp "$viol" "$alerted"; then + echo "invariant-alert: could not update $alerted" >&2 + exit 1 +fi +echo "invariant-alert: NEW violations ($(wc -l < "$viol" | tr -d ' ')); marker updated" +exit 3 diff --git a/.github/workflows/live-refresh.yml b/.github/workflows/live-refresh.yml index d47b655f..6a1ce0b6 100644 --- a/.github/workflows/live-refresh.yml +++ b/.github/workflows/live-refresh.yml @@ -10,20 +10,40 @@ name: Live refresh # scores actually moved. It keeps looping until no points event is live or it # nears the 6h per-job ceiling, then yields. # -# Continuation across that ceiling needs no PAT: the shared concurrency group -# keeps exactly one run active and one queued, and the coarse cron reliably -# leaves a queued successor during any 5.5h window, so it starts the instant -# the active run yields. Between rounds / overnight the loop idles on the cheap -# check (no pip, no sim) and, when nothing is live, exits in seconds. Public -# repo => Actions minutes are free, so holding a runner during an event is fine. +# Continuation across that ceiling is explicit, and needs no PAT: before it +# yields, the run dispatches its own successor (.github/dispatch-successor.sh) +# and confirms a run actually appeared. `workflow_dispatch` is the documented +# exception to the rule that GITHUB_TOKEN-triggered events create no workflow +# run, so this costs only the `actions: write` permission below. +# +# It used to rely instead on the coarse cron having left a queued successor in +# the concurrency group during the 5.5h window. On 2026-08-27 none had been: +# GitHub delivered zero scheduled fires for this workflow between 14:30Z and +# 23:17Z, so the loop yielded into nothing and the site sat on a stale bundle +# for 3h16m during round 2 of Worlds (and 5h38m earlier the same day). The cron +# now only has to START the loop when an event begins; it is no longer what +# keeps it alive. Between rounds / overnight the loop idles on the cheap check +# (no pip, no sim) and, when nothing is live, exits in seconds. Public repo => +# Actions minutes are free, so holding a runner during an event is fine. on: schedule: - - cron: "*/15 * * * *" # only needs to (re)start / keep a successor queued - workflow_dispatch: {} + - cron: "*/15 * * * *" # only needs to (re)start the loop when one goes live + workflow_dispatch: + inputs: + after_failure: + # Deliberately a string, not a boolean: this is dispatched by + # `gh workflow run -f after_failure=true`, which sends inputs as + # strings, and the loop compares it as one. A boolean-typed input + # would put a coercion between the two for no benefit — nothing + # reads this except the [ "$AFTER_FAILURE" = "true" ] below. + description: "Internal: marks the one automatic retry that follows a failed run, so the chain stops there." + type: string + default: "false" permissions: contents: write pages: write + actions: write # dispatch our own successor; see .github/dispatch-successor.sh concurrency: group: refresh-forecast-v2 # share with the weekly full refresh; never overlap @@ -75,6 +95,7 @@ jobs: PDGA_USERNAME: ${{ secrets.PDGA_USERNAME }} PDGA_PASSWORD: ${{ secrets.PDGA_PASSWORD }} GH_TOKEN: ${{ github.token }} + AFTER_FAILURE: ${{ github.event.inputs.after_failure || 'false' }} run: | set -uo pipefail git config user.name "github-actions[bot]" @@ -91,11 +112,24 @@ jobs: python -c 'import sys; from dgpt import schedule; sys.exit(0 if schedule.live_events() else 3)' } + # A crash-out gets ONE automatic retry: this run hands off with + # after_failure=true, and a run carrying that flag does not chain + # again. So a transient break (a PDGA blip, a bad runner) self-heals + # in a minute instead of waiting on a cron fire that may be hours + # away, while a persistent one cannot spin runners in a tight loop — + # past that single hop the cron is the only restart, which is the + # right amount of patience for a break that is not going to fix + # itself. The run still goes red either way; the owner is notified. fail_step() { failures=$((failures + 1)) echo "::warning::$1 (consecutive failures: $failures)" if [ "$failures" -ge 3 ]; then - echo "::error::Three consecutive live-refresh failures — failing the run so it gets seen. The cron restarts the loop." + echo "::error::Three consecutive live-refresh failures — failing the run so it gets seen." + if [ "$AFTER_FAILURE" = "true" ]; then + echo "::warning::Already the post-failure retry — not chaining again. The cron restarts the loop." + elif ! .github/dispatch-successor.sh after_failure=true; then + echo "::warning::Could not hand off after failure; the cron restarts the loop." + fi exit 1 fi } @@ -166,28 +200,75 @@ jobs: fi gh api -X POST "repos/${GITHUB_REPOSITORY}/pages/builds" >/dev/null 2>&1 || true + # Publish-gate invariants: DECIDE here, ALERT after the push. + # The decision has to happen before the state commit so the + # de-dupe marker is committed with it. It used to live in + # data/cache/, which is gitignored and carried only by + # actions/cache — whose save step is skipped when the job ends + # non-zero, which the alerting run always does. So the marker + # written by the run that alerts was the one guaranteed not to + # survive, and every restart re-alerted and died again (observed + # 2026-08-29, runs #1027 and #1029). See .github/invariant-alert.sh. + alert=0 + .github/invariant-alert.sh || alert=$? + if [ "$alert" = 1 ]; then + fail_step "invariant-alert failed" + sleep 300; continue + fi + # Cross-run state only: the bundle no longer lands on main, so - # most live iterations now make no commit here at all. + # most live iterations now make no commit here at all. This is + # also what persists data/invariant_alerted.txt. git add data predictions + state_pushed=1 if git diff --cached --quiet; then echo "no state change after refresh" else git commit -m "Refresh pipeline state ($(date -u +'%F %H:%MZ'))" # a concurrent merge/weekly refresh could have moved main - git push || { git pull --rebase --autostash && git push; } + if git push || { git pull --rebase --autostash && git push; }; then + : + else + state_pushed=0 + echo "::warning::could not push pipeline state to main" + fi fi - # publish-gate invariants: alert on NEW violations, after the push - viol=data/cache/invariant_violations.txt - if [ -s "$viol" ] && ! cmp -s "$viol" data/cache/invariant_alerted.txt; then - cp "$viol" data/cache/invariant_alerted.txt + + if [ "$alert" = 3 ]; then echo "::error::Live data failed invariant checks (data still published — see dgpt/invariants.py):" - cat "$viol" + cat data/cache/invariant_violations.txt + # Hand off before dying. The run still goes red so the owner is + # notified, but the loop no longer stays down until someone + # restarts it by hand: the successor reads the marker committed + # just above, sees this violation set as already alerted, and + # keeps publishing. On 2026-08-29 two of these deaths left the + # site on hourly updates for 2h26m and 70m during round 4. + # Unbounded on purpose — a repeat cannot re-alert, so this + # cannot chain; only a genuinely new violation set alerts again. + # That guarantee rests entirely on the marker having reached + # main, so a failed state push withholds the hand-off: a + # successor starting from the old marker WOULD re-alert, and + # that is the one way this could spin. + if [ "$state_pushed" = 1 ]; then + .github/dispatch-successor.sh \ + || echo "::warning::could not hand off after an invariant alert; the cron restarts the loop" + else + echo "::warning::state push failed, so the marker is not on main — not handing off, because the successor would re-alert. The cron restarts the loop." + fi exit 1 fi fi if [ "$(date +%s)" -ge "$BUDGET" ]; then - echo "Near job time limit — yielding; the queued successor takes over." + # Dispatch BEFORE breaking, so the successor sits in the + # concurrency group as the pending run and starts the moment this + # one releases the slot. A hand-off that cannot be confirmed is + # precisely the 2026-08-27 outage, so it fails the run instead of + # exiting quietly with nothing lined up to take over. + echo "Near job time limit — handing off to a fresh run." + if ! .github/dispatch-successor.sh; then + exit 1 + fi exit_reason=yield break fi diff --git a/HARDENING.md b/HARDENING.md index ec90679b..b54985ff 100644 --- a/HARDENING.md +++ b/HARDENING.md @@ -103,6 +103,54 @@ pattern across them, not any single bug, is what this backlog addresses. UTC-rollover bug waiting for a US Sunday finish, and the second one (permanent caching keyed on date) was sitting behind the first. +11. **The loop yielded at the job ceiling and nothing took over.** The + live loop holds a runner ~5.5h and exits under GitHub's 6h per-job + ceiling; continuation depended on the `*/15` cron having left a + queued successor in the shared concurrency group — described in the + workflow as something the coarse cron did "reliably". On 2026-08-27, + round 2 of Worlds, it had not. Run #1019 yielded at 20:01Z + announcing "the queued successor takes over" and nothing did. GitHub + delivered **zero** scheduled fires for the workflow between 14:30Z + and 23:17Z, when it was restarted by hand — including the 3h11m + after the yield, when the concurrency slot was completely free, so a + congested queue does not explain it. Corroboration that the + scheduler rather than the queue was at fault: `refresh.yml`'s + `0 11 * * *` cron fired that day at 20:57Z, ~10 hours late. That + late run is also the only reason the site was 2h15m stale instead of + 3h16m — the once-a-day job happened to land in the hole. An earlier + gap the same day went unnoticed entirely: #1018 ended 08:52Z, #1019 + started 14:30Z, 5h38m uncovered. Nothing in the pipeline was broken; + the loop, the gate, the publish path and the invariants were all + healthy, which is why no run went red and nobody was notified. + Lesson: "best effort" in GitHub's cron documentation means the + delivery rate can go to zero for hours, so anything load-bearing + built on "a fire will arrive within this window" is a scheduled + outage rather than a design — and a liveness failure that produces + no failing run produces no alert either. + +12. **The publish-gate alert took the live loop down, and could not + de-duplicate.** Two PDGA rows at Worlds carried impossible cumulative + scores on 2026-08-29 — MPO Sander Bahnerth `cur=+981` and, two hours + later, FPO Samantha Zaborowski `cur=+849`, both far outside the + per-round bounds. The checks did exactly their job: published the + data, then failed the run so the owner was notified. What nobody had + noticed is what that costs. `exit 1` ends the run, and the run is the + live loop, so each bad row also stopped the site updating — and the + de-dupe meant to keep a persistent violation from re-alerting could + never fire, because its marker lived in `data/cache/`, which is + gitignored and carried between runs only by `actions/cache`, whose + save step is skipped when the job ends non-zero. The alerting run is + always the failing run, so the marker written by the run that alerts + was precisely the one guaranteed not to survive. Every restart + re-alerted and died again. Round 4 ran on hourly updates (16:10, + 17:26, 18:38Z) instead of six-minute ones, the 17:26 refresh being the + daily cron rather than the loop. Fix: item 10's hand-off extended to + this exit, and the marker moved into tracked state so it is committed + with the rest of the cross-run state before the run dies. Lesson: an + alerting path that shares a process with the thing it monitors will + take that thing down with it, and state that only exists on the + success path cannot protect the failure path. + ## The plan Ordered by expected payoff. "In-season safe" = additive, can't change a @@ -114,7 +162,8 @@ published number. failure posture in `.github/workflows/live-refresh.yml`. Item 6 shipped 2026-08 (`.github/publish-site.sh`). Items 7 and 9 are offseason work; item 8's first two pieces (a resnapshot command, the invariant gate on -recording) are in-season safe and still open.* +recording) are in-season safe and still open. Items 10 and 11 shipped +2026-08 (`.github/dispatch-successor.sh`, `.github/invariant-alert.sh`).* ### 1. Commit the regression corpus; make it a test suite *(shipped 2026-07)* @@ -264,3 +313,54 @@ broken. - `validate.py`'s HTML parsing is regex over `