From 5c532074f681f9bbadc59886e32cc669a687b250 Mon Sep 17 00:00:00 2001 From: Joshua Temple Date: Wed, 24 Jun 2026 18:06:24 -0400 Subject: [PATCH] fix(fleet): tolerate transient watch API errors with a bounded retry The dispatch-suite and auto-promote watchers used gh run watch --exit-status, which fails the whole job on a single transient HTTP 401/403/5xx mid-poll even when the watched run succeeds. Replace with a bounded gh run view poll that retries transient API errors, exits 0 only on completed+success, fails 1 on any other conclusion, and fails closed on a stuck run or repeated errors. Closes #335. Signed-off-by: Joshua Temple --- .github/actions/dispatch-suite/action.yaml | 67 ++++++++++++++++++++-- .github/workflows/auto-promote.yaml | 58 ++++++++++++++++++- 2 files changed, 119 insertions(+), 6 deletions(-) diff --git a/.github/actions/dispatch-suite/action.yaml b/.github/actions/dispatch-suite/action.yaml index 314ed96c..407ac0e3 100644 --- a/.github/actions/dispatch-suite/action.yaml +++ b/.github/actions/dispatch-suite/action.yaml @@ -104,8 +104,65 @@ runs: echo "- **$TARGET_REPO**: [run $RUN_ID]($RUN_URL)" } >> "$GITHUB_STEP_SUMMARY" - # Block on the recovered run's conclusion. --exit-status makes gh return - # non-zero if the run concluded with a non-success result. --interval - # keeps the refresh cadence well under GitHub secondary rate limits when - # many suites are watched at once (the default 3s cadence over-polled). - gh run watch "$RUN_ID" --repo "$TARGET_REPO" --exit-status --interval "$WATCH_INTERVAL" + # Block on the recovered run's conclusion via a bounded poll loop. + # gh run watch --exit-status fails the whole step on a transient + # 401/403/5xx mid-poll, even when the watched run ultimately succeeds. + # This loop retries on transient API errors while still failing closed on + # real run failures and on timeout. + # + # MAX_ATTEMPTS * WATCH_INTERVAL = wall-clock cap (~30 min at 60s each). + MAX_ATTEMPTS=30 + CONSEC_ERRORS=0 + MAX_CONSEC_ERRORS=5 + attempt=0 + + while [ "$attempt" -lt "$MAX_ATTEMPTS" ]; do + attempt=$((attempt + 1)) + + view_output=$(gh run view "$RUN_ID" \ + --repo "$TARGET_REPO" \ + --json status,conclusion 2>&1) + exit_code=$? + + if [ "$exit_code" -ne 0 ]; then + # Decide whether this looks transient (auth blip, rate-limit, 5xx). + if echo "$view_output" | grep -qiE \ + 'HTTP (401|403|5[0-9]{2})|bad credentials|rate.?limit|temporary|connection|timed? ?out|network'; then + echo "Transient API error on attempt $attempt, retrying in ${WATCH_INTERVAL}s..." + CONSEC_ERRORS=$((CONSEC_ERRORS + 1)) + else + echo "gh run view failed (attempt $attempt): $view_output" + CONSEC_ERRORS=$((CONSEC_ERRORS + 1)) + fi + + if [ "$CONSEC_ERRORS" -ge "$MAX_CONSEC_ERRORS" ]; then + echo "::error::$MAX_CONSEC_ERRORS consecutive gh run view failures - failing closed" + exit 1 + fi + + sleep "$WATCH_INTERVAL" + continue + fi + + # Reset consecutive-error counter on a clean response. + CONSEC_ERRORS=0 + + status=$(echo "$view_output" | jq -r '.status // empty') + conclusion=$(echo "$view_output" | jq -r '.conclusion // empty') + + if [ "$status" = "completed" ]; then + if [ "$conclusion" = "success" ]; then + echo "Run $RUN_ID in $TARGET_REPO completed successfully" + exit 0 + else + echo "::error::Run $RUN_ID in $TARGET_REPO completed with conclusion: $conclusion" + exit 1 + fi + fi + + echo "Run $RUN_ID status=$status (attempt $attempt/$MAX_ATTEMPTS), waiting ${WATCH_INTERVAL}s..." + sleep "$WATCH_INTERVAL" + done + + echo "::error::Timed out waiting for run $RUN_ID in $TARGET_REPO after $MAX_ATTEMPTS attempts" + exit 1 diff --git a/.github/workflows/auto-promote.yaml b/.github/workflows/auto-promote.yaml index c5e492f0..d6d86acb 100644 --- a/.github/workflows/auto-promote.yaml +++ b/.github/workflows/auto-promote.yaml @@ -297,7 +297,63 @@ jobs: fi echo "Watching https://github.com/${GITHUB_REPOSITORY}/actions/runs/$RUN_ID" - gh run watch "$RUN_ID" --repo "${GITHUB_REPOSITORY}" --exit-status --interval 30 + + # Poll with bounded retry to survive transient 401/403/5xx errors + # that would otherwise fail this step even when the run succeeds. + # Fail closed: consecutive API errors or timeout both exit 1. + POLL_INTERVAL=30 + MAX_ATTEMPTS=60 # 60 x 30s = 30-minute cap + CONSEC_ERRORS=0 + MAX_CONSEC_ERRORS=5 + attempt=0 + + while [ "$attempt" -lt "$MAX_ATTEMPTS" ]; do + attempt=$((attempt + 1)) + + view_output=$(gh run view "$RUN_ID" \ + --repo "${GITHUB_REPOSITORY}" \ + --json status,conclusion 2>&1) + exit_code=$? + + if [ "$exit_code" -ne 0 ]; then + if echo "$view_output" | grep -qiE \ + 'HTTP (401|403|5[0-9]{2})|bad credentials|rate.?limit|temporary|connection|timed? ?out|network'; then + echo "Transient API error on attempt $attempt, retrying in ${POLL_INTERVAL}s..." + CONSEC_ERRORS=$((CONSEC_ERRORS + 1)) + else + echo "gh run view failed (attempt $attempt): $view_output" + CONSEC_ERRORS=$((CONSEC_ERRORS + 1)) + fi + + if [ "$CONSEC_ERRORS" -ge "$MAX_CONSEC_ERRORS" ]; then + echo "::error::$MAX_CONSEC_ERRORS consecutive gh run view failures - failing closed" + exit 1 + fi + + sleep "$POLL_INTERVAL" + continue + fi + + CONSEC_ERRORS=0 + status=$(echo "$view_output" | jq -r '.status // empty') + conclusion=$(echo "$view_output" | jq -r '.conclusion // empty') + + if [ "$status" = "completed" ]; then + if [ "$conclusion" = "success" ]; then + echo "Release run $RUN_ID completed successfully" + exit 0 + else + echo "::error::Release run $RUN_ID completed with conclusion: $conclusion" + exit 1 + fi + fi + + echo "Release run $RUN_ID status=$status (attempt $attempt/$MAX_ATTEMPTS), waiting ${POLL_INTERVAL}s..." + sleep "$POLL_INTERVAL" + done + + echo "::error::Timed out waiting for Release run $RUN_ID after $MAX_ATTEMPTS attempts" + exit 1 # Verify the published release matches expectations: not a draft, not a # prerelease (GoReleaser's prerelease:auto must classify a non-rc tag as a