diff --git a/.github/workflows/deploy-queue-watchdog.yml b/.github/workflows/deploy-queue-watchdog.yml new file mode 100644 index 00000000..a7670cf2 --- /dev/null +++ b/.github/workflows/deploy-queue-watchdog.yml @@ -0,0 +1,91 @@ +name: deploy-queue-watchdog + +# Alerts when a deploy or rollback run has sat in `queued` or `pending` for +# hours, or in `waiting` for most of a day. The per-cloud deploy jobs share a +# job-level concurrency group +# (`-tfstate-`, cancel-in-progress: false), so one run that never +# starts holds the slot and every later deploy or rollback waits behind it +# (issue #708: GCP and Azure dev deploys frozen from 2026-10-05). +# +# Alert-only on purpose: the job fails (red scheduled run, GitHub notifies) and +# lists the stuck runs; a human cancels them. `waiting` is how the AWS dev +# orphan in #708 showed up (no approval gate on dev), but it is also the state +# of a run awaiting a required environment approval, so it only alerts after +# 720 minutes and is labeled "approval or orphaned". The `test` job runs the +# script's self-test when the script or this workflow changes. + +on: + schedule: + - cron: "17,47 * * * *" + workflow_dispatch: + pull_request: + paths: + - ".github/workflows/deploy-queue-watchdog.yml" + - "scripts/check-stuck-deploy-runs.sh" + - "scripts/test-check-stuck-deploy-runs.sh" + +permissions: + contents: read + +concurrency: + group: deploy-queue-watchdog-${{ github.event_name }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + test: + name: Watchdog script self-test + if: github.event_name == 'pull_request' + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Run self-test + run: bash scripts/test-check-stuck-deploy-runs.sh + + watch: + name: Look for stuck deploy runs + if: github.event_name != 'pull_request' + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + actions: read + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + STUCK_AFTER_MINUTES: "120" + WAITING_AFTER_MINUTES: "720" + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Collect queued, pending and waiting runs of the deploy workflows + run: | + set -euo pipefail + : > runs.jsonl + for wf in deploy-gcp.yml deploy-azure.yml deploy-aws-lambda.yml deploy-aws-fargate.yml deploy-all.yml rollback.yml; do + for status in queued pending waiting; do + gh run list --workflow "$wf" --status "$status" --limit 50 \ + --json databaseId,workflowName,status,createdAt,url,headBranch >> runs.jsonl + done + done + jq -s 'add // []' runs.jsonl > runs.json + + - name: Fail on stuck runs + run: | + set -euo pipefail + status=0 + bash scripts/check-stuck-deploy-runs.sh < runs.json | tee report.txt || status=$? + { + echo '### Deploy queue watchdog' + echo '```' + cat report.txt + echo '```' + } >> "$GITHUB_STEP_SUMMARY" + exit "$status" diff --git a/scripts/check-stuck-deploy-runs.sh b/scripts/check-stuck-deploy-runs.sh new file mode 100644 index 00000000..5d3226d0 --- /dev/null +++ b/scripts/check-stuck-deploy-runs.sh @@ -0,0 +1,56 @@ +#!/usr/bin/env bash +# check-stuck-deploy-runs.sh +# +# Reads a JSON array of workflow runs on stdin (the output of +# `gh run list --json databaseId,workflowName,status,createdAt,url,headBranch`) +# and reports runs that have been queued or pending for STUCK_AFTER_MINUTES +# (default 120) or longer, and runs that have been waiting for +# WAITING_AFTER_MINUTES (default 720) or longer. +# +# Why: a deploy run stuck in `queued` holds the job-level concurrency group +# (`-tfstate-`, cancel-in-progress: false) and every later deploy +# or rollback waits behind it (issue #708). `waiting` is how an orphaned +# holder looked on AWS dev (no approval gate there), but it is also the state +# of a run legitimately waiting for a required environment approval, so it gets +# a much longer threshold and its own label. +# +# Exit codes: 0 nothing stuck, 1 at least one stuck run, 2 bad input. +# NOW_EPOCH overrides the clock (used by the self-test). + +set -euo pipefail + +threshold_minutes="${STUCK_AFTER_MINUTES:-120}" +waiting_minutes="${WAITING_AFTER_MINUTES:-720}" +now_epoch="${NOW_EPOCH:-$(date -u +%s)}" + +for pair in "STUCK_AFTER_MINUTES=$threshold_minutes" "WAITING_AFTER_MINUTES=$waiting_minutes"; do + if ! [[ "${pair#*=}" =~ ^[0-9]+$ ]]; then + echo "${pair%%=*} must be a non-negative integer, got '${pair#*=}'" >&2 + exit 2 + fi +done + +stuck="$(jq -r --argjson now "$now_epoch" --argjson limit "$threshold_minutes" --argjson wlimit "$waiting_minutes" ' + map(select(.status == "queued" or .status == "pending" or .status == "waiting")) + | map(. + {age_min: (($now - (.createdAt | fromdateiso8601)) / 60 | floor)}) + | map(select(.age_min >= (if .status == "waiting" then $wlimit else $limit end))) + | sort_by(.createdAt) + | .[] + | (if .status == "waiting" then "waiting (approval or orphaned)" else .status end) as $label + | "\(.workflowName) run \(.databaseId) (\(.headBranch)) has been \($label) for \(.age_min) min: \(.url)" +' 2>/dev/null)" || { + echo "input is not a JSON array of runs" >&2 + exit 2 +} + +if [[ -z "$stuck" ]]; then + echo "No deploy run has been queued or pending for ${threshold_minutes} min or more, or waiting for ${waiting_minutes} min or more." + exit 0 +fi + +echo "$stuck" +echo +echo "Stuck runs hold the deploy concurrency group and block later deploys and rollbacks." +echo "A waiting run may just be awaiting an environment approval; check the run page before cancelling." +echo "Cancel them with: gh run cancel --repo \"\$GITHUB_REPOSITORY\", then re-run the latest deploy." +exit 1 diff --git a/scripts/test-check-stuck-deploy-runs.sh b/scripts/test-check-stuck-deploy-runs.sh new file mode 100644 index 00000000..2b916c6a --- /dev/null +++ b/scripts/test-check-stuck-deploy-runs.sh @@ -0,0 +1,71 @@ +#!/usr/bin/env bash +# test-check-stuck-deploy-runs.sh +# +# Self-test for check-stuck-deploy-runs.sh. Fixed clock, fixture run lists. +# Exits 0 when all cases pass; exits 1 on any failure. + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +CHECK="${SCRIPT_DIR}/check-stuck-deploy-runs.sh" + +# 2026-10-09T12:00:00Z +NOW=1791547200 + +pass=0 +fail=0 + +run_case() { + local label="$1" expected_exit="$2" input="$3" expected_text="${4:-}" + local out actual_exit=0 + out="$(printf '%s' "$input" | NOW_EPOCH="$NOW" bash "$CHECK" 2>&1)" || actual_exit=$? + if [[ "$actual_exit" -ne "$expected_exit" ]]; then + echo "FAIL: $label (expected exit $expected_exit, got $actual_exit)" + (( fail++ )) || true + return + fi + if [[ -n "$expected_text" && "$out" != *"$expected_text"* ]]; then + echo "FAIL: $label (output missing '$expected_text')" + (( fail++ )) || true + return + fi + echo "PASS: $label" + (( pass++ )) || true +} + +run() { # id status createdAt + printf '{"databaseId":%s,"workflowName":"Deploy to GCP Cloud Run","status":"%s","createdAt":"%s","url":"https://example.invalid/runs/%s","headBranch":"main"}' "$1" "$2" "$3" "$1" +} + +old="2026-10-09T08:00:00Z" # 240 min before NOW +fresh="2026-10-09T11:30:00Z" # 30 min before NOW +edge="2026-10-09T10:00:00Z" # exactly 120 min before NOW +wedge="2026-10-09T00:00:00Z" # exactly 720 min before NOW +wold="2026-10-08T23:00:00Z" # 780 min before NOW + +run_case "empty list is clean" 0 "[]" "No deploy run" +run_case "old queued run is stuck" 1 "[$(run 1 queued "$old")]" "run 1 (main) has been queued for 240 min" +run_case "old pending run is stuck" 1 "[$(run 2 pending "$old")]" "has been pending" +run_case "fresh queued run is fine" 0 "[$(run 3 queued "$fresh")]" +run_case "run exactly at the threshold is stuck" 1 "[$(run 4 queued "$edge")]" +run_case "waiting run under the waiting threshold is fine" 0 "[$(run 5 waiting "$old")]" +run_case "waiting run past the waiting threshold is reported" 1 "[$(run 10 waiting "$wold")]" "waiting (approval or orphaned) for 780 min" +run_case "waiting run exactly at the waiting threshold is reported" 1 "[$(run 11 waiting "$wedge")]" +run_case "old in_progress run is ignored" 0 "[$(run 6 in_progress "$old")]" +run_case "one stuck among fine runs is reported" 1 "[$(run 7 queued "$fresh"),$(run 8 queued "$old"),$(run 9 waiting "$old")]" "run 8" +run_case "a waiting run does not use the short threshold" 0 "[$(run 12 waiting "$edge")]" +run_case "not JSON is a bad-input error" 2 "this is not json" +run_case "JSON object instead of array is a bad-input error" 2 '{"a":1}' + +out_exit=0 +STUCK_AFTER_MINUTES=abc bash "$CHECK" /dev/null 2>&1 || out_exit=$? +if [[ "$out_exit" -eq 2 ]]; then echo "PASS: non-numeric threshold is rejected"; (( pass++ )) || true +else echo "FAIL: non-numeric threshold (got exit $out_exit)"; (( fail++ )) || true; fi + +wexit=0 +WAITING_AFTER_MINUTES=-5 bash "$CHECK" /dev/null 2>&1 || wexit=$? +if [[ "$wexit" -eq 2 ]]; then echo "PASS: invalid waiting threshold is rejected"; (( pass++ )) || true +else echo "FAIL: invalid waiting threshold (got exit $wexit)"; (( fail++ )) || true; fi + +echo "$pass passed, $fail failed" +[[ "$fail" -eq 0 ]]