From 7e160e14d0ad3191ca96920a6280b691b76fc489 Mon Sep 17 00:00:00 2001 From: Cristian Magherusan-Stanciu Date: Fri, 9 Oct 2026 12:06:51 +0200 Subject: [PATCH 1/2] ci: alert when a deploy run sits queued or pending for hours - A deploy run that never starts holds the job-level concurrency group (-tfstate-, cancel-in-progress false), so every later deploy and rollback waits behind it. GCP and Azure dev deploys were frozen from 2026-10-05 this way and nothing alerted. - New scheduled workflow deploy-queue-watchdog lists queued and pending runs of the deploy and rollback workflows every 30 minutes and fails when any is older than 120 minutes, with the run list in the job summary. It is alert-only (actions: read); waiting runs, which are waiting for an environment approval, are not reported. - scripts/check-stuck-deploy-runs.sh holds the decision logic and scripts/test-check-stuck-deploy-runs.sh covers it with a fixed clock (11 cases); the workflow runs the self-test on pull requests that touch these files. Closes #708 --- .github/workflows/deploy-queue-watchdog.yml | 87 +++++++++++++++++++++ scripts/check-stuck-deploy-runs.sh | 48 ++++++++++++ scripts/test-check-stuck-deploy-runs.sh | 61 +++++++++++++++ 3 files changed, 196 insertions(+) create mode 100644 .github/workflows/deploy-queue-watchdog.yml create mode 100644 scripts/check-stuck-deploy-runs.sh create mode 100644 scripts/test-check-stuck-deploy-runs.sh diff --git a/.github/workflows/deploy-queue-watchdog.yml b/.github/workflows/deploy-queue-watchdog.yml new file mode 100644 index 00000000..3ebb85fe --- /dev/null +++ b/.github/workflows/deploy-queue-watchdog.yml @@ -0,0 +1,87 @@ +name: deploy-queue-watchdog + +# Alerts when a deploy or rollback run has sat in `queued` or `pending` for +# hours. The per-cloud deploy jobs share a job-level concurrency group +# (`-tfstate-`, cancel-in-progress: false), so one run that never +# starts holds the slot and every later deploy or rollback waits behind it +# (issue #708: GCP and Azure dev deploys frozen from 2026-10-05). +# +# Alert-only on purpose: the job fails (red scheduled run, GitHub notifies) and +# lists the stuck runs; a human cancels them. `waiting` runs are not reported, +# they are waiting for an environment approval. The `test` job runs the +# script's self-test when the script or this workflow changes. + +on: + schedule: + - cron: "17,47 * * * *" + workflow_dispatch: + pull_request: + paths: + - ".github/workflows/deploy-queue-watchdog.yml" + - "scripts/check-stuck-deploy-runs.sh" + - "scripts/test-check-stuck-deploy-runs.sh" + +permissions: + contents: read + +concurrency: + group: deploy-queue-watchdog-${{ github.event_name }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + test: + name: Watchdog script self-test + if: github.event_name == 'pull_request' + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Run self-test + run: bash scripts/test-check-stuck-deploy-runs.sh + + watch: + name: Look for stuck deploy runs + if: github.event_name != 'pull_request' + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + actions: read + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + STUCK_AFTER_MINUTES: "120" + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + with: + persist-credentials: false + + - name: Collect queued and pending runs of the deploy workflows + run: | + set -euo pipefail + : > runs.jsonl + for wf in deploy-gcp.yml deploy-azure.yml deploy-aws-lambda.yml deploy-aws-fargate.yml deploy-all.yml rollback.yml; do + for status in queued pending; do + gh run list --workflow "$wf" --status "$status" --limit 50 \ + --json databaseId,workflowName,status,createdAt,url,headBranch >> runs.jsonl + done + done + jq -s 'add // []' runs.jsonl > runs.json + + - name: Fail on stuck runs + run: | + set -euo pipefail + status=0 + bash scripts/check-stuck-deploy-runs.sh < runs.json | tee report.txt || status=$? + { + echo '### Deploy queue watchdog' + echo '```' + cat report.txt + echo '```' + } >> "$GITHUB_STEP_SUMMARY" + exit "$status" diff --git a/scripts/check-stuck-deploy-runs.sh b/scripts/check-stuck-deploy-runs.sh new file mode 100644 index 00000000..c63c4c38 --- /dev/null +++ b/scripts/check-stuck-deploy-runs.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +# check-stuck-deploy-runs.sh +# +# Reads a JSON array of workflow runs on stdin (the output of +# `gh run list --json databaseId,workflowName,status,createdAt,url,headBranch`) +# and reports runs that have been queued or pending for longer than +# STUCK_AFTER_MINUTES (default 120). +# +# Why: a deploy run stuck in `queued` holds the job-level concurrency group +# (`-tfstate-`, cancel-in-progress: false) and every later deploy +# or rollback waits behind it (issue #708). A `waiting` run is deliberately +# NOT reported: that is a run waiting for a required environment approval. +# +# Exit codes: 0 nothing stuck, 1 at least one stuck run, 2 bad input. +# NOW_EPOCH overrides the clock (used by the self-test). + +set -euo pipefail + +threshold_minutes="${STUCK_AFTER_MINUTES:-120}" +now_epoch="${NOW_EPOCH:-$(date -u +%s)}" + +if ! [[ "$threshold_minutes" =~ ^[0-9]+$ ]]; then + echo "STUCK_AFTER_MINUTES must be a non-negative integer, got '$threshold_minutes'" >&2 + exit 2 +fi + +stuck="$(jq -r --argjson now "$now_epoch" --argjson limit "$threshold_minutes" ' + map(select(.status == "queued" or .status == "pending")) + | map(. + {age_min: (($now - (.createdAt | fromdateiso8601)) / 60 | floor)}) + | map(select(.age_min >= $limit)) + | sort_by(.createdAt) + | .[] + | "\(.workflowName) run \(.databaseId) (\(.headBranch)) has been \(.status) for \(.age_min) min: \(.url)" +' 2>/dev/null)" || { + echo "input is not a JSON array of runs" >&2 + exit 2 +} + +if [[ -z "$stuck" ]]; then + echo "No deploy run has been queued or pending for ${threshold_minutes} min or more." + exit 0 +fi + +echo "$stuck" +echo +echo "Stuck queued runs hold the deploy concurrency group and block later deploys and rollbacks." +echo "Cancel them with: gh run cancel --repo \"\$GITHUB_REPOSITORY\", then re-run the latest deploy." +exit 1 diff --git a/scripts/test-check-stuck-deploy-runs.sh b/scripts/test-check-stuck-deploy-runs.sh new file mode 100644 index 00000000..ee1fcc9a --- /dev/null +++ b/scripts/test-check-stuck-deploy-runs.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash +# test-check-stuck-deploy-runs.sh +# +# Self-test for check-stuck-deploy-runs.sh. Fixed clock, fixture run lists. +# Exits 0 when all cases pass; exits 1 on any failure. + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +CHECK="${SCRIPT_DIR}/check-stuck-deploy-runs.sh" + +# 2026-10-09T12:00:00Z +NOW=1791547200 + +pass=0 +fail=0 + +run_case() { + local label="$1" expected_exit="$2" input="$3" expected_text="${4:-}" + local out actual_exit=0 + out="$(printf '%s' "$input" | NOW_EPOCH="$NOW" bash "$CHECK" 2>&1)" || actual_exit=$? + if [[ "$actual_exit" -ne "$expected_exit" ]]; then + echo "FAIL: $label (expected exit $expected_exit, got $actual_exit)" + (( fail++ )) || true + return + fi + if [[ -n "$expected_text" && "$out" != *"$expected_text"* ]]; then + echo "FAIL: $label (output missing '$expected_text')" + (( fail++ )) || true + return + fi + echo "PASS: $label" + (( pass++ )) || true +} + +run() { # id status createdAt + printf '{"databaseId":%s,"workflowName":"Deploy to GCP Cloud Run","status":"%s","createdAt":"%s","url":"https://example.invalid/runs/%s","headBranch":"main"}' "$1" "$2" "$3" "$1" +} + +old="2026-10-09T08:00:00Z" # 240 min before NOW +fresh="2026-10-09T11:30:00Z" # 30 min before NOW +edge="2026-10-09T10:00:00Z" # exactly 120 min before NOW + +run_case "empty list is clean" 0 "[]" "No deploy run" +run_case "old queued run is stuck" 1 "[$(run 1 queued "$old")]" "run 1 (main) has been queued for 240 min" +run_case "old pending run is stuck" 1 "[$(run 2 pending "$old")]" "has been pending" +run_case "fresh queued run is fine" 0 "[$(run 3 queued "$fresh")]" +run_case "run exactly at the threshold is stuck" 1 "[$(run 4 queued "$edge")]" +run_case "old waiting run (environment approval) is ignored" 0 "[$(run 5 waiting "$old")]" +run_case "old in_progress run is ignored" 0 "[$(run 6 in_progress "$old")]" +run_case "one stuck among fine runs is reported" 1 "[$(run 7 queued "$fresh"),$(run 8 queued "$old"),$(run 9 waiting "$old")]" "run 8" +run_case "not JSON is a bad-input error" 2 "this is not json" +run_case "JSON object instead of array is a bad-input error" 2 '{"a":1}' + +out_exit=0 +STUCK_AFTER_MINUTES=abc bash "$CHECK" /dev/null 2>&1 || out_exit=$? +if [[ "$out_exit" -eq 2 ]]; then echo "PASS: non-numeric threshold is rejected"; (( pass++ )) || true +else echo "FAIL: non-numeric threshold (got exit $out_exit)"; (( fail++ )) || true; fi + +echo "$pass passed, $fail failed" +[[ "$fail" -eq 0 ]] From d1b45f55ac417f7774b577a9a6e14bf7790cab4e Mon Sep 17 00:00:00 2001 From: Cristian Magherusan-Stanciu Date: Fri, 9 Oct 2026 14:09:06 +0200 Subject: [PATCH 2/2] fix(ci): report waiting deploy runs in the queue watchdog - The AWS dev orphan in #708 (run 37447887010) held its concurrency group in the waiting state, not queued, and dev has no approval gate, so ignoring waiting would have missed one of the three stuck runs. - waiting is also the state of a run awaiting a required environment approval, so it alerts only after WAITING_AFTER_MINUTES (default 720) and is labeled "waiting (approval or orphaned)"; queued and pending keep the 120 minute threshold. - Self-test: waiting runs older than the waiting threshold (and exactly at it) are reported, a waiting run does not use the short threshold, an invalid waiting threshold is rejected. Three of these fail against the previous script. --- .github/workflows/deploy-queue-watchdog.yml | 14 +++++--- scripts/check-stuck-deploy-runs.sh | 36 +++++++++++++-------- scripts/test-check-stuck-deploy-runs.sh | 12 ++++++- 3 files changed, 42 insertions(+), 20 deletions(-) diff --git a/.github/workflows/deploy-queue-watchdog.yml b/.github/workflows/deploy-queue-watchdog.yml index 3ebb85fe..a7670cf2 100644 --- a/.github/workflows/deploy-queue-watchdog.yml +++ b/.github/workflows/deploy-queue-watchdog.yml @@ -1,14 +1,17 @@ name: deploy-queue-watchdog # Alerts when a deploy or rollback run has sat in `queued` or `pending` for -# hours. The per-cloud deploy jobs share a job-level concurrency group +# hours, or in `waiting` for most of a day. The per-cloud deploy jobs share a +# job-level concurrency group # (`-tfstate-`, cancel-in-progress: false), so one run that never # starts holds the slot and every later deploy or rollback waits behind it # (issue #708: GCP and Azure dev deploys frozen from 2026-10-05). # # Alert-only on purpose: the job fails (red scheduled run, GitHub notifies) and -# lists the stuck runs; a human cancels them. `waiting` runs are not reported, -# they are waiting for an environment approval. The `test` job runs the +# lists the stuck runs; a human cancels them. `waiting` is how the AWS dev +# orphan in #708 showed up (no approval gate on dev), but it is also the state +# of a run awaiting a required environment approval, so it only alerts after +# 720 minutes and is labeled "approval or orphaned". The `test` job runs the # script's self-test when the script or this workflow changes. on: @@ -55,18 +58,19 @@ jobs: GH_TOKEN: ${{ github.token }} GH_REPO: ${{ github.repository }} STUCK_AFTER_MINUTES: "120" + WAITING_AFTER_MINUTES: "720" steps: - name: Checkout uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 with: persist-credentials: false - - name: Collect queued and pending runs of the deploy workflows + - name: Collect queued, pending and waiting runs of the deploy workflows run: | set -euo pipefail : > runs.jsonl for wf in deploy-gcp.yml deploy-azure.yml deploy-aws-lambda.yml deploy-aws-fargate.yml deploy-all.yml rollback.yml; do - for status in queued pending; do + for status in queued pending waiting; do gh run list --workflow "$wf" --status "$status" --limit 50 \ --json databaseId,workflowName,status,createdAt,url,headBranch >> runs.jsonl done diff --git a/scripts/check-stuck-deploy-runs.sh b/scripts/check-stuck-deploy-runs.sh index c63c4c38..5d3226d0 100644 --- a/scripts/check-stuck-deploy-runs.sh +++ b/scripts/check-stuck-deploy-runs.sh @@ -3,13 +3,16 @@ # # Reads a JSON array of workflow runs on stdin (the output of # `gh run list --json databaseId,workflowName,status,createdAt,url,headBranch`) -# and reports runs that have been queued or pending for longer than -# STUCK_AFTER_MINUTES (default 120). +# and reports runs that have been queued or pending for STUCK_AFTER_MINUTES +# (default 120) or longer, and runs that have been waiting for +# WAITING_AFTER_MINUTES (default 720) or longer. # # Why: a deploy run stuck in `queued` holds the job-level concurrency group # (`-tfstate-`, cancel-in-progress: false) and every later deploy -# or rollback waits behind it (issue #708). A `waiting` run is deliberately -# NOT reported: that is a run waiting for a required environment approval. +# or rollback waits behind it (issue #708). `waiting` is how an orphaned +# holder looked on AWS dev (no approval gate there), but it is also the state +# of a run legitimately waiting for a required environment approval, so it gets +# a much longer threshold and its own label. # # Exit codes: 0 nothing stuck, 1 at least one stuck run, 2 bad input. # NOW_EPOCH overrides the clock (used by the self-test). @@ -17,32 +20,37 @@ set -euo pipefail threshold_minutes="${STUCK_AFTER_MINUTES:-120}" +waiting_minutes="${WAITING_AFTER_MINUTES:-720}" now_epoch="${NOW_EPOCH:-$(date -u +%s)}" -if ! [[ "$threshold_minutes" =~ ^[0-9]+$ ]]; then - echo "STUCK_AFTER_MINUTES must be a non-negative integer, got '$threshold_minutes'" >&2 - exit 2 -fi +for pair in "STUCK_AFTER_MINUTES=$threshold_minutes" "WAITING_AFTER_MINUTES=$waiting_minutes"; do + if ! [[ "${pair#*=}" =~ ^[0-9]+$ ]]; then + echo "${pair%%=*} must be a non-negative integer, got '${pair#*=}'" >&2 + exit 2 + fi +done -stuck="$(jq -r --argjson now "$now_epoch" --argjson limit "$threshold_minutes" ' - map(select(.status == "queued" or .status == "pending")) +stuck="$(jq -r --argjson now "$now_epoch" --argjson limit "$threshold_minutes" --argjson wlimit "$waiting_minutes" ' + map(select(.status == "queued" or .status == "pending" or .status == "waiting")) | map(. + {age_min: (($now - (.createdAt | fromdateiso8601)) / 60 | floor)}) - | map(select(.age_min >= $limit)) + | map(select(.age_min >= (if .status == "waiting" then $wlimit else $limit end))) | sort_by(.createdAt) | .[] - | "\(.workflowName) run \(.databaseId) (\(.headBranch)) has been \(.status) for \(.age_min) min: \(.url)" + | (if .status == "waiting" then "waiting (approval or orphaned)" else .status end) as $label + | "\(.workflowName) run \(.databaseId) (\(.headBranch)) has been \($label) for \(.age_min) min: \(.url)" ' 2>/dev/null)" || { echo "input is not a JSON array of runs" >&2 exit 2 } if [[ -z "$stuck" ]]; then - echo "No deploy run has been queued or pending for ${threshold_minutes} min or more." + echo "No deploy run has been queued or pending for ${threshold_minutes} min or more, or waiting for ${waiting_minutes} min or more." exit 0 fi echo "$stuck" echo -echo "Stuck queued runs hold the deploy concurrency group and block later deploys and rollbacks." +echo "Stuck runs hold the deploy concurrency group and block later deploys and rollbacks." +echo "A waiting run may just be awaiting an environment approval; check the run page before cancelling." echo "Cancel them with: gh run cancel --repo \"\$GITHUB_REPOSITORY\", then re-run the latest deploy." exit 1 diff --git a/scripts/test-check-stuck-deploy-runs.sh b/scripts/test-check-stuck-deploy-runs.sh index ee1fcc9a..2b916c6a 100644 --- a/scripts/test-check-stuck-deploy-runs.sh +++ b/scripts/test-check-stuck-deploy-runs.sh @@ -40,15 +40,20 @@ run() { # id status createdAt old="2026-10-09T08:00:00Z" # 240 min before NOW fresh="2026-10-09T11:30:00Z" # 30 min before NOW edge="2026-10-09T10:00:00Z" # exactly 120 min before NOW +wedge="2026-10-09T00:00:00Z" # exactly 720 min before NOW +wold="2026-10-08T23:00:00Z" # 780 min before NOW run_case "empty list is clean" 0 "[]" "No deploy run" run_case "old queued run is stuck" 1 "[$(run 1 queued "$old")]" "run 1 (main) has been queued for 240 min" run_case "old pending run is stuck" 1 "[$(run 2 pending "$old")]" "has been pending" run_case "fresh queued run is fine" 0 "[$(run 3 queued "$fresh")]" run_case "run exactly at the threshold is stuck" 1 "[$(run 4 queued "$edge")]" -run_case "old waiting run (environment approval) is ignored" 0 "[$(run 5 waiting "$old")]" +run_case "waiting run under the waiting threshold is fine" 0 "[$(run 5 waiting "$old")]" +run_case "waiting run past the waiting threshold is reported" 1 "[$(run 10 waiting "$wold")]" "waiting (approval or orphaned) for 780 min" +run_case "waiting run exactly at the waiting threshold is reported" 1 "[$(run 11 waiting "$wedge")]" run_case "old in_progress run is ignored" 0 "[$(run 6 in_progress "$old")]" run_case "one stuck among fine runs is reported" 1 "[$(run 7 queued "$fresh"),$(run 8 queued "$old"),$(run 9 waiting "$old")]" "run 8" +run_case "a waiting run does not use the short threshold" 0 "[$(run 12 waiting "$edge")]" run_case "not JSON is a bad-input error" 2 "this is not json" run_case "JSON object instead of array is a bad-input error" 2 '{"a":1}' @@ -57,5 +62,10 @@ STUCK_AFTER_MINUTES=abc bash "$CHECK" /dev/null 2>&1 || out_exit=$? if [[ "$out_exit" -eq 2 ]]; then echo "PASS: non-numeric threshold is rejected"; (( pass++ )) || true else echo "FAIL: non-numeric threshold (got exit $out_exit)"; (( fail++ )) || true; fi +wexit=0 +WAITING_AFTER_MINUTES=-5 bash "$CHECK" /dev/null 2>&1 || wexit=$? +if [[ "$wexit" -eq 2 ]]; then echo "PASS: invalid waiting threshold is rejected"; (( pass++ )) || true +else echo "FAIL: invalid waiting threshold (got exit $wexit)"; (( fail++ )) || true; fi + echo "$pass passed, $fail failed" [[ "$fail" -eq 0 ]]