From 5bf48522a58a75d141adb5c2baef0df728afdbd6 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 15:55:04 +0000 Subject: [PATCH 1/2] Cancel failed merge-queue CI runs right away ci-ok needs every CI job, so a merge-queue entry whose CI fails in minute 2 (a semantic conflict that stops compilation, say) only reports failure after the slowest Gradle e2e leg ends ~45 minutes later. Until then the entry, and every entry stacked on it, holds the queue and keeps burning macOS and Windows runners on a run that can no longer pass. Add a merge_group-only watcher that polls the entry's CI run and cancels it on the first failed or timed-out job. No CI job sets continue-on-error, so such a run can't pass. ci-ok is if: always(), so it still runs on the cancelled run and fails in seconds; the queue then evicts the entry and rebuilds the ones behind it. A test pins both ci.yml properties the watcher relies on. The watcher is a separate workflow, not a required check, and never fails its own job. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01CtHdJ7b6ncYXfs8LSJMsPy --- .github/workflows/merge-queue-fail-fast.yml | 34 ++++++ scripts/merge-queue-fail-fast.py | 120 ++++++++++++++++++++ scripts/tests/test_merge_queue_fail_fast.py | 56 +++++++++ 3 files changed, 210 insertions(+) create mode 100644 .github/workflows/merge-queue-fail-fast.yml create mode 100644 scripts/merge-queue-fail-fast.py create mode 100644 scripts/tests/test_merge_queue_fail_fast.py diff --git a/.github/workflows/merge-queue-fail-fast.yml b/.github/workflows/merge-queue-fail-fast.yml new file mode 100644 index 000000000..c2640dd0b --- /dev/null +++ b/.github/workflows/merge-queue-fail-fast.yml @@ -0,0 +1,34 @@ +name: Merge queue fail-fast + +# Cancels a merge-queue entry's CI run the moment one of its jobs fails, so +# `ci-ok` (if: always()) reports failure in seconds instead of after the +# slowest e2e leg, and the queue evicts the entry and rebuilds the ones +# behind it ~40 minutes sooner. See scripts/merge-queue-fail-fast.py. +# Not a required check; it never fails. + +on: + merge_group: + types: [checks_requested] + +permissions: {} + +jobs: + watch: + runs-on: ubuntu-latest + timeout-minutes: 110 + permissions: + actions: write + contents: read + steps: + - name: Checkout + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + persist-credentials: false + sparse-checkout: scripts/merge-queue-fail-fast.py + sparse-checkout-cone-mode: false + + - name: Cancel the CI run on its first failed job + env: + GH_TOKEN: ${{ github.token }} + HEAD_SHA: ${{ github.event.merge_group.head_sha }} + run: python3 -B scripts/merge-queue-fail-fast.py diff --git a/scripts/merge-queue-fail-fast.py b/scripts/merge-queue-fail-fast.py new file mode 100644 index 000000000..0791acf16 --- /dev/null +++ b/scripts/merge-queue-fail-fast.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""Cancel a merge-queue entry's CI run as soon as one of its jobs fails. + +`ci-ok`, the queue's required check, `needs:` every CI job, so it only +reports once the slowest leg finishes: a compile error at minute 2 kept +the entry, and every entry built on top of it, in the queue for the +~45 minutes the Gradle e2e legs take, burning macOS and Windows runners +on a run that could no longer pass. + +No CI job sets `continue-on-error`, so one failed or timed-out job means +`ci-ok` will fail. Cancelling the run then changes only *when* it fails: +`ci-ok` is `if: always()`, so it still runs on the cancelled run, sees the +failed and cancelled `needs`, and reports failure within seconds, and the +queue evicts the entry and rebuilds the ones behind it. + +The script never fails its own job: any API error is a warning and the +watch carries on (or gives up), leaving the run to finish as before. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +import urllib.error +import urllib.request + +API = os.environ.get("GITHUB_API_URL", "https://api.github.com") +GATE = "ci-ok" +# Job conclusions that make `ci-ok` fail. `cancelled` is left out: it +# means someone (the queue, a janitor, a person) is already cancelling. +FATAL = {"failure", "timed_out"} + + +def fatal_jobs(jobs): + """Names of the completed jobs whose conclusion dooms `ci-ok`.""" + return sorted( + j["name"] + for j in jobs + if j.get("status") == "completed" and j.get("conclusion") in FATAL and j.get("name") != GATE + ) + + +def request(method, path, token): + req = urllib.request.Request( + f"{API}/{path}", + method=method, + headers={ + "Accept": "application/vnd.github+json", + "Authorization": f"Bearer {token}", + "X-GitHub-Api-Version": "2022-11-28", + }, + ) + with urllib.request.urlopen(req, timeout=30) as resp: + body = resp.read() + return json.loads(body) if body else None + + +def find_run(repo, sha, token): + runs = request( + "GET", f"repos/{repo}/actions/workflows/ci.yml/runs?event=merge_group&head_sha={sha}&per_page=10", token + )["workflow_runs"] + return max(runs, key=lambda r: r["id"]) if runs else None + + +def list_jobs(repo, run_id, token): + jobs, page = [], 1 + while True: + batch = request("GET", f"repos/{repo}/actions/runs/{run_id}/jobs?filter=latest&per_page=100&page={page}", token)["jobs"] + jobs += batch + if len(batch) < 100: + return jobs + page += 1 + + +def summary(text): + print(text) + path = os.environ.get("GITHUB_STEP_SUMMARY") + if path: + with open(path, "a", encoding="utf-8") as f: + f.write(text + "\n") + + +def main(): + repo = os.environ["GITHUB_REPOSITORY"] + sha = os.environ["HEAD_SHA"] + token = os.environ["GH_TOKEN"] + interval = int(os.environ.get("POLL_SECONDS", "30")) + deadline = time.monotonic() + int(os.environ.get("WATCH_MINUTES", "100")) * 60 + run = None + while time.monotonic() < deadline: + try: + if run is None: + run = find_run(repo, sha, token) + if run is None: + time.sleep(interval) + continue + print(f"watching CI run {run['html_url']}") + if request("GET", f"repos/{repo}/actions/runs/{run['id']}", token)["status"] == "completed": + print("CI run completed; nothing to do") + return 0 + failed = fatal_jobs(list_jobs(repo, run["id"], token)) + if failed: + request("POST", f"repos/{repo}/actions/runs/{run['id']}/cancel", token) + summary( + f"Cancelled CI run {run['html_url']}: `ci-ok` cannot pass after these jobs failed:\n" + + "".join(f"- {name}\n" for name in failed) + ) + return 0 + except (urllib.error.URLError, OSError, KeyError, ValueError) as e: + # 409: the run is already finishing or being cancelled. + print(f"::warning title=merge-queue fail-fast::{e}") + time.sleep(interval) + print("::warning title=merge-queue fail-fast::watch window ended before the CI run finished") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/tests/test_merge_queue_fail_fast.py b/scripts/tests/test_merge_queue_fail_fast.py new file mode 100644 index 000000000..afc5590e6 --- /dev/null +++ b/scripts/tests/test_merge_queue_fail_fast.py @@ -0,0 +1,56 @@ +"""Tests for scripts/merge-queue-fail-fast.py.""" + +import importlib.util +import re +import unittest +from pathlib import Path + +ROOT = Path(__file__).parents[2] +CI = ROOT / ".github" / "workflows" / "ci.yml" + + +def load_script(): + spec = importlib.util.spec_from_file_location("fail_fast", ROOT / "scripts" / "merge-queue-fail-fast.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +ff = load_script() + + +def job(name, status="completed", conclusion="success"): + return {"name": name, "status": status, "conclusion": conclusion} + + +class FatalJobs(unittest.TestCase): + def test_all_green_or_running_is_not_fatal(self): + jobs = [job("clippy"), job("e2e (ubuntu)", "in_progress", None), job("e2e-full", conclusion="skipped")] + self.assertEqual(ff.fatal_jobs(jobs), []) + + def test_failed_and_timed_out_jobs_are_fatal(self): + jobs = [job("coverage", conclusion="failure"), job("test (windows-latest)", conclusion="timed_out"), job("clippy")] + self.assertEqual(ff.fatal_jobs(jobs), ["coverage", "test (windows-latest)"]) + + def test_cancelled_jobs_and_the_gate_itself_are_ignored(self): + jobs = [job("e2e (ubuntu)", conclusion="cancelled"), job("ci-ok", conclusion="failure")] + self.assertEqual(ff.fatal_jobs(jobs), []) + + +class CiWorkflowContract(unittest.TestCase): + """Cancelling on a failed job is safe only while ci.yml keeps these.""" + + text = CI.read_text(encoding="utf-8") + + def test_no_job_tolerates_failure(self): + # A job-level continue-on-error would let ci-ok pass despite a failed + # job, and this script would then cancel a run that could still land. + self.assertIsNone(re.search(r"^\s*continue-on-error:", self.text, re.M)) + + def test_gate_runs_on_a_cancelled_run(self): + gate = self.text[self.text.index("\n ci-ok:"):] + self.assertRegex(gate, r"\n if: always\(\)\n") + + +if __name__ == "__main__": + unittest.main() From 66aa8642ad6b2ac27fdb705aa1f4df7c4c334668 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 8 Oct 2026 17:22:06 +0000 Subject: [PATCH 2/2] Run the merge-queue canceller from the base branch, not the queue head The watch job holds an actions:write token. Checking out the merge-group head meant a queued PR could edit scripts/merge-queue-fail-fast.py and run arbitrary Python with that token before it ever lands. Check the script out at merge_group.base_sha instead so changing it requires the change to have already merged, and skip cleanly while the script is not on base yet. Co-Authored-By: Claude --- .github/workflows/merge-queue-fail-fast.yml | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/.github/workflows/merge-queue-fail-fast.yml b/.github/workflows/merge-queue-fail-fast.yml index c2640dd0b..1625c3006 100644 --- a/.github/workflows/merge-queue-fail-fast.yml +++ b/.github/workflows/merge-queue-fail-fast.yml @@ -22,7 +22,11 @@ jobs: steps: - name: Checkout uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + # Run the canceller as it is on the base branch, never the queued + # PRs' copy: this job holds an actions:write token, so executing + # merge-group-head bytes would hand that token to any queued PR. with: + ref: ${{ github.event.merge_group.base_sha }} persist-credentials: false sparse-checkout: scripts/merge-queue-fail-fast.py sparse-checkout-cone-mode: false @@ -31,4 +35,10 @@ jobs: env: GH_TOKEN: ${{ github.token }} HEAD_SHA: ${{ github.event.merge_group.head_sha }} - run: python3 -B scripts/merge-queue-fail-fast.py + # Until this workflow lands, the base branch has no canceller to run. + run: | + if [ ! -f scripts/merge-queue-fail-fast.py ]; then + echo "scripts/merge-queue-fail-fast.py is not on the base branch yet; nothing to watch." + exit 0 + fi + python3 -B scripts/merge-queue-fail-fast.py