From 724d3702a1d7d3d834b1f954088acaadd947e664 Mon Sep 17 00:00:00 2001 From: woksin Date: Thu, 1 Oct 2026 16:08:47 +0200 Subject: [PATCH 1/5] Improve work-record caller filtering and job-level Actions reporting --- .github/scripts/actions-usage-report.py | 214 ++++++++++++++++++ .github/scripts/bootstrap-common-workflows.sh | 7 +- .../tests/actions-usage-report.test.py | 120 ++++++++++ .../tests/verify-release-intent.spec.mjs | 11 + .github/workflows/actions-usage-report.yml | 70 +----- .../workflows/verify-actions-usage-report.yml | 36 +++ 6 files changed, 396 insertions(+), 62 deletions(-) create mode 100644 .github/scripts/actions-usage-report.py create mode 100644 .github/scripts/tests/actions-usage-report.test.py create mode 100644 .github/workflows/verify-actions-usage-report.yml diff --git a/.github/scripts/actions-usage-report.py b/.github/scripts/actions-usage-report.py new file mode 100644 index 0000000..e5269f7 --- /dev/null +++ b/.github/scripts/actions-usage-report.py @@ -0,0 +1,214 @@ +#!/usr/bin/env python3 +# Copyright (c) Cratis. All rights reserved. +# Licensed under the MIT license. See LICENSE file in the project root for full license information. +"""Report job execution, estimated hosted billing, and scale-set queue pressure.""" +import datetime as dt +import json +import math +import os +import subprocess +from collections import defaultdict + + +class APIError(Exception): + pass + + +def api(endpoint): + try: + result = subprocess.run(["gh", "api", endpoint], capture_output=True, text=True, timeout=60) + if result.returncode: + raise APIError("GitHub API request failed") + return json.loads(result.stdout) + except (subprocess.TimeoutExpired, ValueError) as error: + raise APIError("GitHub API request timed out or returned invalid JSON") from error + + +def pages(get, endpoint, key=None): + separator = "&" if "?" in endpoint else "?" + for page in range(1, 1001): + response = get(f"{endpoint}{separator}per_page=100&page={page}") + if key == "workflow_runs" and response.get("total_count", 0) > 1000: + raise APIError("Filtered run search exceeds GitHub's 1000-result limit") + items = response[key] if key else response + if not isinstance(items, list): + raise APIError("Invalid API listing") + yield from items + if len(items) < 100: + return + raise APIError("API pagination limit exceeded") + + +def timestamp(value): + if not value: + return None + try: + return dt.datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError: + return None + + +def collect(get, owner, since, until): + repos = [repo for repo in pages(get, f"orgs/{owner}/repos") if not repo.get("archived")] + # A token-filtered repository list alone cannot prove complete private coverage. + private_total = None + errors = [] + try: + private_total = get(f"orgs/{owner}").get("total_private_repos") + except APIError: + errors.append("Organization private-repository count unavailable; private coverage cannot be confirmed.") + runs, jobs, seen, seen_runs = [], [], set(), set() + for repo in repos: + full = repo["full_name"] + visibility = repo.get("visibility", "private" if repo.get("private") else "public") + # Filtered run searches have a 1000-result cap. Daily windows keep busy + # repositories below it; a window over the cap is explicitly incomplete. + cursor = since + while cursor < until: + window_end = min(cursor + dt.timedelta(days=1), until) + try: + start_query = cursor.strftime("%Y-%m-%dT%H:%M:%SZ") + end_query = window_end.strftime("%Y-%m-%dT%H:%M:%SZ") + endpoint = f"repos/{full}/actions/runs?created={start_query}..{end_query}" + for run in pages(get, endpoint, "workflow_runs"): + run_identity = (full, run["id"]) + if run_identity in seen_runs: + continue # GitHub's range endpoints are inclusive. + seen_runs.add(run_identity) + runs.append((full, visibility, run)) + try: + # Include rerun attempts rather than silently dropping earlier jobs. + for job in pages(get, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): + identity = (full, job["id"]) + if identity not in seen: + seen.add(identity) + jobs.append((full, visibility, run.get("name") or "Unnamed workflow", job)) + except APIError: + errors.append(f"Job inventory incomplete for {full}, run {run['id']}.") + except APIError: + errors.append(f"Run inventory incomplete for {full}, {cursor.date()} to {window_end.date()}.") + cursor = window_end + return repos, private_total, runs, jobs, errors + + +def os_multiplier(labels): + labels = [label.lower() for label in labels] + if any(label.startswith(("macos", "mac-")) for label in labels): + return 10 + if any(label.startswith("windows") for label in labels): + return 2 + if any(label.startswith("ubuntu") or label == "linux" for label in labels): + return 1 + return None + + +def percentile95(values): + return sorted(values)[math.ceil(len(values) * .95) - 1] if values else None + + +def render(repos, private_total, runs, jobs, errors, since, until): + private_visible = sum(repo.get("visibility") == "private" or repo.get("private", False) for repo in repos) + private = defaultdict(lambda: [0, 0.0, 0.0, 0]) + workflows = defaultdict(lambda: [0, 0.0, 0]) + arc_queue, arc_jobs, arc_short, queue_unknown, duration_unknown = [], 0, 0, 0, 0 + unknown_runner = 0 + hosted_private = 0 + for full, visibility, workflow, job in jobs: + start, end = timestamp(job.get("started_at")), timestamp(job.get("completed_at")) + if start is None or end is None or end < start: + # In-progress/never-started jobs have no final execution or billing verdict. + duration_unknown += 1 + continue + if not since <= start < until: + continue + seconds = (end - start).total_seconds() + minutes = seconds / 60 + row = workflows[f"{full} / {workflow}"] + row[0] += 1 + row[1] += minutes + row[2] += seconds < 60 + hosted = job.get("runner_group_name") == "GitHub Actions" + labels = job.get("labels") or [] + arc = not hosted and "cratis-arc" in labels + if arc: + arc_jobs += 1 + arc_short += seconds < 60 + created = timestamp(job.get("created_at")) + # Do not substitute run creation: that includes dependency waits, not just job queueing. + if created and created <= start: + arc_queue.append((start - created).total_seconds() / 60) + else: + queue_unknown += 1 + if visibility == "private": + p = private[full] + p[0] += 1 + multiplier = os_multiplier(labels) if hosted else None + if hosted and multiplier is not None: + billed = math.ceil(seconds / 60) * multiplier + p[1] += billed + hosted_private += billed + elif hosted or not job.get("runner_group_name"): + p[3] += 1 + unknown_runner += 1 + else: + p[2] += minutes + lines = [f"Job-level Actions usage from {since.isoformat()} to {until.isoformat()} (UTC).", "", + "Hosted private minutes are estimates: each completed job is rounded up to a minute, " + "then multiplied by Linux 1×, Windows 2× or macOS 10×. Public hosted jobs are free. " + "Self-hosted execution is not billed by GitHub; this is not an invoice. " + "The inventory covers runs created in the period and their completed jobs that started in the period.", "", + f"Visible non-archived repositories: **{len(repos)}**, private: **{private_visible}**. " + f"Inventoried runs: **{len(runs)}**, jobs (all attempts): **{len(jobs)}**.", ""] + if private_visible == 0: + lines.append("**Private repository coverage unavailable:** PAT_WORKFLOWS exposes no private repositories. " + "This is not evidence of zero private usage. Give the reporting token read access to private " + "repository metadata and Actions; on plans without private organization-secret support, use " + "a repository secret in Workflows.") + elif private_total is None: + lines.append("**Private coverage unconfirmed:** the token-filtered repository list may omit private " + "repositories. The organization private-repository count is unavailable.") + elif private_visible < private_total: + lines.append(f"**Partial private coverage:** {private_visible} non-archived private repositories visible " + f"out of {private_total} organization private repositories (including archived repositories). " + "Check PAT_WORKFLOWS repository access; missing usage is not zero.") + else: + lines.append("Private repository visibility matches the organization private-repository count.") + lines.extend(["", "### Private repositories", "", + "| Repository | completed jobs | estimated hosted weighted minutes | self-hosted execution minutes | unclassified jobs |", + "|---|---:|---:|---:|---:|"]) + for repo in sorted((r["full_name"] for r in repos if r.get("visibility") == "private" or r.get("private")), + key=lambda name: -private[name][1]): + n, billed, self_hosted, unknown = private[repo] + lines.append(f"| {repo} | {n} | {billed:,.0f} | {self_hosted:,.1f} | {unknown} |") + if not private_visible: + lines.append("| Not observable with PAT_WORKFLOWS | — | unknown | unknown | — |") + if hosted_private > 300: + lines.extend(["", f"**Alert: observed private hosted usage is {hosted_private:,.0f} weighted minutes, " + "above the 300-minute weekly guardrail.**"]) + lines.extend(["", "### cratis-arc queue pressure", "", + f"Completed jobs: **{arc_jobs}**; execution under one minute: **{arc_short}**."]) + p95 = percentile95(arc_queue) + lines.append(f"Job queue p95: **{p95:.1f} minutes** ({len(arc_queue)} samples)." if p95 is not None + else "Job queue p95: **unavailable** (no valid job creation/start timestamps).") + lines.append(f"Missing queue timestamps: **{queue_unknown}**. Run creation is not substituted for job creation.") + lines.extend(["", "### Top 20 workflows by job execution minutes", "", + "| Workflow | completed jobs | execution minutes | jobs under 1 minute |", + "|---|---:|---:|---:|"]) + for name, (n, minutes, short) in sorted(workflows.items(), key=lambda item: -item[1][1])[:20]: + lines.append(f"| {name.replace('|', '/')} | {n} | {minutes:,.1f} | {short} |") + lines.extend(["", f"Jobs with unavailable final duration: **{duration_unknown}**. " + f"Private jobs with unknown runner group or OS: **{unknown_runner}**; these are not counted as zero billing."]) + if errors: + lines.extend(["", "### Incomplete API coverage", "", "Missing data must not be treated as zero usage.", ""]) + lines.extend(f"- {error}" for error in sorted(set(errors))) + return "\n".join(lines) + "\n" + + +if __name__ == "__main__": + until = dt.datetime.now(dt.timezone.utc) + since = until - dt.timedelta(days=7) + try: + data = collect(api, os.environ["GITHUB_REPOSITORY_OWNER"], since, until) + print(render(*data, since, until), end="") + except APIError: + raise SystemExit("Unable to enumerate repositories; no usage report published.") diff --git a/.github/scripts/bootstrap-common-workflows.sh b/.github/scripts/bootstrap-common-workflows.sh index dfc1f7e..d18ca80 100644 --- a/.github/scripts/bootstrap-common-workflows.sh +++ b/.github/scripts/bootstrap-common-workflows.sh @@ -114,12 +114,17 @@ BOOTSTRAPPED_FILES[".github/workflows/auto-approve-publish-deployments.yml"]="bm # name: Verify No Work Records # on: # pull_request: +# paths: ["**.md", ".ai-work/**"] # push: # branches: ["main"] +# paths: ["**.md", ".ai-work/**"] +# concurrency: +# group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} +# cancel-in-progress: true # jobs: # verify: # uses: Cratis/Workflows/.github/workflows/verify-no-work-records.yml@main -BOOTSTRAPPED_FILES[".github/workflows/verify-no-work-records.yml"]="bmFtZTogVmVyaWZ5IE5vIFdvcmsgUmVjb3JkcwoKb246CiAgcHVsbF9yZXF1ZXN0OgogIHB1c2g6CiAgICBicmFuY2hlczogWyJtYWluIl0KCmpvYnM6CiAgdmVyaWZ5OgogICAgdXNlczogQ3JhdGlzL1dvcmtmbG93cy8uZ2l0aHViL3dvcmtmbG93cy92ZXJpZnktbm8td29yay1yZWNvcmRzLnltbEBtYWluCg==" +BOOTSTRAPPED_FILES[".github/workflows/verify-no-work-records.yml"]="bmFtZTogVmVyaWZ5IE5vIFdvcmsgUmVjb3JkcwoKb246CiAgcHVsbF9yZXF1ZXN0OgogICAgcGF0aHM6IFsiKioubWQiLCAiLmFpLXdvcmsvKioiXQogIHB1c2g6CiAgICBicmFuY2hlczogWyJtYWluIl0KICAgIHBhdGhzOiBbIioqLm1kIiwgIi5haS13b3JrLyoqIl0KCmNvbmN1cnJlbmN5OgogIGdyb3VwOiAke3sgZ2l0aHViLndvcmtmbG93IH19LSR7eyBnaXRodWIuZXZlbnQucHVsbF9yZXF1ZXN0Lm51bWJlciB8fCBnaXRodWIucmVmIH19CiAgY2FuY2VsLWluLXByb2dyZXNzOiB0cnVlCgpqb2JzOgogIHZlcmlmeToKICAgIHVzZXM6IENyYXRpcy9Xb3JrZmxvd3MvLmdpdGh1Yi93b3JrZmxvd3MvdmVyaWZ5LW5vLXdvcmstcmVjb3Jkcy55bWxAbWFpbgo=" # verify-release-notes.yml — fails release-bound PRs whose description (published # verbatim as the release notes) breaks the release-note contract diff --git a/.github/scripts/tests/actions-usage-report.test.py b/.github/scripts/tests/actions-usage-report.test.py new file mode 100644 index 0000000..4d00d40 --- /dev/null +++ b/.github/scripts/tests/actions-usage-report.test.py @@ -0,0 +1,120 @@ +# Copyright (c) Cratis. All rights reserved. +# Licensed under the MIT license. See LICENSE file in the project root for full license information. +"""Offline usage-report fixtures; no token or live GitHub calls.""" +import datetime as dt +import importlib.util +from pathlib import Path +import unittest + +spec = importlib.util.spec_from_file_location("report", Path(__file__).parents[1] / "actions-usage-report.py") +report = importlib.util.module_from_spec(spec) +spec.loader.exec_module(report) +SINCE = dt.datetime(2026, 9, 24, tzinfo=dt.timezone.utc) +UNTIL = dt.datetime(2026, 10, 1, tzinfo=dt.timezone.utc) +PRIVATE = {"full_name": "Cratis/Private", "visibility": "private"} +PUBLIC = {"full_name": "Cratis/Public", "visibility": "public"} + + +def job(identity=1, group="GitHub Actions", labels=None, seconds=61, queue=120): + start = SINCE + dt.timedelta(hours=1) + return {"id": identity, "runner_group_name": group, "labels": labels or ["ubuntu-latest"], + "started_at": start.isoformat(), "completed_at": (start + dt.timedelta(seconds=seconds)).isoformat(), + "created_at": (start - dt.timedelta(seconds=queue)).isoformat()} + + +def render(jobs=(), repos=None, total=1, errors=()): + return report.render(repos if repos is not None else [PRIVATE], total, [], jobs, errors, SINCE, UNTIL) + + +class UsageReportTests(unittest.TestCase): + def test_rounding_and_os_multipliers_apply_per_private_job(self): + jobs = [("Cratis/Private", "private", "Build", job(i, labels=[os])) + for i, os in enumerate(["ubuntu-latest", "windows-latest", "macos-15"], 1)] + self.assertIn("| Cratis/Private | 3 | 26 | 0.0 | 0 |", render(jobs)) + + def test_public_hosted_jobs_are_not_billed(self): + output = render([("Cratis/Public", "public", "Build", job())], repos=[PUBLIC], total=0) + self.assertIn("Private repository coverage unavailable", output) + self.assertIn("This is not evidence of zero private usage", output) + self.assertIn("Not observable with PAT_WORKFLOWS", output) + self.assertNotIn("billed wall-clock", output) + + def test_scale_set_execution_short_jobs_and_queue_percentile(self): + jobs = [("Cratis/Private", "private", "Build", job(i, "Custom runners", ["cratis-arc"], 30, i * 60)) + for i in range(1, 21)] + output = render(jobs) + self.assertIn("| Cratis/Private | 20 | 0 | 10.0 | 0 |", output) + self.assertIn("execution under one minute: **20**", output) + self.assertIn("Job queue p95: **19.0 minutes** (20 samples)", output) + + def test_unknown_runner_group_is_not_silently_self_hosted(self): + output = render([("Cratis/Private", "private", "Build", job(group=None))]) + self.assertIn("| Cratis/Private | 1 | 0 | 0.0 | 1 |", output) + self.assertIn("not counted as zero billing", output) + + def test_unknown_hosted_os_is_unclassified(self): + output = render([("Cratis/Private", "private", "Build", job(labels=["unrecognized"]))]) + self.assertIn("| Cratis/Private | 1 | 0 | 0.0 | 1 |", output) + + def test_unfinished_jobs_are_not_counted_as_final_billing(self): + unfinished = job() + unfinished["completed_at"] = None + self.assertIn("Jobs with unavailable final duration: **1**", render([("Cratis/Private", "private", "Build", unfinished)])) + + def test_missing_queue_timestamp_does_not_use_run_creation(self): + data = job(group="Custom", labels=["cratis-arc"]) + data.pop("created_at") + output = render([("Cratis/Private", "private", "Build", data)]) + self.assertIn("Job queue p95: **unavailable**", output) + self.assertIn("Missing queue timestamps: **1**", output) + + def test_coverage_and_api_failures_are_explicit(self): + self.assertIn("Partial private coverage", render(total=2)) + self.assertIn("Private coverage unconfirmed", render(total=None)) + self.assertIn("Incomplete API coverage", render(errors=["Run inventory incomplete for Cratis/Private."])) + + def test_weekly_hosted_guardrail(self): + data = job(labels=["macos-latest"], seconds=1801) + output = render([("Cratis/Private", "private", "Build", data)]) + self.assertIn("Alert: observed private hosted usage is 310", output) + + def test_collect_includes_all_attempts_and_deduplicates_ids(self): + calls = [] + def get(endpoint): + calls.append(endpoint) + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if "/jobs?" in endpoint: + return {"jobs": [job(), job(), job(2)]} + return {"workflow_runs": [{"id": 42, "name": "Build"}]} + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(data[2]), 1) + self.assertEqual(len(data[3]), 2) + self.assertTrue(any("jobs?filter=all&per_page=100&page=1" in call for call in calls)) + + def test_collection_failure_is_reported_not_zero(self): + def get(endpoint): + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + raise report.APIError("denied") + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(data[4]), 8) + self.assertTrue(any("Run inventory incomplete for Cratis/Private" in error for error in data[4])) + + def test_filtered_search_cap_is_not_silent_truncation(self): + with self.assertRaises(report.APIError): + list(report.pages(lambda _: {"total_count": 1001, "workflow_runs": []}, "runs", "workflow_runs")) + + def test_pagination(self): + calls = [] + def get(endpoint): + calls.append(endpoint) + return list(range(100)) if endpoint.endswith("page=1") else [100] + self.assertEqual(len(list(report.pages(get, "fixture"))), 101) + self.assertEqual(len(calls), 2) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/scripts/tests/verify-release-intent.spec.mjs b/.github/scripts/tests/verify-release-intent.spec.mjs index f7a5887..95b8efb 100644 --- a/.github/scripts/tests/verify-release-intent.spec.mjs +++ b/.github/scripts/tests/verify-release-intent.spec.mjs @@ -167,6 +167,17 @@ test("the normalizer never turns a run red: a token that cannot write, a failed // The bootstrap installs the release-intent caller only where a repository releases. Its decision function is run // with bash and a stub `gh` that serves workflow blobs from the test. const BOOTSTRAP = readFileSync(".github/scripts/bootstrap-common-workflows.sh", "utf8"); + +test("the bootstrapped work-record caller filters markdown and local work records and cancels superseded runs", () => { + const encoded = /BOOTSTRAPPED_FILES\["\.github\/workflows\/verify-no-work-records\.yml"\]="([^"]+)"/.exec(BOOTSTRAP)[1]; + const caller = Buffer.from(encoded, "base64").toString("utf8"); + assert.match(caller, /pull_request:\n paths: \["\*\*\.md", "\.ai-work\/\*\*"\]/); + assert.match(caller, /push:\n branches: \["main"\]\n paths: \["\*\*\.md", "\.ai-work\/\*\*"\]/); + assert(caller.includes("group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}")); + assert(caller.includes("cancel-in-progress: true")); + assert(caller.includes("uses: Cratis/Workflows/.github/workflows/verify-no-work-records.yml@main")); + assert.match(readFileSync(".github/workflows/verify-no-work-records.yml", "utf8"), /timeout-minutes: 5/); +}); const releasesFunction = /^releases_with_release_action\(\) \{\n[\s\S]*?^\}$/m.exec(BOOTSTRAP)[0]; const blobs = mkdtempSync(join(tmpdir(), "bootstrap-blobs-")); writeFileSync(join(blobs, "gh"), `#!/usr/bin/env bash diff --git a/.github/workflows/actions-usage-report.yml b/.github/workflows/actions-usage-report.yml index e4a1778..3e824ab 100644 --- a/.github/workflows/actions-usage-report.yml +++ b/.github/workflows/actions-usage-report.yml @@ -1,11 +1,7 @@ name: Actions Usage Report -# Weekly guardrail against runaway Actions consumption. The 2026-08-25 runner -# starvation was preceded by weeks of silent queue saturation (a single private -# repository was burning ~20,000 billable minutes a month and a comment-triggered -# assistant workflow ran ~150 times a week doing nothing). This report makes that -# kind of regression visible within a week instead of after an outage. - +# Weekly guardrail against runaway hosted consumption and scale-set queue pressure. +# Job-level data separates self-hosted execution from estimated private billing. on: schedule: - cron: "31 5 * * 1" @@ -24,65 +20,17 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 20 steps: + - name: Check out the reporting script + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Build and publish the report env: GH_TOKEN: ${{ secrets.PAT_WORKFLOWS }} shell: bash run: | set -euo pipefail - since=$(date -u -d '7 days ago' +%Y-%m-%d) - tmp=$(mktemp) - - gh api "orgs/${GITHUB_REPOSITORY_OWNER}/repos?per_page=100" --paginate \ - --jq '.[] | select(.archived | not) | .full_name + " " + .visibility' > /tmp/repos.txt - - while read -r full vis; do - gh api "repos/${full}/actions/runs?created=>=${since}&per_page=100" --paginate \ - --jq ".workflow_runs[] | \"${full}|${vis}|\(.name)|\(.conclusion)|\(.run_started_at)|\(.updated_at)\"" \ - >> "$tmp" 2>/dev/null || true - done < /tmp/repos.txt - - python3 - "$tmp" > /tmp/report.md <<'PY' - import sys, datetime - from collections import defaultdict - rows = defaultdict(lambda: [0, 0.0, 0]) - priv = defaultdict(float) - total = cancelled = 0 - for line in open(sys.argv[1]): - p = line.strip().split('|') - if len(p) < 6 or p[4] == 'null': - continue - try: - s = datetime.datetime.fromisoformat(p[4].replace('Z', '+00:00')) - e = datetime.datetime.fromisoformat(p[5].replace('Z', '+00:00')) - except ValueError: - continue - mins = max((e - s).total_seconds() / 60, 0) - key = f"{p[0]} / {p[2]}" - rows[key][0] += 1 - rows[key][1] += mins - total += 1 - if p[3] == 'cancelled': - rows[key][2] += 1 - cancelled += 1 - if p[1] == 'private': - priv[p[0]] += mins - print(f"Total runs: **{total}**, cancelled: **{cancelled}** " - f"({cancelled * 100 // max(total, 1)}%)\n") - print("### Private repositories (billed wall-clock minutes)\n") - print("| Repository | minutes |\n|---|---|") - for k, v in sorted(priv.items(), key=lambda x: -x[1]): - print(f"| {k} | {v:,.0f} |") - print("\n### Top 20 workflows by wall-clock minutes\n") - print("| Workflow | runs | minutes | cancelled |\n|---|---|---|---|") - for k, (n, m, c) in sorted(rows.items(), key=lambda x: -x[1][1])[:20]: - print(f"| {k} | {n} | {m:,.0f} | {c} |") - PY - + report="${RUNNER_TEMP}/actions-usage-report.md" + python3 .github/scripts/actions-usage-report.py > "$report" title="Actions usage report ($(date -u +%Y-%m-%d))" - { - echo "Automated weekly report over the last 7 days (wall-clock run minutes; job-level billable minutes are higher where matrices fan out)." - echo - cat /tmp/report.md - } > /tmp/body.md - gh issue create --repo "${GITHUB_REPOSITORY}" --title "$title" --body-file /tmp/body.md + gh issue create --repo "${GITHUB_REPOSITORY}" --title "$title" --body-file "$report" diff --git a/.github/workflows/verify-actions-usage-report.yml b/.github/workflows/verify-actions-usage-report.yml new file mode 100644 index 0000000..72c4064 --- /dev/null +++ b/.github/workflows/verify-actions-usage-report.yml @@ -0,0 +1,36 @@ +# Copyright (c) Cratis. All rights reserved. +# Licensed under the MIT license. See LICENSE file in the project root for full license information. +name: Verify Actions Usage Report + +on: + pull_request: + paths: + - .github/workflows/actions-usage-report.yml + - .github/workflows/verify-actions-usage-report.yml + - .github/scripts/actions-usage-report.py + - .github/scripts/tests/actions-usage-report.test.py + push: + branches: [main] + paths: + - .github/workflows/actions-usage-report.yml + - .github/workflows/verify-actions-usage-report.yml + - .github/scripts/actions-usage-report.py + - .github/scripts/tests/actions-usage-report.test.py + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + verify: + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: Verify job accounting and incomplete coverage reporting offline + run: python3 -B .github/scripts/tests/actions-usage-report.test.py From a3a8f05d3735804ad6bcbae4b2f6ec65f6a7bec2 Mon Sep 17 00:00:00 2001 From: woksin Date: Thu, 1 Oct 2026 19:03:13 +0200 Subject: [PATCH 2/5] Bound Actions usage collection and exclude jobs that never ran --- .github/scripts/actions-usage-report.py | 204 ++++++++++++++---- .../tests/actions-usage-report.test.py | 200 ++++++++++++++++- .github/workflows/actions-usage-report.yml | 4 +- 3 files changed, 360 insertions(+), 48 deletions(-) diff --git a/.github/scripts/actions-usage-report.py b/.github/scripts/actions-usage-report.py index e5269f7..ca88345 100644 --- a/.github/scripts/actions-usage-report.py +++ b/.github/scripts/actions-usage-report.py @@ -2,22 +2,59 @@ # Copyright (c) Cratis. All rights reserved. # Licensed under the MIT license. See LICENSE file in the project root for full license information. """Report job execution, estimated hosted billing, and scale-set queue pressure.""" +import base64 import datetime as dt import json import math import os import subprocess -from collections import defaultdict +import threading +import time +from collections import Counter, defaultdict +from concurrent.futures import ThreadPoolExecutor +from urllib.parse import quote class APIError(Exception): pass +class CollectionStopped(APIError): + pass + + +class RequestBudget: + """Reserve each request (including pagination) atomically across workers.""" + def __init__(self, get, remaining, max_seconds): + self.get = get + self.remaining = max(0, remaining - 100) # Leave capacity to publish the issue. + self.deadline = time.monotonic() + max_seconds + self.lock = threading.Lock() + self.reason = None + + def __call__(self, endpoint): + with self.lock: + if time.monotonic() >= self.deadline: + self.reason = "Collection time budget exhausted" + if self.remaining <= 0: + self.reason = "API request budget exhausted (100 requests reserved for publication)" + if self.reason: + raise CollectionStopped(self.reason) + self.remaining -= 1 + try: + return self.get(endpoint) + except CollectionStopped as error: + with self.lock: + self.reason = str(error) + raise + + def api(endpoint): try: result = subprocess.run(["gh", "api", endpoint], capture_output=True, text=True, timeout=60) if result.returncode: + if "rate limit" in result.stderr.lower(): + raise CollectionStopped("GitHub API rate limit reached") raise APIError("GitHub API request failed") return json.loads(result.stdout) except (subprocess.TimeoutExpired, ValueError) as error: @@ -48,46 +85,98 @@ def timestamp(value): return None -def collect(get, owner, since, until): - repos = [repo for repo in pages(get, f"orgs/{owner}/repos") if not repo.get("archived")] - # A token-filtered repository list alone cannot prove complete private coverage. - private_total = None - errors = [] +def collect(get, owner, since, until, workers=8, max_seconds=480): + # Do not start a collection whose request budget cannot be established. try: - private_total = get(f"orgs/{owner}").get("total_private_repos") + remaining = get("rate_limit")["resources"]["core"]["remaining"] + except (APIError, KeyError, TypeError): + return [], None, [], [], ["API rate budget unavailable; collection not started."] + request = RequestBudget(get, remaining, max_seconds) + repos, private_total, errors = [], None, [] + try: + # Retain archived metadata for the visibility check, but never collect its usage. + repos = list(pages(request, f"orgs/{owner}/repos")) + private_total = request(f"orgs/{owner}").get("total_private_repos") + except CollectionStopped as error: + return repos, private_total, [], [], [str(error)] except APIError: - errors.append("Organization private-repository count unavailable; private coverage cannot be confirmed.") - runs, jobs, seen, seen_runs = [], [], set(), set() - for repo in repos: + errors.append("Repository visibility or organization private-repository count unavailable; coverage cannot be confirmed.") + + def inventory(repo): full = repo["full_name"] visibility = repo.get("visibility", "private" if repo.get("private") else "public") - # Filtered run searches have a 1000-result cap. Daily windows keep busy - # repositories below it; a window over the cap is explicitly incomplete. + runs, jobs, seen, seen_runs, failures = [], [], set(), set(), Counter() + definitions, trees = {}, {} cursor = since - while cursor < until: - window_end = min(cursor + dt.timedelta(days=1), until) - try: - start_query = cursor.strftime("%Y-%m-%dT%H:%M:%SZ") - end_query = window_end.strftime("%Y-%m-%dT%H:%M:%SZ") - endpoint = f"repos/{full}/actions/runs?created={start_query}..{end_query}" - for run in pages(get, endpoint, "workflow_runs"): - run_identity = (full, run["id"]) - if run_identity in seen_runs: - continue # GitHub's range endpoints are inclusive. - seen_runs.add(run_identity) - runs.append((full, visibility, run)) - try: - # Include rerun attempts rather than silently dropping earlier jobs. - for job in pages(get, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): - identity = (full, job["id"]) - if identity not in seen: - seen.add(identity) - jobs.append((full, visibility, run.get("name") or "Unnamed workflow", job)) - except APIError: - errors.append(f"Job inventory incomplete for {full}, run {run['id']}.") - except APIError: - errors.append(f"Run inventory incomplete for {full}, {cursor.date()} to {window_end.date()}.") - cursor = window_end + try: + while cursor < until: + window_end = min(cursor + dt.timedelta(days=1), until) + try: + start_query = cursor.strftime("%Y-%m-%dT%H:%M:%SZ") + end_query = window_end.strftime("%Y-%m-%dT%H:%M:%SZ") + endpoint = f"repos/{full}/actions/runs?created={start_query}..{end_query}" + for run in pages(request, endpoint, "workflow_runs"): + if run["id"] in seen_runs: + continue # GitHub's range endpoints are inclusive. + seen_runs.add(run["id"]) + runs.append((full, visibility, run)) + try: + if visibility != "private": + # Public ranking needs no jobs. Resolve the historical workflow + # blob via one tree per commit, then cache by content SHA: most + # commits do not change workflows, so no per-run contents call. + path, sha = run.get("path", "").split("@")[0], run.get("head_sha") + if sha not in trees: + trees[sha] = {} + if sha: + tree = request(f"repos/{full}/git/trees/{quote(sha)}?recursive=1") + if not tree.get("truncated"): + trees[sha] = {entry["path"]: entry["sha"] for entry in tree["tree"] + if entry["type"] == "blob"} + blob = trees[sha].get(path) + if not blob: + failures["Public runner discovery incomplete"] += 1 + continue + if blob not in definitions: + definitions[blob] = None + content = request(f"repos/{full}/git/blobs/{quote(blob)}") + definitions[blob] = "cratis-arc" in base64.b64decode(content["content"]).decode() + if definitions[blob] is None: + failures["Public runner discovery incomplete"] += 1 + continue + if not definitions[blob]: + continue + # Include rerun attempts rather than silently dropping earlier jobs. + for job in pages(request, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): + if visibility != "private" and "cratis-arc" not in (job.get("labels") or []): + continue + if job["id"] not in seen: + seen.add(job["id"]) + jobs.append((full, visibility, run.get("name") or "Unnamed workflow", job)) + except CollectionStopped: + raise + except (APIError, KeyError, ValueError, UnicodeError): + failures["Job or runner inventory incomplete"] += 1 + except CollectionStopped: + raise + except APIError: + failures["Run inventory incomplete"] += 1 + cursor = window_end + except CollectionStopped: + failures["Collection stopped before completing repository"] += 1 + return runs, jobs, [f"{kind} for {full}: {count} {'daily windows' if kind == 'Run inventory incomplete' else 'runs'}." + for kind, count in failures.items()] + + active = [repo for repo in repos if not repo.get("archived")] + active.sort(key=lambda repo: not (repo.get("private") or repo.get("visibility") == "private")) + runs, jobs = [], [] + with ThreadPoolExecutor(max_workers=workers) as pool: + for repo_runs, repo_jobs, repo_errors in pool.map(inventory, active): + runs.extend(repo_runs) + jobs.extend(repo_jobs) + errors.extend(repo_errors) + if request.reason: + errors.append(request.reason + "; remaining repositories/runs were not collected. Missing usage is unknown.") return repos, private_total, runs, jobs, errors @@ -108,12 +197,28 @@ def percentile95(values): def render(repos, private_total, runs, jobs, errors, since, until): private_visible = sum(repo.get("visibility") == "private" or repo.get("private", False) for repo in repos) + repos = [repo for repo in repos if not repo.get("archived")] + private_active = sum(repo.get("visibility") == "private" or repo.get("private", False) for repo in repos) private = defaultdict(lambda: [0, 0.0, 0.0, 0]) workflows = defaultdict(lambda: [0, 0.0, 0]) + public_workflows = defaultdict(lambda: [0, 0.0]) + for full, visibility, run in runs: + if visibility != "public" or run.get("status") != "completed": + continue + start, end = timestamp(run.get("run_started_at")), timestamp(run.get("updated_at")) + if start is None or end is None or end < start or not since <= start < until: + continue + row = public_workflows[f"{full} / {run.get('name') or 'Unnamed workflow'}"] + row[0] += 1 + row[1] += (end - start).total_seconds() / 60 arc_queue, arc_jobs, arc_short, queue_unknown, duration_unknown = [], 0, 0, 0, 0 unknown_runner = 0 hosted_private = 0 + excluded = 0 for full, visibility, workflow, job in jobs: + if job.get("conclusion") == "skipped" or not job.get("runner_name") or not job.get("runner_id"): + excluded += 1 + continue start, end = timestamp(job.get("started_at")), timestamp(job.get("completed_at")) if start is None or end is None or end < start: # In-progress/never-started jobs have no final execution or billing verdict. @@ -123,10 +228,11 @@ def render(repos, private_total, runs, jobs, errors, since, until): continue seconds = (end - start).total_seconds() minutes = seconds / 60 - row = workflows[f"{full} / {workflow}"] - row[0] += 1 - row[1] += minutes - row[2] += seconds < 60 + if visibility == "private": + row = workflows[f"{full} / {workflow}"] + row[0] += 1 + row[1] += minutes + row[2] += seconds < 60 hosted = job.get("runner_group_name") == "GitHub Actions" labels = job.get("labels") or [] arc = not hosted and "cratis-arc" in labels @@ -157,7 +263,7 @@ def render(repos, private_total, runs, jobs, errors, since, until): "then multiplied by Linux 1×, Windows 2× or macOS 10×. Public hosted jobs are free. " "Self-hosted execution is not billed by GitHub; this is not an invoice. " "The inventory covers runs created in the period and their completed jobs that started in the period.", "", - f"Visible non-archived repositories: **{len(repos)}**, private: **{private_visible}**. " + f"Visible non-archived repositories: **{len(repos)}**, private: **{private_active}**. " f"Inventoried runs: **{len(runs)}**, jobs (all attempts): **{len(jobs)}**.", ""] if private_visible == 0: lines.append("**Private repository coverage unavailable:** PAT_WORKFLOWS exposes no private repositories. " @@ -168,7 +274,7 @@ def render(repos, private_total, runs, jobs, errors, since, until): lines.append("**Private coverage unconfirmed:** the token-filtered repository list may omit private " "repositories. The organization private-repository count is unavailable.") elif private_visible < private_total: - lines.append(f"**Partial private coverage:** {private_visible} non-archived private repositories visible " + lines.append(f"**Partial private coverage:** {private_visible} private repositories visible (including archived) " f"out of {private_total} organization private repositories (including archived repositories). " "Check PAT_WORKFLOWS repository access; missing usage is not zero.") else: @@ -180,7 +286,7 @@ def render(repos, private_total, runs, jobs, errors, since, until): key=lambda name: -private[name][1]): n, billed, self_hosted, unknown = private[repo] lines.append(f"| {repo} | {n} | {billed:,.0f} | {self_hosted:,.1f} | {unknown} |") - if not private_visible: + if not private_active: lines.append("| Not observable with PAT_WORKFLOWS | — | unknown | unknown | — |") if hosted_private > 300: lines.extend(["", f"**Alert: observed private hosted usage is {hosted_private:,.0f} weighted minutes, " @@ -191,12 +297,20 @@ def render(repos, private_total, runs, jobs, errors, since, until): lines.append(f"Job queue p95: **{p95:.1f} minutes** ({len(arc_queue)} samples)." if p95 is not None else "Job queue p95: **unavailable** (no valid job creation/start timestamps).") lines.append(f"Missing queue timestamps: **{queue_unknown}**. Run creation is not substituted for job creation.") - lines.extend(["", "### Top 20 workflows by job execution minutes", "", + lines.extend(["", "### Top 20 private workflows by job execution minutes", "", "| Workflow | completed jobs | execution minutes | jobs under 1 minute |", "|---|---:|---:|---:|"]) for name, (n, minutes, short) in sorted(workflows.items(), key=lambda item: -item[1][1])[:20]: lines.append(f"| {name.replace('|', '/')} | {n} | {minutes:,.1f} | {short} |") - lines.extend(["", f"Jobs with unavailable final duration: **{duration_unknown}**. " + lines.extend(["", "### Top 20 public workflows by run elapsed minutes", "", + "Run-level elapsed time includes dependency waits and is not job execution or billing. " + "Public job inventory is limited to workflows with literal cratis-arc labels in their historical definitions; " + "indirect runner selection is not observable.", "", + "| Workflow | completed runs | elapsed minutes |", "|---|---:|---:|"]) + for name, (n, minutes) in sorted(public_workflows.items(), key=lambda item: -item[1][1])[:20]: + lines.append(f"| {name.replace('|', '/')} | {n} | {minutes:,.1f} |") + lines.extend(["", f"Skipped or never-assigned jobs excluded from execution and queue statistics: **{excluded}**.", + f"Jobs with unavailable final duration: **{duration_unknown}**. " f"Private jobs with unknown runner group or OS: **{unknown_runner}**; these are not counted as zero billing."]) if errors: lines.extend(["", "### Incomplete API coverage", "", "Missing data must not be treated as zero usage.", ""]) diff --git a/.github/scripts/tests/actions-usage-report.test.py b/.github/scripts/tests/actions-usage-report.test.py index 4d00d40..59a218b 100644 --- a/.github/scripts/tests/actions-usage-report.test.py +++ b/.github/scripts/tests/actions-usage-report.test.py @@ -1,9 +1,12 @@ # Copyright (c) Cratis. All rights reserved. # Licensed under the MIT license. See LICENSE file in the project root for full license information. """Offline usage-report fixtures; no token or live GitHub calls.""" +import base64 import datetime as dt import importlib.util from pathlib import Path +import threading +import time import unittest spec = importlib.util.spec_from_file_location("report", Path(__file__).parents[1] / "actions-usage-report.py") @@ -17,7 +20,7 @@ def job(identity=1, group="GitHub Actions", labels=None, seconds=61, queue=120): start = SINCE + dt.timedelta(hours=1) - return {"id": identity, "runner_group_name": group, "labels": labels or ["ubuntu-latest"], + return {"id": identity, "runner_id": identity, "runner_name": "runner", "runner_group_name": group, "labels": labels or ["ubuntu-latest"], "started_at": start.isoformat(), "completed_at": (start + dt.timedelta(seconds=seconds)).isoformat(), "created_at": (start - dt.timedelta(seconds=queue)).isoformat()} @@ -82,6 +85,8 @@ def test_collect_includes_all_attempts_and_deduplicates_ids(self): calls = [] def get(endpoint): calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} if endpoint == "orgs/Cratis": return {"total_private_repos": 1} if endpoint.startswith("orgs/Cratis/repos?"): @@ -96,13 +101,204 @@ def get(endpoint): def test_collection_failure_is_reported_not_zero(self): def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} if endpoint.startswith("orgs/Cratis/repos?"): return [PRIVATE] raise report.APIError("denied") data = report.collect(get, "Cratis", SINCE, UNTIL) - self.assertEqual(len(data[4]), 8) + self.assertEqual(len(data[4]), 1) + self.assertIn("7 daily windows", data[4][0]) self.assertTrue(any("Run inventory incomplete for Cratis/Private" in error for error in data[4])) + def test_skipped_and_never_assigned_jobs_do_not_distort_statistics(self): + skipped = job(1, None, ["cratis-arc"], seconds=-1, queue=0) + skipped.update(conclusion="skipped", runner_name=None, runner_id=None) + cancelled = job(2, "", ["cratis-arc"], seconds=0, queue=0) + cancelled.update(conclusion="cancelled", runner_name="", runner_id=0) + executed = job(3, "Custom", ["cratis-arc"], seconds=30, queue=120) + executed["conclusion"] = "cancelled" + output = render([("Cratis/Private", "private", "Build", data) + for data in (skipped, cancelled, executed)]) + self.assertIn("| Cratis/Private | 1 | 0 | 0.5 | 0 |", output) + self.assertIn("Completed jobs: **1**; execution under one minute: **1**", output) + self.assertIn("Job queue p95: **2.0 minutes** (1 samples)", output) + self.assertIn("statistics: **2**", output) + self.assertIn("Jobs with unavailable final duration: **0**", output) + self.assertIn("unknown runner group or OS: **0**", output) + + def test_archived_private_visibility_counts_but_usage_does_not(self): + archived = {"full_name": "Cratis/Archived", "private": True, "archived": True} + output = render(repos=[PRIVATE, archived], total=2) + self.assertNotIn("Partial private coverage", output) + self.assertIn("private: **1**", output) + self.assertNotIn("| Cratis/Archived |", output) + + def test_public_ranking_uses_run_elapsed_not_job_execution(self): + run = {"name": "Build", "status": "completed", "run_started_at": SINCE.isoformat(), + "updated_at": (SINCE + dt.timedelta(minutes=10)).isoformat()} + output = report.render([PUBLIC], 0, [("Cratis/Public", "public", run)], [], [], SINCE, UNTIL) + self.assertIn("| Cratis/Public / Build | 1 | 10.0 |", output) + self.assertIn("not job execution or billing", output) + + def test_public_job_inventory_only_for_arc_workflows_and_definition_is_cached(self): + calls = [] + def get(endpoint): + calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint == "orgs/Cratis": + return {"total_private_repos": 0} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PUBLIC] + if "/git/trees/" in endpoint: + return {"tree": [{"path": ".github/workflows/hosted.yml", "type": "blob", "sha": "hosted"}, + {"path": ".github/workflows/arc.yml", "type": "blob", "sha": "arc"}]} + if "/git/blobs/" in endpoint: + text = "runs-on: cratis-arc" if endpoint.endswith("/arc") else "runs-on: ubuntu-latest" + return {"content": base64.b64encode(text.encode()).decode()} + if "/jobs?" in endpoint: + return {"jobs": [job(group="Custom", labels=["cratis-arc"]), job(2)]} + return {"workflow_runs": [{"id": i, "path": path, "head_sha": "abc", "name": "Build"} + for i, path in ((1, ".github/workflows/hosted.yml"), + (2, ".github/workflows/arc.yml"), + (3, ".github/workflows/arc.yml"))]} + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(data[2]), 3) + self.assertEqual(len(data[3]), 1) + self.assertEqual(sum("/git/trees/" in call for call in calls), 1) + self.assertEqual(sum("/git/blobs/" in call for call in calls), 2) + self.assertEqual(sum("/jobs?" in call for call in calls), 2) + self.assertFalse(data[4]) + + def test_budget_exhaustion_is_explicit_and_a_fresh_collection_can_resume(self): + calls = [] + remaining = 105 + def get(endpoint): + calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": remaining}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + if "/jobs?" in endpoint: + return {"jobs": [job(1), job(2)]} + return {"workflow_runs": [{"id": 1, "name": "Build"}]} + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(calls[0], "rate_limit") + self.assertEqual(len(calls), 6) # Preflight plus exactly five reserved requests. + self.assertIn("API request budget exhausted", render(errors=data[4])) + self.assertEqual(len(data[3]), 2) # Successful data retained, not silently zeroed. + remaining = 5000 + self.assertFalse(report.collect(get, "Cratis", SINCE, UNTIL)[4]) + + def test_shared_budget_cannot_be_overspent_by_parallel_workers(self): + calls = [] + lock = threading.Lock() + def get(endpoint): + with lock: + calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 110}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [{"full_name": f"Cratis/P{i}", "private": True} for i in range(20)] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 20} + time.sleep(.001) + return {"workflow_runs": []} + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(calls), 11) + self.assertTrue(any("API request budget exhausted" in error for error in data[4])) + + def test_server_rate_limit_stops_further_requests(self): + calls = [] + def get(endpoint): + calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + raise report.CollectionStopped("GitHub API rate limit reached") + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(calls), 4) + self.assertTrue(any("rate limit reached" in error for error in data[4])) + + def test_rate_preflight_failure_does_not_start_collection(self): + calls = [] + def get(endpoint): + calls.append(endpoint) + raise report.APIError("denied") + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(calls, ["rate_limit"]) + self.assertIn("collection not started", data[4][0]) + + def test_time_budget_stops_with_partial_report(self): + def get(endpoint): + return {"resources": {"core": {"remaining": 5000}}} + data = report.collect(get, "Cratis", SINCE, UNTIL, max_seconds=0) + self.assertIn("Collection time budget exhausted", data[4][0]) + + def test_job_failures_are_collapsed_per_repository(self): + def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + if "/jobs?" in endpoint: + raise report.APIError("denied") + return {"workflow_runs": [{"id": i} for i in range(99)]} + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(data[4]), 1) + self.assertIn("99 runs", data[4][0]) + + def test_observed_org_volume_fits_budget_with_bounded_parallelism(self): + repos = [{"full_name": f"Cratis/R{i}", "visibility": "private" if i < 3 else "public"} + for i in range(20)] + lock = threading.Lock() + active = peak = calls = 0 + def get(endpoint): + nonlocal active, peak, calls + with lock: + active += 1 + peak = max(peak, active) + calls += 1 + try: + time.sleep(.0001) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return repos + if endpoint == "orgs/Cratis": + return {"total_private_repos": 3} + if "/git/trees/" in endpoint: + return {"tree": [{"path": ".github/workflows/build.yml", "type": "blob", "sha": "hosted"}]} + if "/git/blobs/" in endpoint: + return {"content": base64.b64encode(b"runs-on: ubuntu-latest").decode()} + if "/jobs?" in endpoint: + return {"jobs": []} + page = int(endpoint.rsplit("=", 1)[1]) + if page > 1: + return {"workflow_runs": []} + day = int(endpoint.split("created=2026-09-")[1][:2]) + return {"workflow_runs": [{"id": day * 100 + i, "path": ".github/workflows/build.yml", + "head_sha": "abc"} for i in range(100)]} + finally: + with lock: + active -= 1 + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(data[2]), 14000) + self.assertFalse(data[4]) + self.assertLess(calls, 5000) + self.assertGreater(peak, 1) + self.assertLessEqual(peak, 8) + def test_filtered_search_cap_is_not_silent_truncation(self): with self.assertRaises(report.APIError): list(report.pages(lambda _: {"total_count": 1001, "workflow_runs": []}, "runs", "workflow_runs")) diff --git a/.github/workflows/actions-usage-report.yml b/.github/workflows/actions-usage-report.yml index 3e824ab..3e77b32 100644 --- a/.github/workflows/actions-usage-report.yml +++ b/.github/workflows/actions-usage-report.yml @@ -18,7 +18,9 @@ concurrency: jobs: report: runs-on: ubuntu-latest - timeout-minutes: 20 + # Live no-issue collection measured 481s at the 480s collection bound. + # Allow one final 60s API call plus checkout/publication overhead. + timeout-minutes: 12 steps: - name: Check out the reporting script uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 From 28adb023bb4a51be55be946a06f60c5679c7c126 Mon Sep 17 00:00:00 2001 From: woksin Date: Thu, 1 Oct 2026 19:31:25 +0200 Subject: [PATCH 3/5] Fix usage report discovery and parallelize job collection --- .github/scripts/actions-usage-report.py | 86 +++++++++--- .../tests/actions-usage-report.test.py | 132 +++++++++++++++++- 2 files changed, 190 insertions(+), 28 deletions(-) diff --git a/.github/scripts/actions-usage-report.py b/.github/scripts/actions-usage-report.py index ca88345..43cbeda 100644 --- a/.github/scripts/actions-usage-report.py +++ b/.github/scripts/actions-usage-report.py @@ -23,6 +23,10 @@ class CollectionStopped(APIError): pass +class RunWindowTooLarge(APIError): + pass + + class RequestBudget: """Reserve each request (including pagination) atomically across workers.""" def __init__(self, get, remaining, max_seconds): @@ -66,7 +70,7 @@ def pages(get, endpoint, key=None): for page in range(1, 1001): response = get(f"{endpoint}{separator}per_page=100&page={page}") if key == "workflow_runs" and response.get("total_count", 0) > 1000: - raise APIError("Filtered run search exceeds GitHub's 1000-result limit") + raise RunWindowTooLarge("Filtered run search exceeds GitHub's 1000-result limit") items = response[key] if key else response if not isinstance(items, list): raise APIError("Invalid API listing") @@ -76,6 +80,21 @@ def pages(get, endpoint, key=None): raise APIError("API pagination limit exceeded") +def window_runs(get, full, start, end): + """Bisect capped searches; callers deduplicate inclusive range boundaries.""" + start_query = start.strftime("%Y-%m-%dT%H:%M:%SZ") + end_query = end.strftime("%Y-%m-%dT%H:%M:%SZ") + endpoint = f"repos/{full}/actions/runs?created={start_query}..{end_query}" + try: + yield from pages(get, endpoint, "workflow_runs") + except RunWindowTooLarge: + middle = (start + (end - start) / 2).replace(microsecond=0) + if middle <= start or middle >= end: + raise # Even a one-second window is capped: report the gap explicitly. + yield from window_runs(get, full, start, middle) + yield from window_runs(get, full, middle, end) + + def timestamp(value): if not value: return None @@ -89,33 +108,33 @@ def collect(get, owner, since, until, workers=8, max_seconds=480): # Do not start a collection whose request budget cannot be established. try: remaining = get("rate_limit")["resources"]["core"]["remaining"] - except (APIError, KeyError, TypeError): - return [], None, [], [], ["API rate budget unavailable; collection not started."] + except (APIError, KeyError, TypeError) as error: + raise APIError("API rate budget unavailable; collection not started.") from error request = RequestBudget(get, remaining, max_seconds) repos, private_total, errors = [], None, [] try: # Retain archived metadata for the visibility check, but never collect its usage. repos = list(pages(request, f"orgs/{owner}/repos")) + except APIError as error: + raise APIError("Unable to enumerate repositories; collection not started.") from error + try: private_total = request(f"orgs/{owner}").get("total_private_repos") except CollectionStopped as error: return repos, private_total, [], [], [str(error)] except APIError: - errors.append("Repository visibility or organization private-repository count unavailable; coverage cannot be confirmed.") + errors.append("Organization private-repository count unavailable; coverage cannot be confirmed.") def inventory(repo): full = repo["full_name"] visibility = repo.get("visibility", "private" if repo.get("private") else "public") - runs, jobs, seen, seen_runs, failures = [], [], set(), set(), Counter() + runs, candidates, seen_runs, failures = [], [], set(), Counter() definitions, trees = {}, {} cursor = since try: while cursor < until: window_end = min(cursor + dt.timedelta(days=1), until) try: - start_query = cursor.strftime("%Y-%m-%dT%H:%M:%SZ") - end_query = window_end.strftime("%Y-%m-%dT%H:%M:%SZ") - endpoint = f"repos/{full}/actions/runs?created={start_query}..{end_query}" - for run in pages(request, endpoint, "workflow_runs"): + for run in window_runs(request, full, cursor, window_end): if run["id"] in seen_runs: continue # GitHub's range endpoints are inclusive. seen_runs.add(run["id"]) @@ -126,6 +145,8 @@ def inventory(repo): # blob via one tree per commit, then cache by content SHA: most # commits do not change workflows, so no per-run contents call. path, sha = run.get("path", "").split("@")[0], run.get("head_sha") + if path.startswith("dynamic/"): + continue # GitHub-managed hosted workflows have no repository blob. if sha not in trees: trees[sha] = {} if sha: @@ -146,13 +167,7 @@ def inventory(repo): continue if not definitions[blob]: continue - # Include rerun attempts rather than silently dropping earlier jobs. - for job in pages(request, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): - if visibility != "private" and "cratis-arc" not in (job.get("labels") or []): - continue - if job["id"] not in seen: - seen.add(job["id"]) - jobs.append((full, visibility, run.get("name") or "Unnamed workflow", job)) + candidates.append((full, visibility, run)) except CollectionStopped: raise except (APIError, KeyError, ValueError, UnicodeError): @@ -164,17 +179,43 @@ def inventory(repo): cursor = window_end except CollectionStopped: failures["Collection stopped before completing repository"] += 1 - return runs, jobs, [f"{kind} for {full}: {count} {'daily windows' if kind == 'Run inventory incomplete' else 'runs'}." - for kind, count in failures.items()] + return runs, candidates, [f"{kind} for {full}: {count} {'daily windows' if kind == 'Run inventory incomplete' else 'runs'}." + for kind, count in failures.items()] + + def job_inventory(candidate): + full, visibility, run = candidate + found = [] + try: + # Each run is independent: a busy repository uses the entire bounded + # pool instead of serializing thousands of job requests on one worker. + for job in pages(request, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): + if visibility == "private" or "cratis-arc" in (job.get("labels") or []): + found.append((full, visibility, run.get("name") or "Unnamed workflow", job)) + except CollectionStopped: + return found, "Collection stopped before completing repository" + except (APIError, KeyError, ValueError, UnicodeError): + return found, "Job or runner inventory incomplete" + return found, None active = [repo for repo in repos if not repo.get("archived")] active.sort(key=lambda repo: not (repo.get("private") or repo.get("visibility") == "private")) - runs, jobs = [], [] + runs, jobs, candidates, seen_jobs, job_failures = [], [], [], set(), Counter() with ThreadPoolExecutor(max_workers=workers) as pool: - for repo_runs, repo_jobs, repo_errors in pool.map(inventory, active): + for repo_runs, repo_candidates, repo_errors in pool.map(inventory, active): runs.extend(repo_runs) - jobs.extend(repo_jobs) + candidates.extend(repo_candidates) errors.extend(repo_errors) + # Reuse the same pool and request budget for job pagination; at most + # `workers` API requests are in flight across both collection stages. + for candidate, (found, failure) in zip(candidates, pool.map(job_inventory, candidates)): + if failure: + job_failures[(candidate[0], failure)] += 1 + for entry in found: + key = (entry[0], entry[3]["id"]) + if key not in seen_jobs: + seen_jobs.add(key) + jobs.append(entry) + errors.extend(f"{kind} for {full}: {count} runs." for (full, kind), count in job_failures.items()) if request.reason: errors.append(request.reason + "; remaining repositories/runs were not collected. Missing usage is unknown.") return repos, private_total, runs, jobs, errors @@ -265,6 +306,9 @@ def render(repos, private_total, runs, jobs, errors, since, until): "The inventory covers runs created in the period and their completed jobs that started in the period.", "", f"Visible non-archived repositories: **{len(repos)}**, private: **{private_active}**. " f"Inventoried runs: **{len(runs)}**, jobs (all attempts): **{len(jobs)}**.", ""] + dynamic_runs = sum(visibility != "private" and run.get("path", "").startswith("dynamic/") + for _, visibility, run in runs) + lines.extend([f"GitHub-managed dynamic runs (not inspected): **{dynamic_runs}**.", ""]) if private_visible == 0: lines.append("**Private repository coverage unavailable:** PAT_WORKFLOWS exposes no private repositories. " "This is not evidence of zero private usage. Give the reporting token read access to private " diff --git a/.github/scripts/tests/actions-usage-report.test.py b/.github/scripts/tests/actions-usage-report.test.py index 59a218b..3e96b9c 100644 --- a/.github/scripts/tests/actions-usage-report.test.py +++ b/.github/scripts/tests/actions-usage-report.test.py @@ -175,7 +175,7 @@ def get(endpoint): def test_budget_exhaustion_is_explicit_and_a_fresh_collection_can_resume(self): calls = [] - remaining = 105 + remaining = 111 def get(endpoint): calls.append(endpoint) if endpoint == "rate_limit": @@ -186,10 +186,11 @@ def get(endpoint): return {"total_private_repos": 1} if "/jobs?" in endpoint: return {"jobs": [job(1), job(2)]} - return {"workflow_runs": [{"id": 1, "name": "Build"}]} + day = int(endpoint.split("created=2026-09-")[1][:2]) + return {"workflow_runs": [{"id": day, "name": "Build"}]} data = report.collect(get, "Cratis", SINCE, UNTIL) self.assertEqual(calls[0], "rate_limit") - self.assertEqual(len(calls), 6) # Preflight plus exactly five reserved requests. + self.assertEqual(len(calls), 12) # Preflight plus exactly eleven reserved requests. self.assertIn("API request budget exhausted", render(errors=data[4])) self.assertEqual(len(data[3]), 2) # Successful data retained, not silently zeroed. remaining = 5000 @@ -233,15 +234,15 @@ def test_rate_preflight_failure_does_not_start_collection(self): def get(endpoint): calls.append(endpoint) raise report.APIError("denied") - data = report.collect(get, "Cratis", SINCE, UNTIL) + with self.assertRaisesRegex(report.APIError, "collection not started"): + report.collect(get, "Cratis", SINCE, UNTIL) self.assertEqual(calls, ["rate_limit"]) - self.assertIn("collection not started", data[4][0]) def test_time_budget_stops_with_partial_report(self): def get(endpoint): return {"resources": {"core": {"remaining": 5000}}} - data = report.collect(get, "Cratis", SINCE, UNTIL, max_seconds=0) - self.assertIn("Collection time budget exhausted", data[4][0]) + with self.assertRaisesRegex(report.APIError, "Unable to enumerate repositories"): + report.collect(get, "Cratis", SINCE, UNTIL, max_seconds=0) def test_job_failures_are_collapsed_per_repository(self): def get(endpoint): @@ -299,6 +300,123 @@ def get(endpoint): self.assertGreater(peak, 1) self.assertLessEqual(peak, 8) + def test_dynamic_workflows_skip_tree_lookup_and_do_not_mark_coverage_incomplete(self): + calls = [] + def get(endpoint): + calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint == "orgs/Cratis": + return {"total_private_repos": 0} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PUBLIC] + if "/actions/runs?" in endpoint: + return {"workflow_runs": [{"id": i, "path": path, "head_sha": "abc"} + for i, path in enumerate(("dynamic/github-code-quality/codeql", + "dynamic/dependabot/dependabot-updates"))]} + self.fail(f"Dynamic workflows must not request definitions or jobs: {endpoint}") + data = report.collect(get, "Cratis", SINCE, UNTIL) + output = report.render(*data, SINCE, UNTIL) + self.assertFalse(data[4]) + self.assertIn("GitHub-managed dynamic runs (not inspected): **2**", output) + self.assertNotIn("Incomplete API coverage", output) + + def test_over_cap_window_is_recursively_split_and_inclusive_boundaries_are_deduplicated(self): + calls = [] + end = SINCE + dt.timedelta(days=1) + middle = SINCE + dt.timedelta(hours=12) + created = [SINCE + dt.timedelta(seconds=i * 60) for i in range(1201)] + def get(endpoint): + calls.append(endpoint) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if "/jobs?" in endpoint: + return {"jobs": [job(int(endpoint.split("/runs/")[1].split("/")[0]))]} + query = endpoint.split("created=")[1].split("&")[0] + start, stop = (report.timestamp(value) for value in query.split("..")) + entries = [{"id": i, "name": "Build"} for i, date in enumerate(created) if start <= date <= stop] + page = int(endpoint.rsplit("=", 1)[1]) + return {"total_count": len(entries), "workflow_runs": entries[(page - 1) * 100:page * 100]} + data = report.collect(get, "Cratis", SINCE, end) + self.assertFalse(data[4]) + self.assertEqual(len(data[2]), 1201) + self.assertEqual(len(data[3]), 1201) + self.assertTrue(any(f"..{middle.strftime('%Y-%m-%dT%H:%M:%SZ')}" in call for call in calls)) + + def test_even_a_capped_one_second_window_reports_a_gap(self): + with self.assertRaises(report.RunWindowTooLarge): + list(report.window_runs(lambda _: {"total_count": 1001, "workflow_runs": []}, + "Cratis/Private", SINCE, SINCE + dt.timedelta(seconds=1))) + + def test_repository_listing_failure_aborts_instead_of_publishing_a_token_diagnosis(self): + def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + raise report.APIError("transient repository listing failure") + with self.assertRaisesRegex(report.APIError, "Unable to enumerate repositories"): + report.collect(get, "Cratis", SINCE, UNTIL) + + def test_private_count_failure_retains_enumerated_repository_usage(self): + def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if endpoint == "orgs/Cratis": + raise report.APIError("denied") + if "/jobs?" in endpoint: + return {"jobs": [job()]} + return {"workflow_runs": [{"id": 1}]} + data = report.collect(get, "Cratis", SINCE, UNTIL) + self.assertEqual(len(data[3]), 1) + self.assertIn("Private coverage unconfirmed", report.render(*data, SINCE, UNTIL)) + + def test_busy_private_repositories_finish_with_latency_and_shared_worker_and_request_limits(self): + volumes = {"Studio": 1428, "Direct": 509, "Strategy": 474, + "Infrastructure": 191, "Chronicle.Wolverine": 129, "Other": 397} + repos = [{"full_name": f"Cratis/{name}", "private": True} for name in volumes] + lock = threading.Lock() + calls = active = peak = 0 + # Compress 0.8-second API latency and the 480-second deadline by 80x. + # Studio alone would take >14 seconds serially, beyond this 6-second budget. + latency, budget = .01, 6 + def get(endpoint): + nonlocal calls, active, peak + with lock: + calls += 1 + active += 1 + peak = max(peak, active) + try: + time.sleep(latency) + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return repos + if endpoint == "orgs/Cratis": + return {"total_private_repos": len(repos)} + if "/jobs?" in endpoint: + return {"jobs": [job(int(endpoint.split("/runs/")[1].split("/")[0]))]} + name = endpoint.split("repos/Cratis/")[1].split("/")[0] + day = int(endpoint.split("created=2026-09-")[1][:2]) - 24 + entries = [{"id": i, "name": "Build"} for i in range(volumes[name]) if i % 7 == day] + page = int(endpoint.rsplit("=", 1)[1]) + return {"total_count": len(entries), "workflow_runs": entries[(page - 1) * 100:page * 100]} + finally: + with lock: + active -= 1 + start = time.monotonic() + data = report.collect(get, "Cratis", SINCE, UNTIL, max_seconds=budget) + self.assertFalse(data[4]) + self.assertEqual(len(data[3]), sum(volumes.values())) + self.assertLess(time.monotonic() - start, budget) + self.assertLess(calls, 4900) + self.assertGreater(peak, 1) + self.assertLessEqual(peak, 8) + def test_filtered_search_cap_is_not_silent_truncation(self): with self.assertRaises(report.APIError): list(report.pages(lambda _: {"total_count": 1001, "workflow_runs": []}, "runs", "workflow_runs")) From 4ed4d475d3a41c09a945f78f1b34493b388d8266 Mon Sep 17 00:00:00 2001 From: woksin Date: Thu, 1 Oct 2026 20:48:54 +0200 Subject: [PATCH 4/5] Prioritize private usage collection and remove public discovery --- .github/scripts/actions-usage-report.py | 115 ++++++++---------- .../tests/actions-usage-report.test.py | 106 +++++++++++----- 2 files changed, 126 insertions(+), 95 deletions(-) diff --git a/.github/scripts/actions-usage-report.py b/.github/scripts/actions-usage-report.py index 43cbeda..f08270e 100644 --- a/.github/scripts/actions-usage-report.py +++ b/.github/scripts/actions-usage-report.py @@ -2,7 +2,6 @@ # Copyright (c) Cratis. All rights reserved. # Licensed under the MIT license. See LICENSE file in the project root for full license information. """Report job execution, estimated hosted billing, and scale-set queue pressure.""" -import base64 import datetime as dt import json import math @@ -10,9 +9,8 @@ import subprocess import threading import time -from collections import Counter, defaultdict -from concurrent.futures import ThreadPoolExecutor -from urllib.parse import quote +from collections import Counter, defaultdict, deque +from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait class APIError(Exception): @@ -91,8 +89,8 @@ def window_runs(get, full, start, end): middle = (start + (end - start) / 2).replace(microsecond=0) if middle <= start or middle >= end: raise # Even a one-second window is capped: report the gap explicitly. - yield from window_runs(get, full, start, middle) yield from window_runs(get, full, middle, end) + yield from window_runs(get, full, start, middle) def timestamp(value): @@ -128,59 +126,30 @@ def inventory(repo): full = repo["full_name"] visibility = repo.get("visibility", "private" if repo.get("private") else "public") runs, candidates, seen_runs, failures = [], [], set(), Counter() - definitions, trees = {}, {} - cursor = since + inventory_errors = [] + cursor = until try: - while cursor < until: - window_end = min(cursor + dt.timedelta(days=1), until) + while cursor > since: + window_start = max(cursor - dt.timedelta(days=1), since) try: - for run in window_runs(request, full, cursor, window_end): + for run in window_runs(request, full, window_start, cursor): if run["id"] in seen_runs: continue # GitHub's range endpoints are inclusive. seen_runs.add(run["id"]) runs.append((full, visibility, run)) - try: - if visibility != "private": - # Public ranking needs no jobs. Resolve the historical workflow - # blob via one tree per commit, then cache by content SHA: most - # commits do not change workflows, so no per-run contents call. - path, sha = run.get("path", "").split("@")[0], run.get("head_sha") - if path.startswith("dynamic/"): - continue # GitHub-managed hosted workflows have no repository blob. - if sha not in trees: - trees[sha] = {} - if sha: - tree = request(f"repos/{full}/git/trees/{quote(sha)}?recursive=1") - if not tree.get("truncated"): - trees[sha] = {entry["path"]: entry["sha"] for entry in tree["tree"] - if entry["type"] == "blob"} - blob = trees[sha].get(path) - if not blob: - failures["Public runner discovery incomplete"] += 1 - continue - if blob not in definitions: - definitions[blob] = None - content = request(f"repos/{full}/git/blobs/{quote(blob)}") - definitions[blob] = "cratis-arc" in base64.b64decode(content["content"]).decode() - if definitions[blob] is None: - failures["Public runner discovery incomplete"] += 1 - continue - if not definitions[blob]: - continue + # Public repositories use hosted runners and need run-level data only. + if visibility == "private": candidates.append((full, visibility, run)) - except CollectionStopped: - raise - except (APIError, KeyError, ValueError, UnicodeError): - failures["Job or runner inventory incomplete"] += 1 except CollectionStopped: raise except APIError: failures["Run inventory incomplete"] += 1 - cursor = window_end + cursor = window_start except CollectionStopped: - failures["Collection stopped before completing repository"] += 1 - return runs, candidates, [f"{kind} for {full}: {count} {'daily windows' if kind == 'Run inventory incomplete' else 'runs'}." - for kind, count in failures.items()] + inventory_errors.append(f"Run inventory stopped before completing {full}; older runs may be missing.") + candidates.sort(key=lambda entry: timestamp(entry[2].get("created_at")) or since, reverse=True) + inventory_errors.extend(f"{kind} for {full}: {count} daily windows." for kind, count in failures.items()) + return runs, candidates, inventory_errors def job_inventory(candidate): full, visibility, run = candidate @@ -189,32 +158,44 @@ def job_inventory(candidate): # Each run is independent: a busy repository uses the entire bounded # pool instead of serializing thousands of job requests on one worker. for job in pages(request, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): - if visibility == "private" or "cratis-arc" in (job.get("labels") or []): - found.append((full, visibility, run.get("name") or "Unnamed workflow", job)) + found.append((full, visibility, run.get("name") or "Unnamed workflow", job)) except CollectionStopped: - return found, "Collection stopped before completing repository" + return found, "Job inventory stopped before completing repository (older private runs may be missing)" except (APIError, KeyError, ValueError, UnicodeError): return found, "Job or runner inventory incomplete" return found, None active = [repo for repo in repos if not repo.get("archived")] active.sort(key=lambda repo: not (repo.get("private") or repo.get("visibility") == "private")) - runs, jobs, candidates, seen_jobs, job_failures = [], [], [], set(), Counter() + runs, jobs, seen_jobs, job_failures = [], [], set(), Counter() + remaining_repos, candidates, pending = deque(active), deque(), {} with ThreadPoolExecutor(max_workers=workers) as pool: - for repo_runs, repo_candidates, repo_errors in pool.map(inventory, active): - runs.extend(repo_runs) - candidates.extend(repo_candidates) - errors.extend(repo_errors) - # Reuse the same pool and request budget for job pagination; at most - # `workers` API requests are in flight across both collection stages. - for candidate, (found, failure) in zip(candidates, pool.map(job_inventory, candidates)): - if failure: - job_failures[(candidate[0], failure)] += 1 - for entry in found: - key = (entry[0], entry[3]["id"]) - if key not in seen_jobs: - seen_jobs.add(key) - jobs.append(entry) + while remaining_repos or candidates or pending: + # Bound queued work as well as requests. Prioritize private jobs as soon + # as their repository listing finishes, without waiting for public listings. + while len(pending) < workers and (candidates or remaining_repos): + if candidates: + candidate = candidates.popleft() + pending[pool.submit(job_inventory, candidate)] = candidate + else: + pending[pool.submit(inventory, remaining_repos.popleft())] = None + completed, _ = wait(pending, return_when=FIRST_COMPLETED) + for future in completed: + candidate = pending.pop(future) + if candidate is None: + repo_runs, repo_candidates, repo_errors = future.result() + runs.extend(repo_runs) + candidates.extend(repo_candidates) + errors.extend(repo_errors) + continue + found, failure = future.result() + if failure: + job_failures[(candidate[0], failure)] += 1 + for entry in found: + key = (entry[0], entry[3]["id"]) + if key not in seen_jobs: + seen_jobs.add(key) + jobs.append(entry) errors.extend(f"{kind} for {full}: {count} runs." for (full, kind), count in job_failures.items()) if request.reason: errors.append(request.reason + "; remaining repositories/runs were not collected. Missing usage is unknown.") @@ -299,7 +280,7 @@ def render(repos, private_total, runs, jobs, errors, since, until): unknown_runner += 1 else: p[2] += minutes - lines = [f"Job-level Actions usage from {since.isoformat()} to {until.isoformat()} (UTC).", "", + lines = [f"Actions usage from {since.isoformat()} to {until.isoformat()} (UTC).", "", "Hosted private minutes are estimates: each completed job is rounded up to a minute, " "then multiplied by Linux 1×, Windows 2× or macOS 10×. Public hosted jobs are free. " "Self-hosted execution is not billed by GitHub; this is not an invoice. " @@ -348,8 +329,8 @@ def render(repos, private_total, runs, jobs, errors, since, until): lines.append(f"| {name.replace('|', '/')} | {n} | {minutes:,.1f} | {short} |") lines.extend(["", "### Top 20 public workflows by run elapsed minutes", "", "Run-level elapsed time includes dependency waits and is not job execution or billing. " - "Public job inventory is limited to workflows with literal cratis-arc labels in their historical definitions; " - "indirect runner selection is not observable.", "", + "Public repositories use GitHub-hosted runners and are reported from run-level data only. " + "Job-level billing and cratis-arc queue statistics cover private repositories only.", "", "| Workflow | completed runs | elapsed minutes |", "|---|---:|---:|"]) for name, (n, minutes) in sorted(public_workflows.items(), key=lambda item: -item[1][1])[:20]: lines.append(f"| {name.replace('|', '/')} | {n} | {minutes:,.1f} |") diff --git a/.github/scripts/tests/actions-usage-report.test.py b/.github/scripts/tests/actions-usage-report.test.py index 3e96b9c..52d0095 100644 --- a/.github/scripts/tests/actions-usage-report.test.py +++ b/.github/scripts/tests/actions-usage-report.test.py @@ -1,7 +1,6 @@ # Copyright (c) Cratis. All rights reserved. # Licensed under the MIT license. See LICENSE file in the project root for full license information. """Offline usage-report fixtures; no token or live GitHub calls.""" -import base64 import datetime as dt import importlib.util from pathlib import Path @@ -143,35 +142,27 @@ def test_public_ranking_uses_run_elapsed_not_job_execution(self): self.assertIn("| Cratis/Public / Build | 1 | 10.0 |", output) self.assertIn("not job execution or billing", output) - def test_public_job_inventory_only_for_arc_workflows_and_definition_is_cached(self): - calls = [] + def test_public_collection_uses_only_run_level_data_for_all_workflows(self): def get(endpoint): - calls.append(endpoint) if endpoint == "rate_limit": return {"resources": {"core": {"remaining": 5000}}} if endpoint == "orgs/Cratis": return {"total_private_repos": 0} if endpoint.startswith("orgs/Cratis/repos?"): return [PUBLIC] - if "/git/trees/" in endpoint: - return {"tree": [{"path": ".github/workflows/hosted.yml", "type": "blob", "sha": "hosted"}, - {"path": ".github/workflows/arc.yml", "type": "blob", "sha": "arc"}]} - if "/git/blobs/" in endpoint: - text = "runs-on: cratis-arc" if endpoint.endswith("/arc") else "runs-on: ubuntu-latest" - return {"content": base64.b64encode(text.encode()).decode()} - if "/jobs?" in endpoint: - return {"jobs": [job(group="Custom", labels=["cratis-arc"]), job(2)]} - return {"workflow_runs": [{"id": i, "path": path, "head_sha": "abc", "name": "Build"} - for i, path in ((1, ".github/workflows/hosted.yml"), - (2, ".github/workflows/arc.yml"), - (3, ".github/workflows/arc.yml"))]} + if "/actions/runs?" in endpoint: + return {"workflow_runs": [{"id": i, "path": path, "head_sha": str(i), "name": "Build"} + for i, path in enumerate((".github/workflows/hosted.yml", + ".github/workflows/arc.yml", + "dynamic/dependabot/dependabot-updates"))]} + self.fail(f"Public workflows must not request definitions or jobs: {endpoint}") data = report.collect(get, "Cratis", SINCE, UNTIL) self.assertEqual(len(data[2]), 3) - self.assertEqual(len(data[3]), 1) - self.assertEqual(sum("/git/trees/" in call for call in calls), 1) - self.assertEqual(sum("/git/blobs/" in call for call in calls), 2) - self.assertEqual(sum("/jobs?" in call for call in calls), 2) + self.assertFalse(data[3]) self.assertFalse(data[4]) + output = report.render(*data, SINCE, UNTIL) + self.assertIn("reported from run-level data only", output) + self.assertIn("queue statistics cover private repositories only", output) def test_budget_exhaustion_is_explicit_and_a_fresh_collection_can_resume(self): calls = [] @@ -278,10 +269,6 @@ def get(endpoint): return repos if endpoint == "orgs/Cratis": return {"total_private_repos": 3} - if "/git/trees/" in endpoint: - return {"tree": [{"path": ".github/workflows/build.yml", "type": "blob", "sha": "hosted"}]} - if "/git/blobs/" in endpoint: - return {"content": base64.b64encode(b"runs-on: ubuntu-latest").decode()} if "/jobs?" in endpoint: return {"jobs": []} page = int(endpoint.rsplit("=", 1)[1]) @@ -375,10 +362,10 @@ def get(endpoint): self.assertEqual(len(data[3]), 1) self.assertIn("Private coverage unconfirmed", report.render(*data, SINCE, UNTIL)) - def test_busy_private_repositories_finish_with_latency_and_shared_worker_and_request_limits(self): + def test_busy_private_and_300_sha_public_repositories_finish_with_latency_and_shared_limits(self): volumes = {"Studio": 1428, "Direct": 509, "Strategy": 474, "Infrastructure": 191, "Chronicle.Wolverine": 129, "Other": 397} - repos = [{"full_name": f"Cratis/{name}", "private": True} for name in volumes] + repos = [PUBLIC] + [{"full_name": f"Cratis/{name}", "private": True} for name in volumes] lock = threading.Lock() calls = active = peak = 0 # Compress 0.8-second API latency and the 480-second deadline by 80x. @@ -397,12 +384,16 @@ def get(endpoint): if endpoint.startswith("orgs/Cratis/repos?"): return repos if endpoint == "orgs/Cratis": - return {"total_private_repos": len(repos)} + return {"total_private_repos": len(volumes)} + if "/git/" in endpoint or "repos/Cratis/Public/actions/runs/" in endpoint: + self.fail(f"Public collection must not inspect definitions or jobs: {endpoint}") if "/jobs?" in endpoint: return {"jobs": [job(int(endpoint.split("/runs/")[1].split("/")[0]))]} name = endpoint.split("repos/Cratis/")[1].split("/")[0] day = int(endpoint.split("created=2026-09-")[1][:2]) - 24 - entries = [{"id": i, "name": "Build"} for i in range(volumes[name]) if i % 7 == day] + volume = 2100 if name == "Public" else volumes[name] + entries = [{"id": i, "name": "Build", "head_sha": f"sha-{i % 300}"} + for i in range(volume) if i % 7 == day] page = int(endpoint.rsplit("=", 1)[1]) return {"total_count": len(entries), "workflow_runs": entries[(page - 1) * 100:page * 100]} finally: @@ -412,11 +403,70 @@ def get(endpoint): data = report.collect(get, "Cratis", SINCE, UNTIL, max_seconds=budget) self.assertFalse(data[4]) self.assertEqual(len(data[3]), sum(volumes.values())) + self.assertEqual(len(data[2]), sum(volumes.values()) + 2100) + self.assertEqual(len({run[2]["head_sha"] for run in data[2] if run[1] == "public"}), 300) self.assertLess(time.monotonic() - start, budget) self.assertLess(calls, 4900) self.assertGreater(peak, 1) self.assertLessEqual(peak, 8) + def test_private_jobs_start_before_public_run_inventory_finishes(self): + private_job_started = threading.Event() + def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 5000}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PUBLIC, PRIVATE] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + if "/jobs?" in endpoint: + private_job_started.set() + return {"jobs": [job()]} + if "repos/Cratis/Public/" in endpoint: + self.assertTrue(private_job_started.wait(1), "Private jobs waited for public inventory") + return {"workflow_runs": []} + return {"workflow_runs": [{"id": 1}]} + data = report.collect(get, "Cratis", SINCE, UNTIL, workers=2) + self.assertEqual(len(data[3]), 1) + self.assertFalse(data[4]) + + def test_budget_cut_keeps_newest_private_days_and_names_missing_older_runs(self): + listing_days, job_days = [], [] + def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 111}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + if "/jobs?" in endpoint: + day = int(endpoint.split("/runs/")[1].split("/")[0]) + job_days.append(day) + return {"jobs": [job(day)]} + day = int(endpoint.split("created=2026-09-")[1][:2]) + listing_days.append(day) + return {"workflow_runs": [{"id": day, "created_at": f"2026-09-{day}T12:00:00Z"}]} + data = report.collect(get, "Cratis", SINCE, UNTIL, workers=1) + self.assertEqual(listing_days, list(range(30, 23, -1))) + self.assertEqual(job_days, [30, 29]) + self.assertEqual([entry[3]["id"] for entry in data[3]], [30, 29]) + self.assertTrue(any("older private runs may be missing" in error and "5 runs" in error for error in data[4])) + + def test_run_inventory_stop_is_not_misreported_as_one_missing_run(self): + def get(endpoint): + if endpoint == "rate_limit": + return {"resources": {"core": {"remaining": 104}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return [PRIVATE] + if endpoint == "orgs/Cratis": + return {"total_private_repos": 1} + day = int(endpoint.split("created=2026-09-")[1][:2]) + return {"workflow_runs": [{"id": day}]} + data = report.collect(get, "Cratis", SINCE, UNTIL, workers=1) + self.assertIn("Run inventory stopped before completing Cratis/Private; older runs may be missing.", data[4]) + self.assertTrue(any("Job inventory stopped" in error and "2 runs" in error for error in data[4])) + self.assertFalse(any("1 runs" in error for error in data[4])) + def test_filtered_search_cap_is_not_silent_truncation(self): with self.assertRaises(report.APIError): list(report.pages(lambda _: {"total_count": 1001, "workflow_runs": []}, "runs", "workflow_runs")) From 4cc4fe67987ca67935479aeae39ec1b0b1eac578 Mon Sep 17 00:00:00 2001 From: woksin Date: Thu, 1 Oct 2026 21:46:04 +0200 Subject: [PATCH 5/5] Fix global priority in Actions usage collection --- .github/scripts/actions-usage-report.py | 57 ++++++++++++++----- .../tests/actions-usage-report.test.py | 55 +++++++++++++++--- 2 files changed, 91 insertions(+), 21 deletions(-) diff --git a/.github/scripts/actions-usage-report.py b/.github/scripts/actions-usage-report.py index f08270e..cdbc1d2 100644 --- a/.github/scripts/actions-usage-report.py +++ b/.github/scripts/actions-usage-report.py @@ -3,6 +3,7 @@ # Licensed under the MIT license. See LICENSE file in the project root for full license information. """Report job execution, estimated hosted billing, and scale-set queue pressure.""" import datetime as dt +import heapq import json import math import os @@ -34,7 +35,7 @@ def __init__(self, get, remaining, max_seconds): self.lock = threading.Lock() self.reason = None - def __call__(self, endpoint): + def reserve(self): with self.lock: if time.monotonic() >= self.deadline: self.reason = "Collection time budget exhausted" @@ -43,6 +44,13 @@ def __call__(self, endpoint): if self.reason: raise CollectionStopped(self.reason) self.remaining -= 1 + + def __call__(self, endpoint): + self.reserve() + return self.fetch(endpoint) + + def fetch(self, endpoint): + """Fetch a request whose budget was already reserved.""" try: return self.get(endpoint) except CollectionStopped as error: @@ -147,7 +155,6 @@ def inventory(repo): cursor = window_start except CollectionStopped: inventory_errors.append(f"Run inventory stopped before completing {full}; older runs may be missing.") - candidates.sort(key=lambda entry: timestamp(entry[2].get("created_at")) or since, reverse=True) inventory_errors.extend(f"{kind} for {full}: {count} daily windows." for kind, count in failures.items()) return runs, candidates, inventory_errors @@ -157,10 +164,13 @@ def job_inventory(candidate): try: # Each run is independent: a busy repository uses the entire bounded # pool instead of serializing thousands of job requests on one worker. - for job in pages(request, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): + # The scheduler reserves page one in priority order before submission; + # workers reserve any additional pages from the same shared budget. + get_jobs = lambda endpoint: request.fetch(endpoint) if endpoint.endswith("&page=1") else request(endpoint) + for job in pages(get_jobs, f"repos/{full}/actions/runs/{run['id']}/jobs?filter=all", "jobs"): found.append((full, visibility, run.get("name") or "Unnamed workflow", job)) except CollectionStopped: - return found, "Job inventory stopped before completing repository (older private runs may be missing)" + return found, "Job inventory stopped before completing listed private runs" except (APIError, KeyError, ValueError, UnicodeError): return found, "Job or runner inventory incomplete" return found, None @@ -168,24 +178,43 @@ def job_inventory(candidate): active = [repo for repo in repos if not repo.get("archived")] active.sort(key=lambda repo: not (repo.get("private") or repo.get("visibility") == "private")) runs, jobs, seen_jobs, job_failures = [], [], set(), Counter() - remaining_repos, candidates, pending = deque(active), deque(), {} + private_repos = deque(repo for repo in active if repo.get("private") or repo.get("visibility") == "private") + public_repos = deque(repo for repo in active if not (repo.get("private") or repo.get("visibility") == "private")) + candidates, pending, private_inventories = [], {}, set() with ThreadPoolExecutor(max_workers=workers) as pool: - while remaining_repos or candidates or pending: - # Bound queued work as well as requests. Prioritize private jobs as soon - # as their repository listing finishes, without waiting for public listings. - while len(pending) < workers and (candidates or remaining_repos): - if candidates: - candidate = candidates.popleft() + while private_repos or public_repos or candidates or pending: + # Inventory every private repository before jobs so a budget cut drops + # the globally oldest listed runs. Start public inventories next, but + # do not wait for them to finish before collecting private jobs. + while len(pending) < workers: + if private_repos: + future = pool.submit(inventory, private_repos.popleft()) + private_inventories.add(future) + pending[future] = None + elif private_inventories: + break + elif public_repos: + pending[pool.submit(inventory, public_repos.popleft())] = None + elif candidates: + candidate = heapq.heappop(candidates)[3] + try: + request.reserve() + except CollectionStopped: + job_failures[(candidate[0], "Job inventory stopped before completing listed private runs")] += 1 + continue pending[pool.submit(job_inventory, candidate)] = candidate else: - pending[pool.submit(inventory, remaining_repos.popleft())] = None + break completed, _ = wait(pending, return_when=FIRST_COMPLETED) for future in completed: candidate = pending.pop(future) if candidate is None: + private_inventories.discard(future) repo_runs, repo_candidates, repo_errors = future.result() runs.extend(repo_runs) - candidates.extend(repo_candidates) + for entry in repo_candidates: + created = timestamp(entry[2].get("created_at")) or since + heapq.heappush(candidates, (-created.timestamp(), entry[0], entry[2]["id"], entry)) errors.extend(repo_errors) continue found, failure = future.result() @@ -289,7 +318,7 @@ def render(repos, private_total, runs, jobs, errors, since, until): f"Inventoried runs: **{len(runs)}**, jobs (all attempts): **{len(jobs)}**.", ""] dynamic_runs = sum(visibility != "private" and run.get("path", "").startswith("dynamic/") for _, visibility, run in runs) - lines.extend([f"GitHub-managed dynamic runs (not inspected): **{dynamic_runs}**.", ""]) + lines.extend([f"GitHub-managed dynamic runs (included in public run-level totals): **{dynamic_runs}**.", ""]) if private_visible == 0: lines.append("**Private repository coverage unavailable:** PAT_WORKFLOWS exposes no private repositories. " "This is not evidence of zero private usage. Give the reporting token read access to private " diff --git a/.github/scripts/tests/actions-usage-report.test.py b/.github/scripts/tests/actions-usage-report.test.py index 52d0095..b0353b5 100644 --- a/.github/scripts/tests/actions-usage-report.test.py +++ b/.github/scripts/tests/actions-usage-report.test.py @@ -287,7 +287,7 @@ def get(endpoint): self.assertGreater(peak, 1) self.assertLessEqual(peak, 8) - def test_dynamic_workflows_skip_tree_lookup_and_do_not_mark_coverage_incomplete(self): + def test_dynamic_workflows_are_included_in_public_run_level_totals(self): calls = [] def get(endpoint): calls.append(endpoint) @@ -305,7 +305,7 @@ def get(endpoint): data = report.collect(get, "Cratis", SINCE, UNTIL) output = report.render(*data, SINCE, UNTIL) self.assertFalse(data[4]) - self.assertIn("GitHub-managed dynamic runs (not inspected): **2**", output) + self.assertIn("GitHub-managed dynamic runs (included in public run-level totals): **2**", output) self.assertNotIn("Incomplete API coverage", output) def test_over_cap_window_is_recursively_split_and_inclusive_boundaries_are_deduplicated(self): @@ -368,9 +368,9 @@ def test_busy_private_and_300_sha_public_repositories_finish_with_latency_and_sh repos = [PUBLIC] + [{"full_name": f"Cratis/{name}", "private": True} for name in volumes] lock = threading.Lock() calls = active = peak = 0 - # Compress 0.8-second API latency and the 480-second deadline by 80x. - # Studio alone would take >14 seconds serially, beyond this 6-second budget. - latency, budget = .01, 6 + # Compress API latency, with extra deadline margin for shared runners. + # Studio alone would take >14 seconds serially, beyond this 10-second budget. + latency, budget = .01, 10 def get(endpoint): nonlocal calls, active, peak with lock: @@ -430,7 +430,7 @@ def get(endpoint): self.assertEqual(len(data[3]), 1) self.assertFalse(data[4]) - def test_budget_cut_keeps_newest_private_days_and_names_missing_older_runs(self): + def test_budget_cut_keeps_newest_private_days_and_names_incomplete_listed_runs(self): listing_days, job_days = [], [] def get(endpoint): if endpoint == "rate_limit": @@ -450,7 +450,48 @@ def get(endpoint): self.assertEqual(listing_days, list(range(30, 23, -1))) self.assertEqual(job_days, [30, 29]) self.assertEqual([entry[3]["id"] for entry in data[3]], [30, 29]) - self.assertTrue(any("older private runs may be missing" in error and "5 runs" in error for error in data[4])) + self.assertIn("Job inventory stopped before completing listed private runs for Cratis/Private: 5 runs.", data[4]) + self.assertNotIn("older private runs may be missing", "\n".join(data[4])) + + def test_multi_repository_budget_cut_drops_globally_oldest_listed_private_runs(self): + repos = [{"full_name": f"Cratis/P{i}", "private": True} for i in range(9)] + for workers in (1, 8): + with self.subTest(workers=workers): + first_inventory_finished = threading.Event() + finished_inventories, collected = set(), [] + lock = threading.Lock() + def get(endpoint): + if endpoint == "rate_limit": + # Org metadata, nine weekly inventories, then eighteen jobs. + return {"resources": {"core": {"remaining": 100 + 2 + 9 * 7 + 18}}} + if endpoint.startswith("orgs/Cratis/repos?"): + return repos + if endpoint == "orgs/Cratis": + return {"total_private_repos": 9} + full = endpoint.split("repos/")[1].split("/actions/")[0] + if "/jobs?" in endpoint: + identity = int(endpoint.split("/runs/")[1].split("/")[0]) + with lock: + self.assertEqual(finished_inventories, {repo["full_name"] for repo in repos}) + collected.append((full, identity % 100)) + return {"jobs": [job(identity)]} + day = int(endpoint.split("created=2026-09-")[1][:2]) + if full == "Cratis/P8": + self.assertTrue(first_inventory_finished.wait(1)) + if day == 24: + with lock: + finished_inventories.add(full) + if full == "Cratis/P0": + first_inventory_finished.set() + identity = int(full.rsplit("P", 1)[1]) * 100 + day + return {"workflow_runs": [{"id": identity, "created_at": f"2026-09-{day}T12:00:00Z"}]} + data = report.collect(get, "Cratis", SINCE, UNTIL, workers=workers) + self.assertEqual(set(collected), {(repo["full_name"], day) for repo in repos for day in (30, 29)}) + self.assertEqual(len(data[3]), 18) + self.assertEqual(len(data[2]), 63) + for repo in repos: + self.assertIn("Job inventory stopped before completing listed private runs " + f"for {repo['full_name']}: 5 runs.", data[4]) def test_run_inventory_stop_is_not_misreported_as_one_missing_run(self): def get(endpoint):