From 35e2475f056cdc588706ca11c154e403cefb90e5 Mon Sep 17 00:00:00 2001 From: Pablo Ariel Di Loreto Date: Thu, 1 Oct 2026 03:46:32 +0000 Subject: [PATCH] chore(policy): review every risk level with Sonnet 5.5, at low, medium and high effort The maintainer decided one model for every review: the risk now chooses only the effort. The default policy gives the high level Sonnet 5.5 at high effort, the review falls back to Sonnet 5.5 when the policy names none, and the docs, comments and self-tests that described a light and a strong model say so. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/claude-review.yml | 8 ++++---- .github/workflows/issue-triage.yml | 2 +- CONTRIBUTING.md | 2 +- README.md | 4 ++-- policy/review-policy.default.yml | 2 +- scripts/policy.py | 4 ++-- 6 files changed, 11 insertions(+), 11 deletions(-) diff --git a/.github/workflows/claude-review.yml b/.github/workflows/claude-review.yml index 508708a..cc59c75 100644 --- a/.github/workflows/claude-review.yml +++ b/.github/workflows/claude-review.yml @@ -6,9 +6,9 @@ name: Claude review # 1. A deterministic policy (scripts/policy.py) decides the lowest risk the # PR can have from its paths alone: the reviewer can raise it, never # lower it. The floor also picks the model and the effort: a low floor -# (docs, tests, lockfiles) goes to the light model at low effort, a -# medium one to the light model at medium effort, a high one to the -# strong model, as the policy files say; a repository +# (docs, tests, lockfiles) gets low effort, a medium one medium effort +# and a high one high effort, all on the same model, as the policy +# files say; a repository # overrides any level in its .github/review-policy.yml (`review:`), which # is read from the base branch so a change cannot pick its own reviewer. At a low floor the brief carries the profiles and the diff # only; AGENTS.md and docs/architecture.md are added above it. @@ -260,7 +260,7 @@ jobs: mode=incremental; range="$last..$HEAD" fi model=$MODEL; effort=$EFFORT - if [ -z "$model" ]; then echo "::warning::The policy at central-ref gave no model; using the strong default."; model=claude-opus-5-5; fi + if [ -z "$model" ]; then echo "::warning::The policy at central-ref gave no model; using the default."; model=claude-sonnet-5-5; fi if [ -z "$effort" ]; then effort=medium; fi { echo "mode=$mode"; echo "range=$range"; echo "last=$last"; echo "count=$count"; echo "model=$model"; echo "effort=$effort"; } >> "$GITHUB_OUTPUT" echo "Mode: $mode. Model: $model, effort $effort. Previous review: ${last:-none} ($count so far)." diff --git a/.github/workflows/issue-triage.yml b/.github/workflows/issue-triage.yml index b88ca32..4ced8fe 100644 --- a/.github/workflows/issue-triage.yml +++ b/.github/workflows/issue-triage.yml @@ -4,7 +4,7 @@ name: Issue triage # README, readme.txt, docs/roadmap.md and the open issues, classifies it and # does one thing: applies a label and posts one reply. It never closes an # issue, never promises a fix and never gives a date. The `low` reviewer of -# the policy (the light model) does this work. +# the policy (the low-risk reviewer) does this work. # # bug:unconfirmed a bug report with enough to try to reproduce it # (issue-repro.yml picks it up from this label) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 1c5042a..f179f8c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -70,7 +70,7 @@ Pull requests from branches of the repository (not drafts, not forks) are review - The first review reads the whole change; later ones read only what you pushed since, so keep pushes meaningful. Editing the title or description does not trigger a new review; an edited description counts as unchecked (no auto-merge) until the next push or the `review:full` label. - The review decides the type of the change from the diff and sets it as a `type:*` label; it corrects the title's type to match (the version number is computed from these labels). If it is wrong, set a different `type:*` label yourself: the bot keeps a label a person set. -- Low- and medium-risk changes get the light model (low effort for docs, tests and lockfiles, medium for ordinary code); high-risk ones the strong one. +- Every change is reviewed by the same model; the risk decides the effort: low for docs, tests and lockfiles, medium for ordinary code, high for high-risk changes. - After 5 automatic reviews the next push keeps the last verdict and a human merges: add the `review:full` label, then push or re-run the review, to ask for another full pass. - It merges on its own only when the change touches only low-risk paths, the review rated it low risk and low complexity without blocking, the description matches the code (nothing claimed the diff does not do, no behaviour change left unsaid), the author is a trusted maintainer (or Dependabot), every required check is green and the policy has auto-merge on. Anything else, a maintainer merges. diff --git a/README.md b/README.md index 6863224..5122a9c 100644 --- a/README.md +++ b/README.md @@ -86,7 +86,7 @@ is in the run alone. code: - "bin/**" review: - high: { model: claude-opus-5-5, effort: high } + high: { model: claude-sonnet-5-5, effort: xhigh } budget-usd: 2 auto-merge: true ``` @@ -168,7 +168,7 @@ ref (default `v2`), not at the ref of its `uses:` line: a caller that pins | --- | --- | --- | | [`conventions.yml`](.github/workflows/conventions.yml) | Branch, title, commits, description sections, no "Generated with" footer (all in `scripts/conventions.sh`), relative doc links, retired names. Reads the labels and the review App's last record live, so a re-run sees today's `type:*` label. | `retired-names`, `required-sections`, `max-header`, `central-ref`, `review-bot` | | [`claude-review.yml`](.github/workflows/claude-review.yml) | The review described above. Outputs `risk`, `complexity`, `floor`, `trusted`, `blocking`. | `profile`, `central-ref`, `max-auto-reviews` | -| [`review-reply.yml`](.github/workflows/review-reply.yml) | Answers `@dilux-bot` mentions from members and collaborators, with the strong model. | `profile`, `central-ref` | +| [`review-reply.yml`](.github/workflows/review-reply.yml) | Answers `@dilux-bot` mentions from members and collaborators, with the high-risk reviewer. | `profile`, `central-ref` | | [`auto-merge.yml`](.github/workflows/auto-merge.yml) | Turns GitHub's auto-merge on or off from the review's outputs and commits the description verbatim. `pull-request-edited.yml` runs it on `edited` too. | the five review outputs | | [`scripts/conventions.sh`](scripts/conventions.sh) | Not a workflow: the conventions a pull request is held to (branch, title, every commit header, no session trailer, description sections, no "Generated with" footer). `conventions.yml` runs it on a pull request, `local-review.sh` before one; `--test` for its own tests. | env: `BRANCH`, `TITLE`, `BODY`, `BASE`, `HEAD_REF`, `COMMENTS_FILE`, `REVIEW_BOT`, `MAX_HEADER`, `SECTIONS`, `LABELS`, `AUTHOR_TYPE` | | [`scripts/review-brief.sh`](scripts/review-brief.sh) | Not a workflow: the review brief, the one file the Claude review reads (profiles, `AGENTS.md`, `docs/architecture.md`, policy floor, description, diff). `claude-review.yml` and `local-review.sh` build it with the same script; `--test` for its own tests. | env: see the script's header | diff --git a/policy/review-policy.default.yml b/policy/review-policy.default.yml index 2cd9ceb..d6107f2 100644 --- a/policy/review-policy.default.yml +++ b/policy/review-policy.default.yml @@ -30,7 +30,7 @@ high-risk: review: low: { model: claude-sonnet-5-5, effort: low } medium: { model: claude-sonnet-5-5, effort: medium } - high: { model: claude-opus-5-5, effort: medium } + high: { model: claude-sonnet-5-5, effort: high } # The most one review run may spend, in USD. A repository can lower or raise it. budget-usd: 3 diff --git a/scripts/policy.py b/scripts/policy.py index 0e739ca..e0e7b3e 100644 --- a/scripts/policy.py +++ b/scripts/policy.py @@ -208,7 +208,7 @@ def self_test(): here = os.path.dirname(os.path.abspath(__file__)) default = os.path.join(here, "..", "policy", "review-policy.default.yml") with tempfile.NamedTemporaryFile("w", suffix=".yml", delete=False, encoding="utf-8") as fh: - fh.write("auto-merge: false\nreview:\n low: {model: claude-sonnet-5-5}\n medium: {model: claude-sonnet-5-5}\n high: {model: claude-opus-5-5}\n") + fh.write("auto-merge: false\nreview:\n low: {model: claude-sonnet-5-5}\n medium: {model: claude-sonnet-5-5}\n high: {model: claude-sonnet-5-5}\n") org_off = fh.name cases = [ # (name, defaults, repo policy, files, extra env, expected outputs, expected exit) @@ -217,7 +217,7 @@ def self_test(): ("compiled translations are not low: nobody can read them", default, "", ["languages/x-es_AR.mo"], {}, {"floor": "medium"}, 0), ("their sources and template are", default, "", ["languages/x-es_AR.po", "languages/x.pot"], {}, {"floor": "low"}, 0), ("code is medium", default, "", ["includes/a.php"], {}, {"floor": "medium", "model": "claude-sonnet-5-5", "effort": "medium"}, 0), - ("a workflow gets the strong reviewer", default, "", [".github/workflows/a.yml"], {}, {"floor": "high", "model": "claude-opus-5-5"}, 0), + ("a workflow gets high effort", default, "", [".github/workflows/a.yml"], {}, {"floor": "high", "model": "claude-sonnet-5-5", "effort": "high"}, 0), ("docs plus code is medium", default, "", ["docs/a.md", "includes/a.php"], {}, {"floor": "medium"}, 0), ("a workflow is high", default, "", [".github/workflows/a.yml"], {}, {"floor": "high"}, 0), ("AGENTS.md is high, though it is Markdown", default, "", ["AGENTS.md"], {}, {"floor": "high"}, 0),