diff --git a/dev/.claude-plugin/plugin.json b/dev/.claude-plugin/plugin.json index bc87913..5733a85 100644 --- a/dev/.claude-plugin/plugin.json +++ b/dev/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dev", - "version": "3.4.0", + "version": "3.5.0", "description": "Development workflow skills: scope changes with argued decisions, build across unit/integration/e2e with every scenario proven by tests, ship with a deterministic quality gauntlet and an adversarially verified review, create structured commits that feed a decision ledger, and render pitches or comprehension quizzes.", "author": { "name": "Tobrun" diff --git a/dev/README.md b/dev/README.md index 4fd60f7..ada9de0 100644 --- a/dev/README.md +++ b/dev/README.md @@ -2,8 +2,9 @@ Development workflow skills for Claude Code, Codex, opencode, and Pi, built around two ideas: layered tests are the enforceable spec for behavior, and every phase produces something a human actually reviews as HTML, not markdown scrolling. -The skills chain loosely rather than as a rigid pipeline: `/scope` interviews for the real problem, argues every design decision against alternatives, and writes a self-contained spec whose change plan carries layer-tagged test scenarios; `/scope-review` puts the settled spec through a fresh-context, adversarially verified agent panel that checks the plan against the actual repo and refines the spec in place, looping without a human and closing with a short interview for the few findings only the user can decide, so a finished run hands `build` a spec ready to implement; `/build` executes the spec's change sets across unit/integration/e2e, proving every scenario with a test that has been seen to fail, in parallel waves where file lists allow, keeping a running implementation-notes log; `/ship` runs a deterministic quality gauntlet - the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, flakiness, mutation testing - looping fix agents until the checkers pass, then verifies the result with a fan-out review panel that checks spec conformance, e2e coverage, and logged deviations; `/commit` groups pending changes into granular commits with structured what/why messages; `/to-pitch` and `/to-quiz` turn finished work into a buy-in doc or a comprehension check. +The skills chain loosely rather than as a rigid pipeline: `/scope` interviews for the real problem, argues every design decision against alternatives, and writes a self-contained spec whose change plan carries layer-tagged test scenarios; `/scope-review` puts the settled spec through a fresh-context, adversarially verified agent panel that checks the plan against the actual repo and refines the spec in place, looping without a human and closing with a short interview for the few findings only the user can decide, so a finished run hands `build` a spec ready to implement; `/build` executes the spec's change sets across unit/integration/e2e, proving every scenario with a real test at its tagged layer, in parallel waves where file lists allow, keeping a running implementation-notes log; `/ship` runs a deterministic quality gauntlet - the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, flakiness, mutation testing - looping fix agents until the checkers pass, then verifies the result with a fan-out review panel that checks spec conformance, e2e coverage, and logged deviations; `/commit` groups pending changes into granular commits with structured what/why messages; `/to-pitch` and `/to-quiz` turn finished work into a buy-in doc or a comprehension check. The durable context is deliberately small: the code, its tests, the active spec under `.dev/{plan-name}/`, and three repo-tracked registries the skills maintain in the consuming project - `docs/decisions.md` (design decisions with their argued alternatives, read only after a review forms its findings), `docs/contracts.md` (boundary guarantees, read as premises before a review walks the diff), and `docs/dependencies.md` (machine-checkable module dependency rules, enforced by `ship`). +The files under `.dev/{plan-name}/` are written as a run goes, not when a stage closes: the spec opens during the interview, a report opens before its panel returns, the implementation notes gain an entry per change set and per fixup, and the PR body fills check by check, so a run can be followed from its files and a dead session loses only what was in flight. Alongside them, `docs/architecture.md` is a plain high-level overview of the system - components, flows, boundaries, entry points - captured in full the first time a skill needs it and finds it absent, then kept current by build and commit whenever the structure changes, with a small checker that catches stale paths and files no component covers. Every producing skill renders its own output as self-contained HTML under `/tmp/{project-slug}/reports/`. It publishes only when the user requests a shareable link and the host provides an artifact-publishing tool. Every skill is explicit-invocation only: Claude Code and Pi use `disable-model-invocation: true`, the generated Codex distribution uses `agents/openai.yaml` with `allow_implicit_invocation: false`, and opencode enforces it with a `permission.skill` rule set to `ask` (see opencode installation below). Skills recommend the next step rather than launching each other. diff --git a/dev/evals/README.md b/dev/evals/README.md index 0dbef3c..12c3ece 100644 --- a/dev/evals/README.md +++ b/dev/evals/README.md @@ -6,7 +6,7 @@ Eval definitions for the `dev` plugin's skills: realistic prompts and objective - `{skill}.json` - one file per skill: the eval prompt(s), the fixture each expects, and the assertions to grade the output against. Covers all 7 skills: `scope`, `scope-review`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz`. - `results.md` - the record of the most recent full run: scores, methodology, and findings. -- `tests/` - unit tests for the deterministic scripts the skills loop against (`lint-spec.py`, `change-set-brief.py`), run by `scripts/validate.sh` as check D01. +- `tests/` - unit tests for the deterministic scripts the skills loop against (`lint-spec.py`, `change-set-brief.py`, `check-tests.py`), run by `scripts/validate.sh` as check D01. `build` runs in `"functional"` mode (a real fixture, a real subagent run, assertions checked against the actual output). `scope`, `scope-review`, `commit`, `ship`, `to-pitch`, and `to-quiz` run in `"comprehension"` mode instead - each depends on either an interactive question loop, a live codebase, or prior artifacts (a finished spec, implementation notes, an e2e report) that are too expensive to stage on every iteration, so these check policy comprehension of the skill text directly. diff --git a/dev/evals/build.json b/dev/evals/build.json index d3b1cf5..b6ac0a8 100644 --- a/dev/evals/build.json +++ b/dev/evals/build.json @@ -25,6 +25,7 @@ "The spec's Validation block is run once per change set as the gate before its commit, not repeated on an unchanged tree", "No test is run just to be seen failing and no working code is broken to prove a test red: each test is written with its slice and run to green", ".dev/{plan-name}/implementation-notes.md is created and has one entry per change set", + "implementation-notes.md opens with a Build run entry naming the Validation commands and the waves, written before change set 1 is implemented, and change set 1's entry is in the file before change set 2's work starts", "After both change sets are done, the user is asked to run ship rather than a review panel or gauntlet being launched automatically", "Each implementation-notes.md entry carries a Tests added: line naming real path::test name references", "scripts/check-tests.py is run against the plan directory and looped on until it exits clean, before the e2e pass" @@ -71,6 +72,7 @@ "The failing e2e scenario is diagnosed and fixed rather than reported as an accepted failure or a limitation", "The fix adds a test at the cheapest layer that can catch the bug, rather than only patching until the e2e passes", "The scenario is re-run and re-captured after the fix, not flipped to pass with the original capture", + "implementation-notes.md gains a Fixup entry for the fix, written with the fix rather than at the end of the run, naming the e2e scenario that found it and the test added", "The final report's summary counts match the scenarios array and reflect the passing re-run", "The user is asked to run ship only after the e2e run is green" ] @@ -85,7 +87,7 @@ "The screenshot failure is diagnosed and fixed rather than labeled pre-existing, flaky, unrelated, or an accepted deviation", "The fix removes the time-dependent assumption without adding a retry, sleep, timeout increase, or looser assertion", "The full screenshot-matrix command is rerun and green before the PR question", - "The discovered CI commands and outcomes are recorded in implementation-notes.md", + "The discovered CI commands and outcomes are recorded in implementation-notes.md, each as its command finishes", "If the user approves a PR, required checks are watched to a terminal state and a deterministic failure is fixed and pushed rather than merely reported as restarted" ] }, diff --git a/dev/evals/results.md b/dev/evals/results.md index 7cd9888..f6c733b 100644 --- a/dev/evals/results.md +++ b/dev/evals/results.md @@ -1,6 +1,6 @@ # Eval Results -Status: no full run recorded for the current skill set; one partial run below. +Status: no full run recorded for the current skill set; two partial runs below. The last recorded run (2026-07-24) covered the pre-pivot five-skill chain and was invalidated by the pivot to the scope pipeline; its scores were removed rather than left to invite false confidence. Run the harness below against the current 6 skills (`scope`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz`) and replace this file with the dated results. @@ -32,6 +32,43 @@ Findings: - Wall time did not improve on this fixture: the suite costs 20 seconds, so two saved runs are under a minute, less than the difference between two model runs. The saving grows with the cost of the block and the number of change sets; a real build is needed to measure it. - The old run's notes named a test with a comma in it, which `check-tests.py` then split in two. The checker now splits only where a `path::name` follows the comma. +## Partial run, 2026-10-02: plan files written as the run goes + +Old is commit c32e4fb, new is the version that opens each plan file early and writes it as the run goes. +One subagent per skill and version answered that skill's comprehension prompts from the skill text and its linked references. + +| Skill | Eval | Old | New | +| ----- | ---- | --- | --- | +| scope | `scope-confidence-threshold` | 4/4 | 4/4 | +| scope | `scope-written-as-the-run-goes` | 0/5 | 5/5 | +| scope-review | `scope-review-contract` | 3/4 | 4/4 | +| scope-review | `scope-review-loop-limits` | 4/4 | 4/4 | +| scope-review | `scope-review-report-as-the-run-goes` | 0/5 | 5/5 | +| ship | `gauntlet-policy` | 8/8 | 8/8 | +| ship | `review-policy` | 10/10 | 10/10 | +| ship | `ship-files-as-the-run-goes` | 2/6 | 6/6 | + +The old `scope-confidence-threshold` run was graded against the old wording of its answer 3, and the old `scope-review-contract` run missed only the report in answer 1, which the old text did not name. +The two `ship-files-as-the-run-goes` answers the old text got right are the two whose answer is "no" in both versions. + +One functional `build` run on the new text only: a Node CLI fixture with two change sets, the second consuming the first, and a launcher that strips trailing zeros in the built CLI. +A watcher logged every change of the headings in `implementation-notes.md` next to the commit count. + +| Time | Commits | Notes file | +| ---- | ------- | ---------- | +| 15:56:22 | 1 (fixture) | absent | +| 15:58:41 | 1 | `## Build run` entry: Validation commands, waves `[1] [2]` | +| 15:59:20 | 2 | plus `## Change set 1` | +| 16:00:41 | 3 | plus `## Change set 2` with four deviations | +| 16:01:39 | 5 | plus `## E2E pass` and `## CI parity`, one line per command | + +Findings: + +- The run entry was in the file before any change set was committed, and each change set's entry was in the file before the next one started; `check-tests.py` exited clean. +- The fixup entry was not exercised: the agent found the launcher bug while exploring, fixed it inside change set 2, and logged it as a deviation there, so the e2e pass was green on its first run. Only the unit tests in `tests/` cover the fixup entry. +- The `ship` comprehension run showed that nothing forbade writing unverified findings into the open report, and that the question about the Quality rows assumed dead code and duplication run one after the other. The skill now says a finding enters the report only once verified, the gauntlet writes one row per check for the batched five, and the question was reworded. +- Answer 7 of `gauntlet-policy` still expected `commit` to be recommended at every wrap-up; it now says that holds only when phase 3 did not run. + ## Re-running this harness 1. Pick a baseline commit (the last commit before the change under test) and the working tree as "new". diff --git a/dev/evals/scope-review.json b/dev/evals/scope-review.json index 9e95f61..0351677 100644 --- a/dev/evals/scope-review.json +++ b/dev/evals/scope-review.json @@ -7,7 +7,7 @@ "id": "scope-review-contract", "prompt": "Answer briefly, from the skill text only: (1) What may scope-review edit, and what must it never touch? (2) Can a finding reach the spec without being verified, and can the orchestrator apply its own opinion as a refinement? (3) What happens when lint-spec.py reports problems before the loop starts? (4) A verified finding shows a change set implements the S3 alternative that its linked decision rejected in favor of local disk - is that refinable, and if so what does the refine agent write?", "assertions": [ - "Answer 1: it edits spec.md and nothing else - never code, never docs/decisions.md, never docs/contracts.md", + "Answer 1: it edits spec.md, keeps its own spec-review_N.md report, and promotes settled changes to docs/decisions.md and docs/contracts.md - never code, and never any other file", "Answer 2: no - findings reach the spec only through verification (REFUTED findings drop, a BLOCK survives only when CONFIRMED), and the orchestrator's own reading is not a lens", "Answer 3: stop and recommend finishing the scope run - refinement presumes a mechanically settled spec, and repairing an unfinished draft is scope's job", "Answer 4: refinable - a settled chosen decision outranks the spec's prose in the authority order, so the change set is rewritten to match the decision (local disk), not the other way around" @@ -22,6 +22,17 @@ "Answer 3: recorded user intent, then the repo's reality, then settled chosen decisions, then the spec's prose - each level beats everything below it", "Answer 4: APPROVED when every finding was refined or answered (build can start directly); APPROVED WITH DEFERRALS when something remains - an escalation defers only when the answer invalidates the premise or opens a genuinely new effort, when there is no interactive channel, or when the user declines to answer" ] + }, + { + "id": "scope-review-report-as-the-run-goes", + "prompt": "Answer briefly, from the skill text and the references it links only: (1) When is spec-review_N.md created, and what does its Verdict line say then? (2) Round 1's refine agents have just returned with four refinements applied, and round 2's panel is about to start. What does the report hold now? (3) The user answers the first of three escalations. When is that recorded in the report? (4) What is left to write into the report at the wrap-up? (5) A later scope run finds a spec-review_2.md whose verdict is still IN PROGRESS and no scope-review run is going. May it take the report's findings as its opening agenda?", + "assertions": [ + "Answer 1: in step 1, after the lint gate passes and before round 1's panel runs, at the next free index; Verdict: IN PROGRESS", + "Answer 2: round 1's finding counts and each of the four refinements, written when the refine agents returned - not held back until the end of the run", + "Answer 3: at once, when it is answered (or deferred) - not after all three escalations are done", + "Answer 4: only the verdict - everything else has been filling since step 1", + "Answer 5: no - an unfinished file left by a run that is no longer going is a draft, never a result; scope-review picks it up at the same index and no other skill builds on it" + ] } ] } diff --git a/dev/evals/scope.json b/dev/evals/scope.json index 8189bbd..683ab55 100644 --- a/dev/evals/scope.json +++ b/dev/evals/scope.json @@ -73,7 +73,18 @@ "Answer 0: no question; the table is applied and its chosen line carries `auto-applied at Confidence: 82%` in its because clause", "Answer 1: yes, 55% is below the 75% threshold, so the user is asked", "Answer 2: yes, whatever the score, because it changes what was asked for (a different problem than the request names)", - "Answer 3: once, in one block at the end of the interview, so a single reply can overturn any of them before the spec is written" + "Answer 3: once, in one block at the end of the interview, so a single reply can overturn any of them before the catalog builds on them" + ] + }, + { + "id": "scope-written-as-the-run-goes", + "prompt": "Answer briefly, from the skill text and the references it links only. (1) At what point in a run is spec.md created, and what does it hold at that moment? (2) You have just cataloged nine decisions and are about to talk the big ones through with the user. Where are those nine decisions right now? (3) The user answers a question that settles D-file-storage. When does the spec change? (4) The blind spot subagent is hunting through the repo while you catalog. What keeps it from reading your catalog? (5) The session dies halfway through the decision talk-through. What does the next scope run find, and how does it treat it?", + "assertions": [ + "Answer 1: during the interview, once the first answers say what the change is about - not after the interview or the catalog; it holds the title, the date, what was understood so far, and empty Research, Scope, and Change plan sections", + "Answer 2: in the research section of spec.md, each written as an [open] entry the moment it was cataloged - not only in the conversation", + "Answer 3: at once, when the answer lands - the decision's marks change in the file as the talk-through settles it, not in a later write-up", + "Answer 4: it is handed only the original request and the relevant code, and is told to stay out of .dev/, where the catalog is being written", + "Answer 5: a spec.md that lint-spec.py does not pass, holding everything established before the session died; it is a draft that scope picks up where it stopped, and no other skill builds on it" ] } ] diff --git a/dev/evals/ship.json b/dev/evals/ship.json index 7736ece..271f3f4 100644 --- a/dev/evals/ship.json +++ b/dev/evals/ship.json @@ -14,7 +14,7 @@ "Answer 4: the accepted threshold is recorded as a D- entry in docs/decisions.md, dated and sourced to this run, so the next run reads it instead of re-arguing; never adjusted silently", "Answer 5: the branch diff against the default branch; full-repo only when the user asks", "Answer 6: it becomes a human call - presented as a real defect, a threshold worth changing, or a rule the spec should have amended; the loop does not grind on", - "Answer 7: mechanics only, not meaning; by default the review phase runs next over the post-fix diff, and commit is recommended for the accumulated fixes at wrap up, never invoked", + "Answer 7: mechanics only, not meaning; by default the review phase runs next over the post-fix diff, then the pull request phase; commit is recommended at wrap up only when phase 3 did not run, never invoked", "Answer 8: no; ship always runs required PR commands before review, the red screenshot is a violation even if called flaky or pre-existing, and pre-existing status requires merge-base proof plus a human call" ] }, @@ -33,6 +33,18 @@ "Answer 9: no; a single phase runs on request, and the mutating gauntlet is skipped entirely in a review-only run", "Answer 10: scripts/aggregate-findings.py does both - plan numbers the BLOCK and CONCERN findings for the verifiers, aggregate applies the verdicts; never another agent, and never hand-aggregation" ] + }, + { + "id": "ship-files-as-the-run-goes", + "prompt": "Answer briefly, from the skill text and the references it links only: (1) In a default run, when is pr.md created, and what is in it after the five read-only analyzers have finished but before the complexity check starts? (2) The user asked for a gauntlet only run. Is pr.md written? (3) When is review_N.md created, and what does it hold while the lens agents are still working? (4) Batch 1 has returned eleven findings and the verifiers have not run yet. Do those findings go into review_N.md's Blockers and Concerns now? (5) The gauntlet has finished and phase 2 has not; someone opens pr.md and sees no Evidence section and no review line. Is that a defect? (6) Remediation round 1 has ended with one blocker still standing. Where and when is that attempt recorded?", + "assertions": [ + "Answer 1: before the first check, by a run that will reach phase 3; it holds the Summary and one Quality row per check of the batched five, each with its found, fixed, and surviving counts, plus any human call raised so far under Open calls - no Evidence, no review line", + "Answer 2: no - only a run that will reach phase 3 keeps a pr.md, so a partial run cannot overwrite the body an open PR was proposed with", + "Answer 3: when the panel is selected, at the next free index; Verdict: IN PROGRESS, the panel, the base, the brief's summary, and on a re-review the previous findings as open verification items", + "Answer 4: no - only verified findings reach the report; they are written when the aggregate script returns them, together with the verdict", + "Answer 5: no - a part the run has not reached is absent from the file, never a placeholder or a guess; Evidence is added in phase 3 and the review line when a verdict is set", + "Answer 6: under Open calls in pr.md, when the round ends - not composed from memory in phase 3" + ] } ] } diff --git a/dev/evals/tests/test_build_speed.py b/dev/evals/tests/test_build_speed.py index 1d36bf1..1dc5219 100644 --- a/dev/evals/tests/test_build_speed.py +++ b/dev/evals/tests/test_build_speed.py @@ -247,6 +247,53 @@ def test_named_test_that_is_not_in_the_file_still_fails(self) -> None: self.assertIn("contains no test named 'never written'", result.stdout) +class RunAndFixupEntryTest(PlanCase): + """The notes are written as the run goes, so they hold more than change set entries.""" + + RUN = "## Build run 2026-10-02\n- Validation: `npm test`, from the spec\n- Waves: [1]\n\n" + + def check(self, entry: str, fixup: str) -> subprocess.CompletedProcess[str]: + self.write_spec(change_set(1, "`src/a.ts`", 2)) + repo = self.plan.parent + (repo / "a.test.ts").write_text( + 'test("reads a coupon", () => {});\ntest("rejects an expired coupon", () => {});\n', + encoding="utf-8", + ) + self.write_notes( + f"{self.RUN}## Change set 1: Change set 1\n- Tests added: {entry}\n\n" + f"## Fixup: the built CLI printed no newline\n- Found by: e2e scenario 3\n- Tests added: {fixup}\n" + ) + return run(CHECK, str(self.plan), "--repo-root", str(repo)) + + def test_run_entry_and_fixup_entry_leave_a_complete_change_set_clean(self) -> None: + result = self.check( + "a.test.ts::reads a coupon, a.test.ts::rejects an expired coupon", "a.test.ts::reads a coupon" + ) + self.assertEqual(result.returncode, 0, result.stdout) + + def test_fixup_tests_do_not_count_toward_the_change_set_above(self) -> None: + result = self.check("a.test.ts::reads a coupon", "a.test.ts::rejects an expired coupon") + self.assertEqual(result.returncode, 1) + self.assertIn("change set 1: 2 scenario(s) specced, 1 test(s) named", result.stdout) + + def test_fixup_test_that_was_never_written_fails_naming_the_fixup(self) -> None: + result = self.check( + "a.test.ts::reads a coupon, a.test.ts::rejects an expired coupon", "a.test.ts::never written" + ) + self.assertEqual(result.returncode, 1) + self.assertIn("a fixup entry: a.test.ts contains no test named 'never written'", result.stdout) + + def test_brief_carries_the_fixup_to_later_change_sets(self) -> None: + self.write_spec(change_set(1, "`src/a.ts`"), change_set(2, "`src/b.ts`")) + self.write_notes( + f"{self.RUN}## Change set 1: Change set 1\n- What was done: it\n\n" + "## Fixup: the built CLI printed no newline\n- What was done: print one\n" + ) + out = run(BRIEF, str(self.plan), "2").stdout + self.assertIn("### Fixup: the built CLI printed no newline", out) + self.assertIn("- What was done: print one", out) + + class FactoryCopiesTest(unittest.TestCase): """The factory build and scope phases ship the same scripts.""" diff --git a/dev/references/ci-parity.md b/dev/references/ci-parity.md index 92157c9..6a28a54 100644 --- a/dev/references/ci-parity.md +++ b/dev/references/ci-parity.md @@ -15,7 +15,9 @@ Do not attempt to reproduce GitHub-owned setup actions locally. Reproduce the project command after performing its documented local setup. Prefer a repository-provided aggregate target when it covers the same jobs. -Record the commands and outcomes in the plan's `implementation-notes.md`. +Record each command and its outcome as it finishes, not after the whole set: +`build` in the plan's `implementation-notes.md`, `ship` on the CI parity line +of the `pr.md` its run keeps. ## Run before proposing a PR diff --git a/dev/references/plan-layout.md b/dev/references/plan-layout.md index 9276132..549b2bf 100644 --- a/dev/references/plan-layout.md +++ b/dev/references/plan-layout.md @@ -7,16 +7,27 @@ This reference owns the layout, the locating convention, and the diff scope; ski | File | Written by | Read by | | ---- | ---------- | ------- | -| `spec.md` | `scope`; `scope-review` (verified refinements only) | everyone downstream | -| `spec-review_N.md` | `scope-review` (next free index) | `scope` (remediation), re-reviews | +| `spec.md` | `scope` (from the interview on); `scope-review` (verified refinements only) | everyone downstream | +| `spec-review_N.md` | `scope-review` (next free index, opened before round 1) | `scope` (remediation), re-reviews | | `implementation-notes.md` | `build` (append-only) | `ship`, `to-pitch`, `to-quiz` | -| `review_N.md` | `ship` (next free index) | `scope` (remediation), re-reviews | -| `pr.md` | `ship` (phase 3, overwritten per run) | the PR tool via `--body-file`; re-runs | +| `review_N.md` | `ship` (next free index, opened when the panel is selected) | `scope` (remediation), re-reviews | +| `pr.md` | `ship` (opened at the start of a run that reaches phase 3, overwritten per run) | the PR tool via `--body-file`; re-runs | | `.dev/config.json` | the user | any skill with Jira behavior | Each producing skill also renders an HTML companion under `/tmp/{project-slug}/reports/` per [reporting.md](reporting.md), named by that skill. One writer per file; every other skill only reads. +## Written as the run goes + +The plan directory is a run's working state, not its closing summary. +A skill creates its file as soon as the run has something true to put in it, and every later fact - an answer, a decision, a finding, a tally, a verdict - reaches the file when it is established, before the run moves on. +A run that dies or compacts mid-stage then loses only what was in flight, the user can open the file at any moment to see where the run stands, and what the file says was recorded when it happened rather than recalled at the end. +Never hold content back to write it once when the stage closes, and never rebuild a file from memory of the conversation. +Each skill names the moments it writes at. + +A file its run has not finished says so itself: a report carries `Verdict: IN PROGRESS` until its verdict is set, a spec is unfinished until `lint-spec.py` exits clean, and a `pr.md` until `pr-evidence.py check` passes. +An unfinished file left by a run that is no longer going is a draft, never a result: its own skill picks it up where it stopped, at the same index, and no other skill builds on it. + ## Locating the plan directory Match the current branch name to a `.dev/{plan-name}/` slug; fall back to commit messages, then to the only directory in a plausible state for the skill (e.g. the only spec whose change sets aren't done). diff --git a/dev/skills/build/SKILL.md b/dev/skills/build/SKILL.md index ae5498e..f168889 100644 --- a/dev/skills/build/SKILL.md +++ b/dev/skills/build/SKILL.md @@ -16,7 +16,7 @@ Read [references/layers.md](references/layers.md), [references/tests.md](referen ## Workflow 1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. -2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. +2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. Append the run entry to `implementation-notes.md` before the first wave launches. 3. For each wave, run its change sets in parallel per the same reference, pass its wave gate once, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). 4. Move straight to the next wave. Never stop after one change set or wave to ask about review. 5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. @@ -82,15 +82,27 @@ E2E scenarios are proven by running the actual application against the **fully m ## Implementation notes -Maintain `.dev/{plan-name}/implementation-notes.md`, appended after each change set completes, never written once at the end. +Maintain `.dev/{plan-name}/implementation-notes.md` as the run goes, per [../../references/plan-layout.md](../../references/plan-layout.md), never written once at the end: the run entry before the first wave launches, a change set's entry when that change set completes, and a fixup entry each time the wave gate, the e2e loop, or the CI-parity gate makes you change code, written with the fix. It is the shared state across waves - parallel change-set agents can't see each other's conversation, only this file and the code - and the evidence `ship`, `to-pitch`, and `to-quiz` read later. ```markdown +## Build run {date} +- Validation: {the commands this run gates on, and where they came from} +- Waves: [1 2] [3] [4 5] + ## Change set {n}: {title} - What was done: ... - Seams tested: ... - Tests added: {path::test name}, ... # or "none - {reason}"; the checker reads this line - Deviations from spec: {edge case found} -> conservative choice made: {what/why} # only when a deviation occurred + +## Fixup: {what was wrong} +- Found by: {the wave gate | e2e scenario {id} | CI-parity command} +- What was done: ... +- Tests added: ... + +## CI parity +- {command} -> {green | red: what failed | remote-only: why} # one line as each command finishes ``` This file is a short running log, not a rendered report. diff --git a/dev/skills/build/scripts/check-tests.py b/dev/skills/build/scripts/check-tests.py index ccbb1aa..41c2b04 100755 --- a/dev/skills/build/scripts/check-tests.py +++ b/dev/skills/build/scripts/check-tests.py @@ -51,15 +51,22 @@ def spec_scenarios(spec: Path, problem) -> dict[int, int]: return counts -def note_entries(notes: Path) -> dict[int, list[str]]: - """Tests named per change set, from implementation-notes.md.""" +def note_entries(notes: Path) -> tuple[dict[int, list[str]], list[str]]: + """Tests named per change set, and those a fixup entry names, from implementation-notes.md. + + Any other ## heading ends a change set's entry, so a fixup's tests never + count toward the scenarios of the change set above it. + """ entries: dict[int, list[str]] = {} - current = None + fixups: list[str] = [] + current: list[str] | None = None for line in notes.read_text(encoding="utf-8").splitlines(): entry = NOTE_ENTRY.match(line) if entry: - current = int(entry.group(1)) - entries.setdefault(current, []) + current = entries.setdefault(int(entry.group(1)), []) + continue + if HEADING.match(line): + current = fixups continue named = NOTE_TESTS.match(line) if not named or current is None: @@ -67,25 +74,22 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) - return entries + current.extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) + return entries, fixups -def check_named_test(reference: str, repo_root: Path, change_set: int, problem) -> None: +def check_named_test(reference: str, repo_root: Path, entry: str, problem) -> None: """A named test must be path::name, and that name must be in that file.""" if "::" not in reference: - problem(f"change set {change_set}: '{reference}' is not in path::test name form") + problem(f"{entry}: '{reference}' is not in path::test name form") return path, _, name = reference.partition("::") target = repo_root / path.strip() if not target.is_file(): - problem(f"change set {change_set}: {path.strip()} does not exist") + problem(f"{entry}: {path.strip()} does not exist") return if name.strip() not in target.read_text(encoding="utf-8", errors="replace"): - problem( - f"change set {change_set}: {path.strip()} contains no test " - f"named '{name.strip()}'" - ) + problem(f"{entry}: {path.strip()} contains no test named '{name.strip()}'") def main(argv: list[str]) -> int: @@ -115,7 +119,7 @@ def problem(message: str) -> None: problems.append(message) counts = spec_scenarios(spec, problem) - entries = note_entries(notes) + entries, fixups = note_entries(notes) for change_set, scenarios in sorted(counts.items()): if change_set not in entries: @@ -128,11 +132,14 @@ def problem(message: str) -> None: f"{len(named)} test(s) named" ) for reference in named: - check_named_test(reference, repo_root, change_set, problem) + check_named_test(reference, repo_root, f"change set {change_set}", problem) for change_set in sorted(set(entries) - set(counts)): problem(f"change set {change_set} is in {notes} but not in the spec's change plan") + for reference in fixups: + check_named_test(reference, repo_root, "a fixup entry", problem) + for message in problems: print(message) if problems: diff --git a/dev/skills/scope-review/SKILL.md b/dev/skills/scope-review/SKILL.md index 52dac79..da282ee 100644 --- a/dev/skills/scope-review/SKILL.md +++ b/dev/skills/scope-review/SKILL.md @@ -11,7 +11,7 @@ A defect caught here costs a spec edit; the same defect after build costs a re-i The few findings only the user can decide are asked as questions at the end of the run, and the answers are applied before it finishes. The panel judges the plan against the actual repo, not against the conversation that produced it. You are the orchestrator: run tools, dispatch agents, apply the loop, report - your own reading of the spec is not a lens, and findings reach the spec only through verification. -This skill edits `spec.md`, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. +This skill edits `spec.md`, keeps its own report, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. `scope`'s own phase-5 reviewer hunts while the spec is still being drafted, from the spec file alone. This skill is the standalone deeper pass: fresh agents with repo access, adversarial verification, and automatic refinement - worth running when the change is large or risky, or when build will run in a different session. @@ -25,6 +25,7 @@ First run `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py star - Run `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` once. If it reports anything, stop and recommend finishing the `scope` run: refinement here presumes a mechanically settled spec, and repairing an unfinished draft is `scope`'s job, not this loop's. - Read the target repo's `docs/decisions.md`, `docs/contracts.md`, and `docs/dependencies.md` where they exist; their entries are premises the lenses cite. - `spec-review_N.md` files present -> unresolved escalations from the highest-numbered one become verification items for round 1, and new reports continue the numbering. +- Open this run's `spec-review_N.md` now - the next free index, or the one a dead run left `IN PROGRESS` - in the shape of step 7 with `Verdict: IN PROGRESS`, and write it as the run goes per [../../references/plan-layout.md](../../references/plan-layout.md): a round's counts and each refinement when its refine agent returns, each escalation when it is answered or deferred, each promotion when it lands. ## 2. The review-refine loop @@ -96,14 +97,15 @@ python3 {ship-skill-root}/scripts/aggregate-findings.py plan {batch-1 results} python3 {ship-skill-root}/scripts/aggregate-findings.py aggregate {batch-1 results} {batch-2 results} --expected {lenses} ``` -## 7. Write the report +## 7. The report -Write `.dev/{plan-name}/spec-review_N.md` at the next free index, one per run, covering all rounds: +`.dev/{plan-name}/spec-review_N.md` sits at the next free index, one per run, covering all rounds. +It has been filling since step 1, so by the wrap-up only its verdict is left to set: ```markdown # Spec review N - {plan-name} - {date} -Verdict: APPROVED | APPROVED WITH DEFERRALS +Verdict: IN PROGRESS | APPROVED | APPROVED WITH DEFERRALS Rounds: {R} - {finding counts per round} ## Refinements applied diff --git a/dev/skills/scope/SKILL.md b/dev/skills/scope/SKILL.md index 298c3bb..3dccad7 100644 --- a/dev/skills/scope/SKILL.md +++ b/dev/skills/scope/SKILL.md @@ -14,12 +14,16 @@ Before the interview, run `python3 {scope-skill-root}/../../scripts/skill-metric ## 1. Interview -Interview the user until you understand what they want, one consequential question at a time, using the host's structured user-input tool when available. Then write down what you understood. +Interview the user until you understand what they want, one consequential question at a time, using the host's structured user-input tool when available. Whenever you recommend an option - here, in the decision talk-through, or in follow-up questions - put a confidence score next to it (`Confidence: 70%`) with the one fact that would most change it. Score how likely it is the right solution given the evidence, not how strongly you prefer it: a taste call, an unverified premise, or an option chosen without having read the relevant code scores low. A recommendation at `Confidence: 75%` or above is not a question: apply it and keep going, and write `auto-applied at Confidence: NN%` into the chosen line's because clause so the ledger shows who decided. Ask only below 75%, and, whatever the score, when the recommendation changes what was asked for: a different problem than the request names, a requirement dropped, or a new non-goal. -At the end of the interview, list every auto-applied recommendation once, in one block, so a single reply can overturn any of them before the spec is written down. +At the end of the interview, list every auto-applied recommendation once, in one block, so a single reply can overturn any of them before the catalog builds on them. + +The spec opens with the interview, not after it: once the first answers say what the change is about, pick a kebab-case `{plan-name}` naming the outcome and create `spec.md` in the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)) with its title, the date, what you understood, and empty `## Research`, `## Scope`, and `## Change plan` sections. +From then on the file is the run's working state, written as the run goes per that reference: every answer, auto-applied recommendation, cataloged decision, and resolution reaches it when it happens, so the conversation never holds something the file lacks. +Rename the directory when the problem climbs to a different outcome. The request often arrives one level too low: a solution ("add rate limiting with Redis") hides the problem it solves, a symptom hides the cause that picks the fix. Climb up before interviewing about the change itself - what led to this, what they observed, what would look different if it worked - then pressure-test the premise: what data shows the problem is real, and does it point where they think? Committing to a fix before the cause means speccing the wrong change well. @@ -45,8 +49,9 @@ Look at the code, starting from the phase 1 explorers' results - prior art first - **Small** - timeout lengths, shapes of data structures. Pay special attention to edge cases and error handling. How the change gets verified is a decision too: what level to test at, what needs a real dependency versus a fake, what can't be tested and why. +Each decision enters the research section as an `[open]` entry the moment it is cataloged, and its marks change in the file as the talk-through settles it. -**Blind spot pass** (full-size only): don't grade your own catalog - you'll reread it the way you wrote it. Get a second one from something that hasn't seen your reasoning, reliably a subagent handed only the user's original request and the relevant code - not the conversation, not your catalog. +**Blind spot pass** (full-size only): don't grade your own catalog - you'll reread it the way you wrote it. Get a second one from something that hasn't seen your reasoning, reliably a subagent handed only the user's original request and the relevant code - not the conversation, not your catalog, and told to stay out of `.dev/`, where that catalog is being written. Both of its inputs are settled when the interview ends, so launch it before you start cataloging; it hunts while you do, and the fold-in is the sync point. Fold the diff in: what it found and you didn't are blind spots, what you found and it didn't deserves a second look. Then tell the user what their framing didn't account for: constraints already in the code, behavior the change would break, second-order work, and what a mature solution handles in this domain that they wouldn't know to ask about. @@ -56,7 +61,7 @@ Talk through the big and medium decisions with the user, highest-impact first; w ## 3. Research section -Pick a kebab-case `{plan-name}` naming the outcome and write `spec.md` in the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)), dated under its title, in three sections: `## Research`, `## Scope`, `## Change plan`. +`spec.md`, open since the interview, holds three sections under its dated title: `## Research`, `## Scope`, `## Change plan`. Research comes first, one entry per decision, ordered by impact. Every decision gets a stable ID: `D-` plus a short kebab slug (`D-file-storage`); keep a slug once assigned. diff --git a/dev/skills/ship/SKILL.md b/dev/skills/ship/SKILL.md index 720518d..f41e880 100644 --- a/dev/skills/ship/SKILL.md +++ b/dev/skills/ship/SKILL.md @@ -45,6 +45,7 @@ Never weaken or skip a check because acquiring its tool is work. 8. **Mutation testing** over the in-scope source - scored and reported in the wrap-up, never a reason on its own to block shipping (see gauntlet.md's "Mutation testing is reported, not a threshold"). Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). +A run that will reach phase 3 opens `.dev/{plan-name}/pr.md` before the first check and records each check's result in it as that check finishes, per the Body section of [references/pull-request.md](references/pull-request.md). When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. Before phase 2, always run the required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), even when no @@ -76,6 +77,7 @@ Read [references/lenses.md](references/lenses.md); select only lenses with surfa Include `spec-conformance` whenever a spec was found, and always include `simplify` - every diff has simplification surface. At most one diff-specific custom lens (migrations, concurrency, i18n) when clearly warranted, defined in the same shape as the built-ins. Tell the user which lenses you selected and why before launching. +Then open `.dev/{plan-name}/review_N.md` at the next free index, or the one a dead run left `IN PROGRESS`, in the shape of [references/report-format.md](references/report-format.md), with `Verdict: IN PROGRESS`, the panel, the base, the brief's summary, and on a re-review the previous findings as open verification items; a finding enters it only once verified. ### 4. Run the review panel @@ -102,9 +104,9 @@ A failed lens doesn't abort the review - the report names it, so the verdict is Then, with findings verified, read the target repo's `docs/decisions.md` if it keeps one, and classify each colliding finding per the recommender contract in [../../references/decision-ledger.md](../../references/decision-ledger.md). Read-only - this phase never edits the ledger; classifications land in the report's Decision reconciliation section, and ledger writes belong to the follow-up `scope` run. -### 6. Write the report +### 6. Complete the report -Write `.dev/{plan-name}/review_N.md` in the shape of [references/report-format.md](references/report-format.md), at the next free index. +`review_N.md` has been open since step 3: write the aggregated findings and the verdict into it as soon as the script returns them, and the Decision reconciliation section once its classifications are made - written as the run goes per [../../references/plan-layout.md](../../references/plan-layout.md), never composed in one pass at the end. ### 7. Publish the report @@ -122,7 +124,7 @@ Open the PR automatically, with evidence a reviewer can see before reading the d 1. Commit what the gauntlet left uncommitted, branch off the default branch if still on it, and push. 2. Build the Evidence section from the e2e report with `python3 {ship-skill-root}/scripts/pr-evidence.py extract`, publishing frontend screenshots to the `pr-evidence` branch with its `publish` command; without an e2e report, capture the evidence now per the reference - screenshots for a UI, a labeled before/after pair otherwise, and a red-on-base, green-on-branch reproducing test for every bug fix. -3. Write `.dev/{plan-name}/pr.md` in the reference's body shape and loop `pr-evidence.py check` on it until it passes; the check, not your judgment, decides whether the proof is real enough. +3. Add that section to the run's `.dev/{plan-name}/pr.md`, in the reference's body shape, and loop `pr-evidence.py check` on it until it passes; the check, not your judgment, decides whether the proof is real enough. 4. Create the PR (draft when a real blocker survived remediation) or update the one that already exists, then follow its required checks to green per [../../references/ci-parity.md](../../references/ci-parity.md). ## Wrap up diff --git a/dev/skills/ship/references/gauntlet.md b/dev/skills/ship/references/gauntlet.md index 7f79b6f..2e0980e 100644 --- a/dev/skills/ship/references/gauntlet.md +++ b/dev/skills/ship/references/gauntlet.md @@ -14,6 +14,7 @@ For each tool (the batched five count as one), in order: 2. Dispatch fixes: one fresh-context agent per independent area, launched in a single message, each given only the violation list for its area, the relevant file paths, and the fix vocabulary below. 3. Re-run the tool until clean, then run the spec's Validation block (or the repo's test suite) to prove the fixes broke nothing; skip that run when the tool dispatched no fixes. 4. A violation that resists two fix rounds on the same root cause, or that the change seems to legitimately require, is a human call: stop and present it - it is either a real defect, a threshold worth changing, or a rule the spec should have amended. +5. When the run keeps a `pr.md` ([pull-request.md](pull-request.md)), write the tool's Quality row - found, fixed, surviving, one row per check for the batched five - and any human call under Open calls before starting the next tool; the wrap-up reads its tallies from there. ## Exit: refresh the e2e evidence diff --git a/dev/skills/ship/references/pull-request.md b/dev/skills/ship/references/pull-request.md index 2ac4295..c4023e7 100644 --- a/dev/skills/ship/references/pull-request.md +++ b/dev/skills/ship/references/pull-request.md @@ -37,7 +37,9 @@ On a non-GitHub remote pass `--url-template` with wherever the images are hosted ## Body -Write the body to `.dev/{plan-name}/pr.md` and hand it to the PR tool with `--body-file`; the file stays as the record of what was proposed. +The body lives at `.dev/{plan-name}/pr.md` and is handed to the PR tool with `--body-file`; the file stays as the record of what was proposed. +A run that will reach this phase opens the file in the shape below before its first check and fills it as the run goes, per [../../../references/plan-layout.md](../../../references/plan-layout.md): the Summary at once, a Quality row when its check finishes, the CI parity line as each command returns, the review line when a verdict is set, an Open call when it is raised and each remediation round's attempt when the round ends. +Only the Evidence section is left for this phase; a part the run has not reached is absent from the file, never a placeholder or a guess. ```markdown ## Summary diff --git a/dev/skills/ship/references/remediation.md b/dev/skills/ship/references/remediation.md index 37fbce6..d0a3639 100644 --- a/dev/skills/ship/references/remediation.md +++ b/dev/skills/ship/references/remediation.md @@ -38,5 +38,6 @@ A re-run of `ship` after the human call starts a fresh budget. ## Real blockers A real blocker is listed in the PR under Open calls with its history: the finding, the round-1 and round-2 attempts (commit and what each changed), and why it still stands - unfixable in scope, escalated, or a new blocker the fixes introduced. +Each attempt goes into `pr.md` when its round ends, so the history is a record of the rounds and not a recollection of them. Phase 3 opens the PR as a draft. The wrap-up presents each real blocker as the human call it is: a real defect the fixes could not reach, a decision worth reopening, or a spec the change outgrew - the same three outcomes the gauntlet's human calls have. diff --git a/dev/skills/ship/references/report-format.md b/dev/skills/ship/references/report-format.md index 2f990bf..f342893 100644 --- a/dev/skills/ship/references/report-format.md +++ b/dev/skills/ship/references/report-format.md @@ -1,12 +1,13 @@ # Review Report Format The shape of `.dev/{plan-name}/review_N.md`, written by phase 2 at the next free index starting at 1. +Phase 2 opens it when the panel is selected, holding `Verdict: IN PROGRESS` and only the sections it can already fill, and completes it as the aggregator and the reconciliation return. `REVIEW_DATA` ([data-schema.md](data-schema.md)) mirrors this file section for section; the aggregator's JSON fills both. ```markdown # Review {N}: {title} -Verdict: {PASS | CONCERNS | BLOCK} +Verdict: {IN PROGRESS | PASS | CONCERNS | BLOCK} Panel: {lenses run}, {failed lenses if any} Base: {branch or PR}, {date} diff --git a/docs/decisions.md b/docs/decisions.md index 8e81597..1eb845d 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -331,6 +331,20 @@ D-change-set-brief: What does a change-set agent read before it starts? (2026-10 ✗ the whole spec and the notes - every agent pays for every other change set's plan ✗ a summary the orchestrator writes per agent - output tokens are the slow ones, and a paraphrase can be wrong +## Plan files + +D-written-as-the-run-goes: When does a skill write its file under `.dev/{plan-name}/`? (2026-10-02, user feedback) + ✓ as the run goes: the file is created as soon as the run has something true to put in it, and each later fact reaches it when it is established - `scope` opens `spec.md` during the interview and catalogs decisions into it as `[open]` entries, `scope-review` opens `spec-review_N.md` before round 1, `build` adds a run entry and fixup entries to the notes, `ship` opens `review_N.md` when the panel is selected and `pr.md` before the first check; the skills wrote at the end of a stage, and continuously updated files were asked for (user, 2026-10-02); the rule lives once in `plan-layout.md` and each skill names its moments ⚠ a spec, a report, or a PR body can now be found half-written: `Verdict: IN PROGRESS`, a `lint-spec.py` that is not clean, and a failing `pr-evidence.py check` mark those, and no other skill builds on one + ✗ keep writing when the stage closes - a two-hour run holds its result only in the conversation, so a dead or compacted session loses it, nobody can follow the run from its files, and the file is a recollection instead of a record + ⊘ a checkpoint command that fails when a file did not change between two phases - not doing, against P-deterministic-guards-over-prose: the late write was what the skill text itself ordered (`scope` phase 3, `scope-review` step 7, `ship` phase 2 step 6), so moving the instruction removes the cause; reopen if runs under the new text still write at the end of a stage +D-ship-record-in-pr-body: Where does ship record its gauntlet results while it runs? (2026-10-02, follows D-written-as-the-run-goes) + ✓ in `pr.md`, opened before the first check by a run that will reach phase 3 - its Quality table and Open calls already are that record, so the file fills check by check instead of being composed in phase 3; a part not reached yet is absent, never a placeholder ⚠ a gauntlet-only, review-only, or no-PR run still keeps its tallies only in the conversation + ✗ `pr.md` in every mode - a partial run would overwrite the record of the body an open PR was proposed with + ✗ a new gauntlet log file - a second artifact and a second reader for what the PR body already holds; reconsider if partial runs need a durable record +D-notes-fixup-entry: How do the notes record code changed outside a change set's own loop? (2026-10-02, follows D-written-as-the-run-goes) + ✓ a `## Fixup:` entry written with the fix, naming what found it, and `check-tests.py` ends a change set's entry at any other `##` heading - builds already wrote such entries unasked ("Orchestrator fixup after wave 9" in the contexia github-sync-teams notes, read 2026-10-02), and a fixup's `Tests added:` line counted toward the scenarios of the change set above it (reproduced by a unit test, 2026-10-02); fixup tests are still checked to exist + ✗ a Deviations line on the nearest change set - that change set is committed and its entry says what its agent did + ## Bootstrap plugin D-bootstrap-plugin: Where does the skill that writes a repository's AGENTS.md live? (2026-09-25, user request) diff --git a/factory/phases/build/scripts/check-tests.py b/factory/phases/build/scripts/check-tests.py index ccbb1aa..41c2b04 100755 --- a/factory/phases/build/scripts/check-tests.py +++ b/factory/phases/build/scripts/check-tests.py @@ -51,15 +51,22 @@ def spec_scenarios(spec: Path, problem) -> dict[int, int]: return counts -def note_entries(notes: Path) -> dict[int, list[str]]: - """Tests named per change set, from implementation-notes.md.""" +def note_entries(notes: Path) -> tuple[dict[int, list[str]], list[str]]: + """Tests named per change set, and those a fixup entry names, from implementation-notes.md. + + Any other ## heading ends a change set's entry, so a fixup's tests never + count toward the scenarios of the change set above it. + """ entries: dict[int, list[str]] = {} - current = None + fixups: list[str] = [] + current: list[str] | None = None for line in notes.read_text(encoding="utf-8").splitlines(): entry = NOTE_ENTRY.match(line) if entry: - current = int(entry.group(1)) - entries.setdefault(current, []) + current = entries.setdefault(int(entry.group(1)), []) + continue + if HEADING.match(line): + current = fixups continue named = NOTE_TESTS.match(line) if not named or current is None: @@ -67,25 +74,22 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) - return entries + current.extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) + return entries, fixups -def check_named_test(reference: str, repo_root: Path, change_set: int, problem) -> None: +def check_named_test(reference: str, repo_root: Path, entry: str, problem) -> None: """A named test must be path::name, and that name must be in that file.""" if "::" not in reference: - problem(f"change set {change_set}: '{reference}' is not in path::test name form") + problem(f"{entry}: '{reference}' is not in path::test name form") return path, _, name = reference.partition("::") target = repo_root / path.strip() if not target.is_file(): - problem(f"change set {change_set}: {path.strip()} does not exist") + problem(f"{entry}: {path.strip()} does not exist") return if name.strip() not in target.read_text(encoding="utf-8", errors="replace"): - problem( - f"change set {change_set}: {path.strip()} contains no test " - f"named '{name.strip()}'" - ) + problem(f"{entry}: {path.strip()} contains no test named '{name.strip()}'") def main(argv: list[str]) -> int: @@ -115,7 +119,7 @@ def problem(message: str) -> None: problems.append(message) counts = spec_scenarios(spec, problem) - entries = note_entries(notes) + entries, fixups = note_entries(notes) for change_set, scenarios in sorted(counts.items()): if change_set not in entries: @@ -128,11 +132,14 @@ def problem(message: str) -> None: f"{len(named)} test(s) named" ) for reference in named: - check_named_test(reference, repo_root, change_set, problem) + check_named_test(reference, repo_root, f"change set {change_set}", problem) for change_set in sorted(set(entries) - set(counts)): problem(f"change set {change_set} is in {notes} but not in the spec's change plan") + for reference in fixups: + check_named_test(reference, repo_root, "a fixup entry", problem) + for message in problems: print(message) if problems: diff --git a/package.json b/package.json index bed5e4f..633e697 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@tobrun/dev-workflow", - "version": "3.4.0", + "version": "3.5.0", "description": "Test-focused development workflow skills for Claude Code, Codex, and Pi.", "keywords": [ "pi-package", diff --git a/plugins/dev/.codex-plugin/plugin.json b/plugins/dev/.codex-plugin/plugin.json index 649ec36..bda2dd5 100644 --- a/plugins/dev/.codex-plugin/plugin.json +++ b/plugins/dev/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dev", - "version": "3.4.0", + "version": "3.5.0", "description": "Development workflow skills: scope changes with argued decisions, build across unit/integration/e2e with every scenario proven by tests, ship with a deterministic quality gauntlet and an adversarially verified review, create structured commits that feed a decision ledger, and render pitches or comprehension quizzes.", "author": { "name": "Tobrun" diff --git a/plugins/dev/references/ci-parity.md b/plugins/dev/references/ci-parity.md index 92157c9..6a28a54 100644 --- a/plugins/dev/references/ci-parity.md +++ b/plugins/dev/references/ci-parity.md @@ -15,7 +15,9 @@ Do not attempt to reproduce GitHub-owned setup actions locally. Reproduce the project command after performing its documented local setup. Prefer a repository-provided aggregate target when it covers the same jobs. -Record the commands and outcomes in the plan's `implementation-notes.md`. +Record each command and its outcome as it finishes, not after the whole set: +`build` in the plan's `implementation-notes.md`, `ship` on the CI parity line +of the `pr.md` its run keeps. ## Run before proposing a PR diff --git a/plugins/dev/references/plan-layout.md b/plugins/dev/references/plan-layout.md index 9276132..549b2bf 100644 --- a/plugins/dev/references/plan-layout.md +++ b/plugins/dev/references/plan-layout.md @@ -7,16 +7,27 @@ This reference owns the layout, the locating convention, and the diff scope; ski | File | Written by | Read by | | ---- | ---------- | ------- | -| `spec.md` | `scope`; `scope-review` (verified refinements only) | everyone downstream | -| `spec-review_N.md` | `scope-review` (next free index) | `scope` (remediation), re-reviews | +| `spec.md` | `scope` (from the interview on); `scope-review` (verified refinements only) | everyone downstream | +| `spec-review_N.md` | `scope-review` (next free index, opened before round 1) | `scope` (remediation), re-reviews | | `implementation-notes.md` | `build` (append-only) | `ship`, `to-pitch`, `to-quiz` | -| `review_N.md` | `ship` (next free index) | `scope` (remediation), re-reviews | -| `pr.md` | `ship` (phase 3, overwritten per run) | the PR tool via `--body-file`; re-runs | +| `review_N.md` | `ship` (next free index, opened when the panel is selected) | `scope` (remediation), re-reviews | +| `pr.md` | `ship` (opened at the start of a run that reaches phase 3, overwritten per run) | the PR tool via `--body-file`; re-runs | | `.dev/config.json` | the user | any skill with Jira behavior | Each producing skill also renders an HTML companion under `/tmp/{project-slug}/reports/` per [reporting.md](reporting.md), named by that skill. One writer per file; every other skill only reads. +## Written as the run goes + +The plan directory is a run's working state, not its closing summary. +A skill creates its file as soon as the run has something true to put in it, and every later fact - an answer, a decision, a finding, a tally, a verdict - reaches the file when it is established, before the run moves on. +A run that dies or compacts mid-stage then loses only what was in flight, the user can open the file at any moment to see where the run stands, and what the file says was recorded when it happened rather than recalled at the end. +Never hold content back to write it once when the stage closes, and never rebuild a file from memory of the conversation. +Each skill names the moments it writes at. + +A file its run has not finished says so itself: a report carries `Verdict: IN PROGRESS` until its verdict is set, a spec is unfinished until `lint-spec.py` exits clean, and a `pr.md` until `pr-evidence.py check` passes. +An unfinished file left by a run that is no longer going is a draft, never a result: its own skill picks it up where it stopped, at the same index, and no other skill builds on it. + ## Locating the plan directory Match the current branch name to a `.dev/{plan-name}/` slug; fall back to commit messages, then to the only directory in a plausible state for the skill (e.g. the only spec whose change sets aren't done). diff --git a/plugins/dev/skills/build/SKILL.md b/plugins/dev/skills/build/SKILL.md index 7466a66..9224d76 100644 --- a/plugins/dev/skills/build/SKILL.md +++ b/plugins/dev/skills/build/SKILL.md @@ -15,7 +15,7 @@ Read [references/layers.md](references/layers.md), [references/tests.md](referen ## Workflow 1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. -2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. +2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. Append the run entry to `implementation-notes.md` before the first wave launches. 3. For each wave, run its change sets in parallel per the same reference, pass its wave gate once, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). 4. Move straight to the next wave. Never stop after one change set or wave to ask about review. 5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. @@ -81,15 +81,27 @@ E2E scenarios are proven by running the actual application against the **fully m ## Implementation notes -Maintain `.dev/{plan-name}/implementation-notes.md`, appended after each change set completes, never written once at the end. +Maintain `.dev/{plan-name}/implementation-notes.md` as the run goes, per [../../references/plan-layout.md](../../references/plan-layout.md), never written once at the end: the run entry before the first wave launches, a change set's entry when that change set completes, and a fixup entry each time the wave gate, the e2e loop, or the CI-parity gate makes you change code, written with the fix. It is the shared state across waves - parallel change-set agents can't see each other's conversation, only this file and the code - and the evidence `ship`, `to-pitch`, and `to-quiz` read later. ```markdown +## Build run {date} +- Validation: {the commands this run gates on, and where they came from} +- Waves: [1 2] [3] [4 5] + ## Change set {n}: {title} - What was done: ... - Seams tested: ... - Tests added: {path::test name}, ... # or "none - {reason}"; the checker reads this line - Deviations from spec: {edge case found} -> conservative choice made: {what/why} # only when a deviation occurred + +## Fixup: {what was wrong} +- Found by: {the wave gate | e2e scenario {id} | CI-parity command} +- What was done: ... +- Tests added: ... + +## CI parity +- {command} -> {green | red: what failed | remote-only: why} # one line as each command finishes ``` This file is a short running log, not a rendered report. diff --git a/plugins/dev/skills/build/scripts/check-tests.py b/plugins/dev/skills/build/scripts/check-tests.py index ccbb1aa..41c2b04 100755 --- a/plugins/dev/skills/build/scripts/check-tests.py +++ b/plugins/dev/skills/build/scripts/check-tests.py @@ -51,15 +51,22 @@ def spec_scenarios(spec: Path, problem) -> dict[int, int]: return counts -def note_entries(notes: Path) -> dict[int, list[str]]: - """Tests named per change set, from implementation-notes.md.""" +def note_entries(notes: Path) -> tuple[dict[int, list[str]], list[str]]: + """Tests named per change set, and those a fixup entry names, from implementation-notes.md. + + Any other ## heading ends a change set's entry, so a fixup's tests never + count toward the scenarios of the change set above it. + """ entries: dict[int, list[str]] = {} - current = None + fixups: list[str] = [] + current: list[str] | None = None for line in notes.read_text(encoding="utf-8").splitlines(): entry = NOTE_ENTRY.match(line) if entry: - current = int(entry.group(1)) - entries.setdefault(current, []) + current = entries.setdefault(int(entry.group(1)), []) + continue + if HEADING.match(line): + current = fixups continue named = NOTE_TESTS.match(line) if not named or current is None: @@ -67,25 +74,22 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) - return entries + current.extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) + return entries, fixups -def check_named_test(reference: str, repo_root: Path, change_set: int, problem) -> None: +def check_named_test(reference: str, repo_root: Path, entry: str, problem) -> None: """A named test must be path::name, and that name must be in that file.""" if "::" not in reference: - problem(f"change set {change_set}: '{reference}' is not in path::test name form") + problem(f"{entry}: '{reference}' is not in path::test name form") return path, _, name = reference.partition("::") target = repo_root / path.strip() if not target.is_file(): - problem(f"change set {change_set}: {path.strip()} does not exist") + problem(f"{entry}: {path.strip()} does not exist") return if name.strip() not in target.read_text(encoding="utf-8", errors="replace"): - problem( - f"change set {change_set}: {path.strip()} contains no test " - f"named '{name.strip()}'" - ) + problem(f"{entry}: {path.strip()} contains no test named '{name.strip()}'") def main(argv: list[str]) -> int: @@ -115,7 +119,7 @@ def problem(message: str) -> None: problems.append(message) counts = spec_scenarios(spec, problem) - entries = note_entries(notes) + entries, fixups = note_entries(notes) for change_set, scenarios in sorted(counts.items()): if change_set not in entries: @@ -128,11 +132,14 @@ def problem(message: str) -> None: f"{len(named)} test(s) named" ) for reference in named: - check_named_test(reference, repo_root, change_set, problem) + check_named_test(reference, repo_root, f"change set {change_set}", problem) for change_set in sorted(set(entries) - set(counts)): problem(f"change set {change_set} is in {notes} but not in the spec's change plan") + for reference in fixups: + check_named_test(reference, repo_root, "a fixup entry", problem) + for message in problems: print(message) if problems: diff --git a/plugins/dev/skills/scope-review/SKILL.md b/plugins/dev/skills/scope-review/SKILL.md index 0997d4c..55c2078 100644 --- a/plugins/dev/skills/scope-review/SKILL.md +++ b/plugins/dev/skills/scope-review/SKILL.md @@ -10,7 +10,7 @@ A defect caught here costs a spec edit; the same defect after build costs a re-i The few findings only the user can decide are asked as questions at the end of the run, and the answers are applied before it finishes. The panel judges the plan against the actual repo, not against the conversation that produced it. You are the orchestrator: run tools, dispatch agents, apply the loop, report - your own reading of the spec is not a lens, and findings reach the spec only through verification. -This skill edits `spec.md`, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. +This skill edits `spec.md`, keeps its own report, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. `scope`'s own phase-5 reviewer hunts while the spec is still being drafted, from the spec file alone. This skill is the standalone deeper pass: fresh agents with repo access, adversarial verification, and automatic refinement - worth running when the change is large or risky, or when build will run in a different session. @@ -24,6 +24,7 @@ First run `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py star - Run `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` once. If it reports anything, stop and recommend finishing the `scope` run: refinement here presumes a mechanically settled spec, and repairing an unfinished draft is `scope`'s job, not this loop's. - Read the target repo's `docs/decisions.md`, `docs/contracts.md`, and `docs/dependencies.md` where they exist; their entries are premises the lenses cite. - `spec-review_N.md` files present -> unresolved escalations from the highest-numbered one become verification items for round 1, and new reports continue the numbering. +- Open this run's `spec-review_N.md` now - the next free index, or the one a dead run left `IN PROGRESS` - in the shape of step 7 with `Verdict: IN PROGRESS`, and write it as the run goes per [../../references/plan-layout.md](../../references/plan-layout.md): a round's counts and each refinement when its refine agent returns, each escalation when it is answered or deferred, each promotion when it lands. ## 2. The review-refine loop @@ -95,14 +96,15 @@ python3 {ship-skill-root}/scripts/aggregate-findings.py plan {batch-1 results} python3 {ship-skill-root}/scripts/aggregate-findings.py aggregate {batch-1 results} {batch-2 results} --expected {lenses} ``` -## 7. Write the report +## 7. The report -Write `.dev/{plan-name}/spec-review_N.md` at the next free index, one per run, covering all rounds: +`.dev/{plan-name}/spec-review_N.md` sits at the next free index, one per run, covering all rounds. +It has been filling since step 1, so by the wrap-up only its verdict is left to set: ```markdown # Spec review N - {plan-name} - {date} -Verdict: APPROVED | APPROVED WITH DEFERRALS +Verdict: IN PROGRESS | APPROVED | APPROVED WITH DEFERRALS Rounds: {R} - {finding counts per round} ## Refinements applied diff --git a/plugins/dev/skills/scope/SKILL.md b/plugins/dev/skills/scope/SKILL.md index 8767cde..c35f513 100644 --- a/plugins/dev/skills/scope/SKILL.md +++ b/plugins/dev/skills/scope/SKILL.md @@ -13,12 +13,16 @@ Before the interview, run `python3 {scope-skill-root}/../../scripts/skill-metric ## 1. Interview -Interview the user until you understand what they want, one consequential question at a time, using the host's structured user-input tool when available. Then write down what you understood. +Interview the user until you understand what they want, one consequential question at a time, using the host's structured user-input tool when available. Whenever you recommend an option - here, in the decision talk-through, or in follow-up questions - put a confidence score next to it (`Confidence: 70%`) with the one fact that would most change it. Score how likely it is the right solution given the evidence, not how strongly you prefer it: a taste call, an unverified premise, or an option chosen without having read the relevant code scores low. A recommendation at `Confidence: 75%` or above is not a question: apply it and keep going, and write `auto-applied at Confidence: NN%` into the chosen line's because clause so the ledger shows who decided. Ask only below 75%, and, whatever the score, when the recommendation changes what was asked for: a different problem than the request names, a requirement dropped, or a new non-goal. -At the end of the interview, list every auto-applied recommendation once, in one block, so a single reply can overturn any of them before the spec is written down. +At the end of the interview, list every auto-applied recommendation once, in one block, so a single reply can overturn any of them before the catalog builds on them. + +The spec opens with the interview, not after it: once the first answers say what the change is about, pick a kebab-case `{plan-name}` naming the outcome and create `spec.md` in the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)) with its title, the date, what you understood, and empty `## Research`, `## Scope`, and `## Change plan` sections. +From then on the file is the run's working state, written as the run goes per that reference: every answer, auto-applied recommendation, cataloged decision, and resolution reaches it when it happens, so the conversation never holds something the file lacks. +Rename the directory when the problem climbs to a different outcome. The request often arrives one level too low: a solution ("add rate limiting with Redis") hides the problem it solves, a symptom hides the cause that picks the fix. Climb up before interviewing about the change itself - what led to this, what they observed, what would look different if it worked - then pressure-test the premise: what data shows the problem is real, and does it point where they think? Committing to a fix before the cause means speccing the wrong change well. @@ -44,8 +48,9 @@ Look at the code, starting from the phase 1 explorers' results - prior art first - **Small** - timeout lengths, shapes of data structures. Pay special attention to edge cases and error handling. How the change gets verified is a decision too: what level to test at, what needs a real dependency versus a fake, what can't be tested and why. +Each decision enters the research section as an `[open]` entry the moment it is cataloged, and its marks change in the file as the talk-through settles it. -**Blind spot pass** (full-size only): don't grade your own catalog - you'll reread it the way you wrote it. Get a second one from something that hasn't seen your reasoning, reliably a subagent handed only the user's original request and the relevant code - not the conversation, not your catalog. +**Blind spot pass** (full-size only): don't grade your own catalog - you'll reread it the way you wrote it. Get a second one from something that hasn't seen your reasoning, reliably a subagent handed only the user's original request and the relevant code - not the conversation, not your catalog, and told to stay out of `.dev/`, where that catalog is being written. Both of its inputs are settled when the interview ends, so launch it before you start cataloging; it hunts while you do, and the fold-in is the sync point. Fold the diff in: what it found and you didn't are blind spots, what you found and it didn't deserves a second look. Then tell the user what their framing didn't account for: constraints already in the code, behavior the change would break, second-order work, and what a mature solution handles in this domain that they wouldn't know to ask about. @@ -55,7 +60,7 @@ Talk through the big and medium decisions with the user, highest-impact first; w ## 3. Research section -Pick a kebab-case `{plan-name}` naming the outcome and write `spec.md` in the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)), dated under its title, in three sections: `## Research`, `## Scope`, `## Change plan`. +`spec.md`, open since the interview, holds three sections under its dated title: `## Research`, `## Scope`, `## Change plan`. Research comes first, one entry per decision, ordered by impact. Every decision gets a stable ID: `D-` plus a short kebab slug (`D-file-storage`); keep a slug once assigned. diff --git a/plugins/dev/skills/ship/SKILL.md b/plugins/dev/skills/ship/SKILL.md index 9d43057..61c690d 100644 --- a/plugins/dev/skills/ship/SKILL.md +++ b/plugins/dev/skills/ship/SKILL.md @@ -44,6 +44,7 @@ Never weaken or skip a check because acquiring its tool is work. 8. **Mutation testing** over the in-scope source - scored and reported in the wrap-up, never a reason on its own to block shipping (see gauntlet.md's "Mutation testing is reported, not a threshold"). Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). +A run that will reach phase 3 opens `.dev/{plan-name}/pr.md` before the first check and records each check's result in it as that check finishes, per the Body section of [references/pull-request.md](references/pull-request.md). When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. Before phase 2, always run the required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), even when no @@ -75,6 +76,7 @@ Read [references/lenses.md](references/lenses.md); select only lenses with surfa Include `spec-conformance` whenever a spec was found, and always include `simplify` - every diff has simplification surface. At most one diff-specific custom lens (migrations, concurrency, i18n) when clearly warranted, defined in the same shape as the built-ins. Tell the user which lenses you selected and why before launching. +Then open `.dev/{plan-name}/review_N.md` at the next free index, or the one a dead run left `IN PROGRESS`, in the shape of [references/report-format.md](references/report-format.md), with `Verdict: IN PROGRESS`, the panel, the base, the brief's summary, and on a re-review the previous findings as open verification items; a finding enters it only once verified. ### 4. Run the review panel @@ -101,9 +103,9 @@ A failed lens doesn't abort the review - the report names it, so the verdict is Then, with findings verified, read the target repo's `docs/decisions.md` if it keeps one, and classify each colliding finding per the recommender contract in [../../references/decision-ledger.md](../../references/decision-ledger.md). Read-only - this phase never edits the ledger; classifications land in the report's Decision reconciliation section, and ledger writes belong to the follow-up `scope` run. -### 6. Write the report +### 6. Complete the report -Write `.dev/{plan-name}/review_N.md` in the shape of [references/report-format.md](references/report-format.md), at the next free index. +`review_N.md` has been open since step 3: write the aggregated findings and the verdict into it as soon as the script returns them, and the Decision reconciliation section once its classifications are made - written as the run goes per [../../references/plan-layout.md](../../references/plan-layout.md), never composed in one pass at the end. ### 7. Publish the report @@ -121,7 +123,7 @@ Open the PR automatically, with evidence a reviewer can see before reading the d 1. Commit what the gauntlet left uncommitted, branch off the default branch if still on it, and push. 2. Build the Evidence section from the e2e report with `python3 {ship-skill-root}/scripts/pr-evidence.py extract`, publishing frontend screenshots to the `pr-evidence` branch with its `publish` command; without an e2e report, capture the evidence now per the reference - screenshots for a UI, a labeled before/after pair otherwise, and a red-on-base, green-on-branch reproducing test for every bug fix. -3. Write `.dev/{plan-name}/pr.md` in the reference's body shape and loop `pr-evidence.py check` on it until it passes; the check, not your judgment, decides whether the proof is real enough. +3. Add that section to the run's `.dev/{plan-name}/pr.md`, in the reference's body shape, and loop `pr-evidence.py check` on it until it passes; the check, not your judgment, decides whether the proof is real enough. 4. Create the PR (draft when a real blocker survived remediation) or update the one that already exists, then follow its required checks to green per [../../references/ci-parity.md](../../references/ci-parity.md). ## Wrap up diff --git a/plugins/dev/skills/ship/references/gauntlet.md b/plugins/dev/skills/ship/references/gauntlet.md index 7f79b6f..2e0980e 100644 --- a/plugins/dev/skills/ship/references/gauntlet.md +++ b/plugins/dev/skills/ship/references/gauntlet.md @@ -14,6 +14,7 @@ For each tool (the batched five count as one), in order: 2. Dispatch fixes: one fresh-context agent per independent area, launched in a single message, each given only the violation list for its area, the relevant file paths, and the fix vocabulary below. 3. Re-run the tool until clean, then run the spec's Validation block (or the repo's test suite) to prove the fixes broke nothing; skip that run when the tool dispatched no fixes. 4. A violation that resists two fix rounds on the same root cause, or that the change seems to legitimately require, is a human call: stop and present it - it is either a real defect, a threshold worth changing, or a rule the spec should have amended. +5. When the run keeps a `pr.md` ([pull-request.md](pull-request.md)), write the tool's Quality row - found, fixed, surviving, one row per check for the batched five - and any human call under Open calls before starting the next tool; the wrap-up reads its tallies from there. ## Exit: refresh the e2e evidence diff --git a/plugins/dev/skills/ship/references/pull-request.md b/plugins/dev/skills/ship/references/pull-request.md index 2ac4295..c4023e7 100644 --- a/plugins/dev/skills/ship/references/pull-request.md +++ b/plugins/dev/skills/ship/references/pull-request.md @@ -37,7 +37,9 @@ On a non-GitHub remote pass `--url-template` with wherever the images are hosted ## Body -Write the body to `.dev/{plan-name}/pr.md` and hand it to the PR tool with `--body-file`; the file stays as the record of what was proposed. +The body lives at `.dev/{plan-name}/pr.md` and is handed to the PR tool with `--body-file`; the file stays as the record of what was proposed. +A run that will reach this phase opens the file in the shape below before its first check and fills it as the run goes, per [../../../references/plan-layout.md](../../../references/plan-layout.md): the Summary at once, a Quality row when its check finishes, the CI parity line as each command returns, the review line when a verdict is set, an Open call when it is raised and each remediation round's attempt when the round ends. +Only the Evidence section is left for this phase; a part the run has not reached is absent from the file, never a placeholder or a guess. ```markdown ## Summary diff --git a/plugins/dev/skills/ship/references/remediation.md b/plugins/dev/skills/ship/references/remediation.md index 37fbce6..d0a3639 100644 --- a/plugins/dev/skills/ship/references/remediation.md +++ b/plugins/dev/skills/ship/references/remediation.md @@ -38,5 +38,6 @@ A re-run of `ship` after the human call starts a fresh budget. ## Real blockers A real blocker is listed in the PR under Open calls with its history: the finding, the round-1 and round-2 attempts (commit and what each changed), and why it still stands - unfixable in scope, escalated, or a new blocker the fixes introduced. +Each attempt goes into `pr.md` when its round ends, so the history is a record of the rounds and not a recollection of them. Phase 3 opens the PR as a draft. The wrap-up presents each real blocker as the human call it is: a real defect the fixes could not reach, a decision worth reopening, or a spec the change outgrew - the same three outcomes the gauntlet's human calls have. diff --git a/plugins/dev/skills/ship/references/report-format.md b/plugins/dev/skills/ship/references/report-format.md index 2f990bf..f342893 100644 --- a/plugins/dev/skills/ship/references/report-format.md +++ b/plugins/dev/skills/ship/references/report-format.md @@ -1,12 +1,13 @@ # Review Report Format The shape of `.dev/{plan-name}/review_N.md`, written by phase 2 at the next free index starting at 1. +Phase 2 opens it when the panel is selected, holding `Verdict: IN PROGRESS` and only the sections it can already fill, and completes it as the aggregator and the reconciliation return. `REVIEW_DATA` ([data-schema.md](data-schema.md)) mirrors this file section for section; the aggregator's JSON fills both. ```markdown # Review {N}: {title} -Verdict: {PASS | CONCERNS | BLOCK} +Verdict: {IN PROGRESS | PASS | CONCERNS | BLOCK} Panel: {lenses run}, {failed lenses if any} Base: {branch or PR}, {date} diff --git a/plugins/factory/phases/build/scripts/check-tests.py b/plugins/factory/phases/build/scripts/check-tests.py index ccbb1aa..41c2b04 100755 --- a/plugins/factory/phases/build/scripts/check-tests.py +++ b/plugins/factory/phases/build/scripts/check-tests.py @@ -51,15 +51,22 @@ def spec_scenarios(spec: Path, problem) -> dict[int, int]: return counts -def note_entries(notes: Path) -> dict[int, list[str]]: - """Tests named per change set, from implementation-notes.md.""" +def note_entries(notes: Path) -> tuple[dict[int, list[str]], list[str]]: + """Tests named per change set, and those a fixup entry names, from implementation-notes.md. + + Any other ## heading ends a change set's entry, so a fixup's tests never + count toward the scenarios of the change set above it. + """ entries: dict[int, list[str]] = {} - current = None + fixups: list[str] = [] + current: list[str] | None = None for line in notes.read_text(encoding="utf-8").splitlines(): entry = NOTE_ENTRY.match(line) if entry: - current = int(entry.group(1)) - entries.setdefault(current, []) + current = entries.setdefault(int(entry.group(1)), []) + continue + if HEADING.match(line): + current = fixups continue named = NOTE_TESTS.match(line) if not named or current is None: @@ -67,25 +74,22 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) - return entries + current.extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) + return entries, fixups -def check_named_test(reference: str, repo_root: Path, change_set: int, problem) -> None: +def check_named_test(reference: str, repo_root: Path, entry: str, problem) -> None: """A named test must be path::name, and that name must be in that file.""" if "::" not in reference: - problem(f"change set {change_set}: '{reference}' is not in path::test name form") + problem(f"{entry}: '{reference}' is not in path::test name form") return path, _, name = reference.partition("::") target = repo_root / path.strip() if not target.is_file(): - problem(f"change set {change_set}: {path.strip()} does not exist") + problem(f"{entry}: {path.strip()} does not exist") return if name.strip() not in target.read_text(encoding="utf-8", errors="replace"): - problem( - f"change set {change_set}: {path.strip()} contains no test " - f"named '{name.strip()}'" - ) + problem(f"{entry}: {path.strip()} contains no test named '{name.strip()}'") def main(argv: list[str]) -> int: @@ -115,7 +119,7 @@ def problem(message: str) -> None: problems.append(message) counts = spec_scenarios(spec, problem) - entries = note_entries(notes) + entries, fixups = note_entries(notes) for change_set, scenarios in sorted(counts.items()): if change_set not in entries: @@ -128,11 +132,14 @@ def problem(message: str) -> None: f"{len(named)} test(s) named" ) for reference in named: - check_named_test(reference, repo_root, change_set, problem) + check_named_test(reference, repo_root, f"change set {change_set}", problem) for change_set in sorted(set(entries) - set(counts)): problem(f"change set {change_set} is in {notes} but not in the spec's change plan") + for reference in fixups: + check_named_test(reference, repo_root, "a fixup entry", problem) + for message in problems: print(message) if problems: