From ac4388c6328c3a44e7fa512e7ea4dca78013ca2f Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 11:14:43 +0200 Subject: [PATCH 01/30] feat(factory): plugin skeleton and packaging What: add the factory plugin skeleton - plugin.json, README, marketplace entries, the generator's PluginConfig, and the committed Codex distribution - plus the nested-validate test harness under factory/evals/tests/. Why: the factory orchestrator needs its own plugin identity in both marketplaces and a Codex-buildable distribution before any skill exists; docs/architecture.md and CLAUDE.md now state the invoker exception the run skill relies on. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- .agents/plugins/marketplace.json | 12 + .claude-plugin/marketplace.json | 5 + CLAUDE.md | 7 +- README.md | 1 + docs/architecture.md | 16 +- docs/decisions.md | 136 ++++++- factory/.claude-plugin/plugin.json | 8 + factory/README.md | 25 ++ factory/evals/tests/__init__.py | 0 factory/evals/tests/test_factory_checks.py | 190 ++++++++++ plugins/factory/.codex-plugin/plugin.json | 24 ++ plugins/factory/README.md | 3 + plugins/factory/references/architecture.md | 92 +++++ plugins/factory/references/ci-parity.md | 52 +++ plugins/factory/references/contracts.md | 67 ++++ plugins/factory/references/decision-ledger.md | 120 +++++++ .../factory/references/dependency-rules.md | 47 +++ plugins/factory/references/factory-run.md | 59 ++++ plugins/factory/references/jira.md | 115 ++++++ plugins/factory/references/plan-layout.md | 34 ++ plugins/factory/references/reporting.md | 23 ++ plugins/factory/scripts/architecture-check.py | 135 +++++++ plugins/factory/scripts/skill-metrics.py | 333 ++++++++++++++++++ plugins/factory/skills/build/SKILL.md | 104 ++++++ .../factory/skills/build/agents/openai.yaml | 6 + .../skills/build/references/e2e-report.md | 50 +++ .../factory/skills/build/references/layers.md | 50 +++ .../skills/build/references/mocking.md | 52 +++ .../skills/build/references/parallel.md | 63 ++++ .../factory/skills/build/references/tests.md | 91 +++++ .../skills/build/scripts/check-tests.py | 144 ++++++++ .../skills/build/templates/e2e-report.html | 248 +++++++++++++ plugins/factory/skills/run/SKILL.md | 50 +++ plugins/factory/skills/run/agents/openai.yaml | 6 + .../factory/skills/run/references/judgment.md | 40 +++ .../factory/skills/run/references/launch.md | 32 ++ .../factory/skills/run/scripts/run-state.py | 276 +++++++++++++++ plugins/factory/skills/scope-review/SKILL.md | 131 +++++++ .../skills/scope-review/agents/openai.yaml | 6 + .../skills/scope-review/references/lenses.md | 55 +++ plugins/factory/skills/scope/SKILL.md | 162 +++++++++ .../factory/skills/scope/agents/openai.yaml | 6 + .../skills/scope/references/bootstrap.md | 72 ++++ .../skills/scope/references/data-schema.md | 64 ++++ .../skills/scope/references/reverse-mode.md | 20 ++ .../factory/skills/scope/scripts/lint-spec.py | 208 +++++++++++ .../factory/skills/scope/templates/spec.html | 151 ++++++++ plugins/factory/skills/ship/SKILL.md | 142 ++++++++ .../factory/skills/ship/agents/openai.yaml | 6 + .../skills/ship/references/data-schema.md | 46 +++ .../skills/ship/references/gauntlet.md | 56 +++ .../factory/skills/ship/references/lenses.md | 102 ++++++ .../ship/references/orchestration-heavy.md | 108 ++++++ .../skills/ship/references/orchestration.md | 150 ++++++++ .../skills/ship/references/pull-request.md | 75 ++++ .../skills/ship/references/remediation.md | 42 +++ .../skills/ship/references/report-format.md | 50 +++ .../factory/skills/ship/references/tools.md | 79 +++++ .../skills/ship/scripts/aggregate-findings.py | 185 ++++++++++ .../skills/ship/scripts/pr-evidence.py | 311 ++++++++++++++++ .../skills/ship/scripts/run-pi-agents.sh | 69 ++++ .../factory/skills/ship/templates/review.html | 214 +++++++++++ scripts/build_codex_plugin.py | 36 ++ 63 files changed, 5253 insertions(+), 9 deletions(-) create mode 100644 factory/.claude-plugin/plugin.json create mode 100644 factory/README.md create mode 100644 factory/evals/tests/__init__.py create mode 100644 factory/evals/tests/test_factory_checks.py create mode 100644 plugins/factory/.codex-plugin/plugin.json create mode 100644 plugins/factory/README.md create mode 100644 plugins/factory/references/architecture.md create mode 100644 plugins/factory/references/ci-parity.md create mode 100644 plugins/factory/references/contracts.md create mode 100644 plugins/factory/references/decision-ledger.md create mode 100644 plugins/factory/references/dependency-rules.md create mode 100644 plugins/factory/references/factory-run.md create mode 100644 plugins/factory/references/jira.md create mode 100644 plugins/factory/references/plan-layout.md create mode 100644 plugins/factory/references/reporting.md create mode 100755 plugins/factory/scripts/architecture-check.py create mode 100755 plugins/factory/scripts/skill-metrics.py create mode 100644 plugins/factory/skills/build/SKILL.md create mode 100644 plugins/factory/skills/build/agents/openai.yaml create mode 100644 plugins/factory/skills/build/references/e2e-report.md create mode 100644 plugins/factory/skills/build/references/layers.md create mode 100644 plugins/factory/skills/build/references/mocking.md create mode 100644 plugins/factory/skills/build/references/parallel.md create mode 100644 plugins/factory/skills/build/references/tests.md create mode 100755 plugins/factory/skills/build/scripts/check-tests.py create mode 100644 plugins/factory/skills/build/templates/e2e-report.html create mode 100644 plugins/factory/skills/run/SKILL.md create mode 100644 plugins/factory/skills/run/agents/openai.yaml create mode 100644 plugins/factory/skills/run/references/judgment.md create mode 100644 plugins/factory/skills/run/references/launch.md create mode 100644 plugins/factory/skills/run/scripts/run-state.py create mode 100644 plugins/factory/skills/scope-review/SKILL.md create mode 100644 plugins/factory/skills/scope-review/agents/openai.yaml create mode 100644 plugins/factory/skills/scope-review/references/lenses.md create mode 100644 plugins/factory/skills/scope/SKILL.md create mode 100644 plugins/factory/skills/scope/agents/openai.yaml create mode 100644 plugins/factory/skills/scope/references/bootstrap.md create mode 100644 plugins/factory/skills/scope/references/data-schema.md create mode 100644 plugins/factory/skills/scope/references/reverse-mode.md create mode 100755 plugins/factory/skills/scope/scripts/lint-spec.py create mode 100644 plugins/factory/skills/scope/templates/spec.html create mode 100644 plugins/factory/skills/ship/SKILL.md create mode 100644 plugins/factory/skills/ship/agents/openai.yaml create mode 100644 plugins/factory/skills/ship/references/data-schema.md create mode 100644 plugins/factory/skills/ship/references/gauntlet.md create mode 100644 plugins/factory/skills/ship/references/lenses.md create mode 100644 plugins/factory/skills/ship/references/orchestration-heavy.md create mode 100644 plugins/factory/skills/ship/references/orchestration.md create mode 100644 plugins/factory/skills/ship/references/pull-request.md create mode 100644 plugins/factory/skills/ship/references/remediation.md create mode 100644 plugins/factory/skills/ship/references/report-format.md create mode 100644 plugins/factory/skills/ship/references/tools.md create mode 100755 plugins/factory/skills/ship/scripts/aggregate-findings.py create mode 100755 plugins/factory/skills/ship/scripts/pr-evidence.py create mode 100755 plugins/factory/skills/ship/scripts/run-pi-agents.sh create mode 100644 plugins/factory/skills/ship/templates/review.html diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index 3a07536..e1b760a 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -15,6 +15,18 @@ "authentication": "ON_INSTALL" }, "category": "Developer Tools" + }, + { + "name": "factory", + "source": { + "source": "local", + "path": "./plugins/factory" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "Developer Tools" } ] } diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index a230864..8b3a74a 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -11,6 +11,11 @@ "name": "dev", "source": "./dev", "description": "Development workflow skills: scope changes with argued decisions, build across unit/integration/e2e with every scenario proven by tests, ship with a deterministic quality gauntlet and an adversarially verified review, create structured commits that feed a decision ledger, and render pitches or comprehension quizzes." + }, + { + "name": "factory", + "source": "./factory", + "description": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself, ending in a pull request or a report." } ] } diff --git a/CLAUDE.md b/CLAUDE.md index 4ebdee3..278ee63 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -8,7 +8,10 @@ This is a monorepo for Tobrun's Claude Code, Codex, and Pi skills. The root `.claude-plugin/marketplace.json` references the Claude source plugins. `.agents/plugins/marketplace.json` references generated Codex plugins under `plugins/`. The root `package.json` exposes the `dev` source skills as a Pi -package. There is one plugin: `dev`, the hand-invoked development workflow. +package. There are two plugins: `dev`, the hand-invoked development workflow, +and `factory`, whose `run` skill is the one sanctioned exception to "skills +never invoke each other" - it launches the copied `scope`, `scope-review`, +`build`, and `ship` phase skills by path and judges their completion itself. ## Repository Structure @@ -45,7 +48,7 @@ package. There is one plugin: `dev`, the hand-invoked development workflow. - All changes must pass `scripts/validate.sh` before committing. - Every plugin directory name must match its `plugin.json` name and marketplace entry name. - Every `SKILL.md` must have YAML frontmatter with `name` and `description`. -- Every `SKILL.md` must set `disable-model-invocation: true`; all skills in this repo are human-triggered only, and skills recommend the next step instead of invoking each other. +- Every `SKILL.md` must set `disable-model-invocation: true`; all skills in this repo are human-triggered only, and skills recommend the next step instead of invoking each other - except the factory `run` skill, the one sanctioned invoker, which launches its copied phase skills by path. - Plan files under `.dev/` are never committed. - Do not edit `plugins/` directly. Run `python3 scripts/build_codex_plugin.py` after changing `dev/`; the generator builds every configured diff --git a/README.md b/README.md index f62a957..462be10 100644 --- a/README.md +++ b/README.md @@ -3,6 +3,7 @@ | Plugin | Use When | Tools | | ------ | -------- | ----- | | [dev](dev/) | A test-focused development workflow for Claude Code, Codex, opencode, and Pi. | `scope`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz` | +| [factory](factory/) | Take a request from scope to a shipped pull request unattended, on Claude Code or Codex. | `run`, `scope`, `scope-review`, `build`, `ship` | ## Claude Code diff --git a/docs/architecture.md b/docs/architecture.md index 8cc1c35..906df2e 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1,11 +1,11 @@ # Architecture Purpose: this repository is a monorepo of agent skills and the tooling that ships them. -One plugin lives here: `dev`, the hand-invoked development workflow (scope, scope-review, build, ship, commit, and two presentation skills). -A person invokes each skill by hand; skills never invoke each other, and each one recommends the next step instead. -The generated Codex distribution under `plugins/` is built from the source plugin and never edited by hand. +Two plugins live here: `dev`, the hand-invoked development workflow (scope, scope-review, build, ship, commit, and two presentation skills), and `factory`, whose `run` skill orchestrates copies of the same four phases unattended. +A person invokes each `dev` skill by hand; skills never invoke each other, and each one recommends the next step instead - except the factory `run` skill, the one sanctioned invoker, which launches its copied phase skills by path and judges their completion itself. +The generated Codex distributions under `plugins/` are built from their source plugin and never edited by hand. -Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin and its runner were removed) +Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin returns as an orchestrator skill; its removed runner stays gone) ## Components @@ -17,6 +17,7 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin and | repo scripts | validation of the whole repository, the plugin generator, and the Pi transport self-test | `scripts/` | every other component | | harden tools | optional local analyses a ship gauntlet can draw on: added lines, coverage-weighted complexity, flaky reruns, mutation | `tools/harden/` | nothing; run by hand against a target repository | | research and todo | plans, findings, and reading notes | `research/`, `todo/` | nothing | +| factory plugin | the `run` orchestrator skill, the four phase copies, their references and scripts | `factory/skills/`, `factory/references/`, `factory/scripts/`, `factory/evals/` | the host's subagent tool, the consuming repository's `.dev/` | ## Flows @@ -27,7 +28,12 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin and ### Building the Codex distribution 1. `scripts/build_codex_plugin.py` copies skills/, references/, and `scripts/` of the source plugin into plugins/{name}/, strips Claude-only frontmatter, and writes agents/openai.yaml. -2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), and the Pi package and transport (P01); `--check` mode of the generator fails when `plugins/` is stale. +2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), the Pi package and transport (P01), and the factory plugin's unattended wording and result protocol; `--check` mode of the generator fails when `plugins/` is stale. + +### A factory run +1. A person invokes `/factory:run "a request"` (or a path to one); `run` scopes the change inline with them and ends with one go question. +2. From the go, `run` launches `scope-review`, `build`, and `ship` in turn as fresh-context subagents, judges each result's file against the phase's own deterministic checks, repairs or relaunches on failure, and never lets a phase invoke another. +3. The run ends in an open pull request, or a report naming the exact action a person could take - the pushed branch and the run's state file are what survives a closed session. ## Boundaries diff --git a/docs/decisions.md b/docs/decisions.md index 9d6801f..f1a6464 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -3,7 +3,7 @@ ## Principles P-deterministic-guards-over-prose: when a loop or skill must be constrained, prefer a check it cannot argue with over an instruction - promoted 2026-09-17 - recurred across several decisions in the factory work (removed 2026-09-20); matches CLAUDE.md "prefer a deterministic check it loops against" + promoted 2026-09-17 - recurred across several decisions in the factory work (removed 2026-09-20); matches CLAUDE.md "prefer a deterministic check it loops against"; recurred again 2026-09-20 in D-unattended-wording-check, D-protocol-check, D-run-state-script (factory-plugin/spec.md); set aside once and deliberately, in D-verification - the principle buys a guard where a guard may hold, and a count that refuses a fourth attempt would be the deterministic gate above the model's judgment D-completion-authority rejects ## Quality gauntlet @@ -15,5 +15,137 @@ D-complexity-threshold: Where does the ship gauntlet's coverage-weighted complex ## Removed D-remove-factory: The factory plugin and its runner are removed from this repository (2026-09-20) - ✓ delete `factory/`, `plugins/factory/`, the runner end-to-end test, and the F01 through F04 validator checks - the implementation was not working out (user, 2026-09-20); the repository goes back to shipping `dev` alone + ✓ delete `factory/`, `plugins/factory/`, the runner end-to-end test, and the F01 through F04 validator checks - the implementation was not working out (user, 2026-09-20); the repository goes back to shipping `dev` alone ⚠ superseded the same day for the plugin, not the runner: factory-plugin/spec.md brings `factory/` back as skills only (D-orchestrator-form); the runner and its gates stay removed ✗ keep it unmaintained behind a flag - dead weight in every validation run and every plugin build, and it would keep drifting from `dev` + +## Factory plugin + +risk: moderate - domains: orchestration, unattended git and GitHub writes (updated 2026-09-20, factory-plugin/spec.md) + +D-orchestrator-form: What is the orchestrator? (2026-09-20, factory-plugin/spec.md) + ✓ a skill, `/factory:run`, that runs inside the interactive host session and launches subagents - no process to install, no per-host argv, works on every host with a subagent tool (auto-applied at Confidence: 85%) ⚠ the run lives as long as the session; the run state file is what survives a closed session + ✗ a runner process outside the host - this is the removed design (D-remove-factory in the ledger stands for the runner; the plugin returns without it), 23k lines of Python with 13k lines of tests, and the gates it owned are what parked runs + ✗ a host-specific agent definition per host - Claude agents, Codex `.codex/agents/` TOML, and opencode agents differ in format; a skill is the one artifact all hosts read +D-completion-authority: Who decides that a phase is complete? (2026-09-20, factory-plugin/spec.md) + ✓ the orchestrator judges, after reading the phase's result file and re-running the phase's own deterministic checks as evidence - the checks are `lint-spec.py`, the review verdict line, `check-tests.py`, the Validation block commands, `pr-evidence.py check`, `git log`, `gh pr view` (auto-applied at Confidence: 80%) ⚠ a wrong judgment advances bad work, and the orchestrator grades work it guided and repaired itself; every judgment is recorded with the check outputs it read, the copied ship's fresh-context panel is the one independent signal on the code, and `skill-metrics.py` numbers are never evidence + ✗ a deterministic gate that overrules the result - the old "gate is authoritative" rule produced the oscillating build attempts and parks on codes nobody could act on (user (2026-09-20)) + ✗ a separate judge subagent - fresh context per judgment costs a launch per phase and the judge would read the same files the orchestrator reads +D-phase-skills: Where do the phase instructions come from? (2026-09-20, factory-plugin/spec.md) + ✓ factory-owned copies of `scope`, `scope-review`, `build`, and `ship` under `factory/skills/`, with their references and scripts copied under `factory/references/` and `factory/scripts/` - the copies are meant to diverge: the long-term goal is to split the monolith skills into smaller factory-only pieces (user (2026-09-20)) ⚠ a fix in `dev` has to be ported by hand; no drift check, because drift is the intent + ✗ reuse the dev skills by path plus an unattended preamble - one source of truth, but it blocks the planned decoupling and depends on `dev` being installed next to `factory`, which plugin installs do not guarantee + ✗ generated copies rewritten by the build script - the rewrite rules would encode the unattended policy twice, once in the script and once in prose +D-phase-transport: How does a phase run? (2026-09-20, factory-plugin/spec.md) + ✓ one fresh-context subagent per phase attempt through the host's plain subagent tool (the Agent tool on Claude Code, `spawn_agent` on Codex, `task` on opencode) - a fresh context per attempt keeps four phases out of one window, following the transport ladder in ship's `orchestration.md`; the subagent's result is a file, never its final message (auto-applied at Confidence: 85%) + ✗ the orchestrator runs the phase inline - one context for four phases loses the middle of everything, and scope-review and ship each spawn panels of their own + ✗ a subprocess per phase (`codex exec`, `claude -p`) - per-host argv again; the removed runner's `hosts.py` was exactly this +D-human-touchpoints: Where is the human? (2026-09-20, factory-plugin/spec.md) + ✓ scope runs inline with the human in the orchestrator's session and ends with one explicit go - scope is an interview and the request names it as the only human phase; from then on no phase asks anyone anything, and a run that is out of options ends with a written report instead of a question (auto-applied at Confidence: 80%) ⚠ the scope interview lives in the orchestrator's context; the state file lets the user clear the session and invoke `/factory:run` again to continue with a fresh context + ✗ scope as a subagent too - subagents have no user-input tool on Claude Code (docs, sub-agents: AskUserQuestion is removed from subagents), and scope is an interview + ✗ a second human gate before the pull request - the request names scope as the only human phase +D-orchestrator-hands: What may the orchestrator change itself when a phase fails? (2026-09-20, factory-plugin/spec.md) + ✓ anything, including source files, committed on the run branch and recorded in the run state as a repair - the old parks were faults a person fixes in two commands, and the user would rather trust the model with an audit trail than add rules (user (2026-09-20)) ⚠ orchestrator context grows with every hands-on fix; the phase's own checks re-run after a fix, so a wrong fix fails the next judgment instead of shipping; the two hard stops bind these hands too: the orchestrator never rewrites history, force-pushes, or deletes outside the run branch + ✗ plan files and git only - keeps the orchestrator lean but sends one-line faults back through a full phase attempt + ✗ nothing, relaunch only - repeats the parking pattern for tiny faults +D-plan-files-ignored: How does "plan files are never committed" hold in a consuming repository? (2026-09-20, factory-plugin/spec.md) + ✓ the `run` skill's preflight runs `git check-ignore -q .dev/` in the consuming repository and, when it is not ignored, the orchestrator appends the entry to `.gitignore` itself and records it as a repair - the check sits before the handoff next to the transport check, while a person is still at the keyboard, so scope's first file is already ignored when it is written; the write is an ordinary repair under D-orchestrator-hands, committed on the run branch and recorded in `factory-run.json` with rationale and evidence, and bound by the two hard stops like any other (user (2026-09-20)) ⚠ the accepted cost: the factory edits a file the user never asked it to touch, as one of its first actions, before any phase has run ⚠ the edit is made at preflight but only committed once the run branch exists at handoff, so the default branch never carries it, and a run abandoned before the handoff leaves the line in the working tree + ✗ a preflight that ends the run with the one-line fix instead of writing it - parking on a fault a person fixes in one command is the pattern D-orchestrator-hands exists to remove, and this one would park before any work has been done at all + ✗ leaving it as a documented Input and a README prerequisite - nothing checks a prerequisite, so the first consuming repository without the entry commits the plan files silently; this repository is itself such a repository today, and only the fixture's `setup.sh` writes an entry +D-failure-ladder: What happens after a failed or stopped phase? (2026-09-20, factory-plugin/spec.md) + ✓ cheapest sufficient step, recorded each time - fix it myself and re-run the checks; relaunch the phase in a fresh subagent with guidance naming what the previous attempt did and what is required instead; end the run with a report when three attempts of one phase have not produced a `done` the orchestrator accepts; at most two repairs per attempt, after which the next fix is a relaunch and counts as an attempt (auto-applied at Confidence: 80%) ⚠ three and two are guesses; reopen when a real run needs one more that would have succeeded ⚠ a repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches; the copied ship's remediation loop is built for that ⚠ three and two are prose the orchestrator follows, not counts it is held to: `run-state.py record` tallies every attempt and repair and `show` prints them, so the numbers are on disk, but nothing refuses a fourth attempt, and the refusal that would catch a miscount is the deterministic gate D-completion-authority rejects + ✗ unlimited attempts - the old run with ten build attempts burned 18 hours on an oscillation + ✗ one attempt then stop - most faults are fixable on the second try with guidance +D-hard-stops: Which situations end a run without judgment? (2026-09-20, factory-plugin/spec.md) + ✓ two - a found secret and a destructive action outside the run branch (history rewrite, force push, deleting data or remote resources); the phase reports them as `stopped`, the orchestrator never overrides them, and the same two rules bind the orchestrator's own repairs (auto-applied at Confidence: 80%) + ✗ the old seven-code park list - missing credentials, no launch command, and an invalidated premise are judgment calls the orchestrator now makes: mock, skip, decide, or end the run with the exact action a person must take + ✗ no hard stops - a secret in a pushed branch is not recoverable by judgment +D-run-state: How does a run survive the session? (2026-09-20, factory-plugin/spec.md) + ✓ `.dev/{plan}/factory-run.json`, written only by the orchestrator - it holds phases with attempts, each attempt's outcome and the orchestrator's decision with rationale and evidence, repairs, the branch and base, the approved scenario texts, the `⊘` lines, and the spec hash at handoff; the file exists from `init` on, before any branch; invoking `/factory:run` without a request resumes the plan whose branch is checked out, else the single state file not `done` under `.dev/`, and an attempt left `launched` without a result is marked failed and relaunched (auto-applied at Confidence: 75%) + ✗ infer state from which plan files exist - cannot tell a crashed build from an unstarted one, and holds no decisions + ✗ a run directory outside the repo - the old `~/.factory/runs/` needed a CLI to look at; `.dev/` is where every other plan artifact lives and is never committed +D-checkout: Where does the work happen? (2026-09-20, factory-plugin/spec.md) + ✓ the current checkout on a new branch `factory/{plan}` created at handoff from the default branch - the human just scoped there, subagents on both hosts share the cwd, and no worktree bookkeeping exists (auto-applied at Confidence: 75%) ⚠ a user edit during the run lands in the checkout; the orchestrator records the dirty files at handoff and judges any new ones + ✗ a git worktree per run - Claude's Agent tool offers worktree isolation, Codex's does not; the old runner's worktrees were a source of stale-state parks + ✗ the user's current branch - the run must be revertable as one branch +D-result-envelope: How does a phase report back? (2026-09-20, factory-plugin/spec.md) + ✓ `.dev/{plan}/results/{phase}-{attempt}.json` written as the phase's last action - one file per attempt, so a stale `done` from attempt 1 can never satisfy attempt 2 and an abandoned subagent that finishes late writes a file nobody reads; fields `schema` `factory.result/1`, `phase`, `skill_path` (the phase skill path the launch prompt passed, echoed back so that the subagent having opened the file it was given is visible in a file the orchestrator already reads), `status` (`done`, `failed`, `stopped`), `reason`, `artifacts[]`, `counts{}`, `auto_decided[]`, and `stop {kind, action}` only when stopped; the orchestrator reads it leniently and treats a missing or unparseable file as a failed attempt with that reason (auto-applied at Confidence: 80%) + ✗ one `{phase}-result.json` overwritten per attempt - a relaunch after a crash could read the previous attempt's file + ✗ the old schema 2 with typed conditions and strict parsing - a malformed file failed attempts, and the condition codes were the park vocabulary this design removes + ✗ the subagent's final message - unreliable on hosts where subagents finish in the background (ship's orchestration.md records this) +D-skill-locating: How does a subagent find the phase skill? (2026-09-20, factory-plugin/spec.md) + ✓ the orchestrator passes the absolute path of the phase's `SKILL.md` and the phase skill root in the prompt, and tells the subagent to read `{phase}-skill-root` as that path - the `run` skill learns its own base directory from the host's injected "Base directory for this skill" line on Claude Code, and the dev skills already use `{scope-skill-root}` style placeholders that a subagent reading a file cannot resolve alone; the subagent reads the file and follows it (auto-applied at Confidence: 85%) ⚠ how Codex tells a skill its path is undocumented ? verify: each e2e run's result file echoes `skill_path` as the exact path the launch prompt passed, the Codex one included + ✗ the subagent invokes the skill by name - `disable-model-invocation: true` and `allow_implicit_invocation: false` block that on both hosts, and both flags are repo rules + ✗ paste the skill text into the prompt - four skills of 100 to 160 lines plus references would bloat every launch +D-panel-nesting: Where do scope-review's and ship's panels run? (2026-09-20, factory-plugin/spec.md) + ✓ inside the phase subagent, so the orchestrator is depth 0, the phase is depth 1, and panel agents are depth 2 - the copied skills already own their panels (auto-applied at Confidence: 75%) ? verify: both e2e runs; Codex has `agents.max_depth` config and the old runner set it, Claude's default depth is 3 + ✗ hoist panels to the orchestrator - rewrites two skills and puts panel results into the orchestrator's context +D-hosts: Which hosts must the first slice run on? (2026-09-20, factory-plugin/spec.md) + ✓ Claude Code and Codex, each proven by one real end-to-end run on the fixture repository - the user launches runs from both; opencode carries the same wording (`task` tool) but is not run; the fixture's bare origin makes a host run's pass criterion the branch pushed to that origin with one commit per change set and `pr.md` written, not an open pull request (user (2026-09-20)) ⚠ opencode support is a claim until someone runs it, and the pull-request step stays unproven by both runs + ✗ Claude Code end to end and Codex structural only - the user launches runs from Codex +D-pi: Ship the factory to Pi? (2026-09-20, factory-plugin/spec.md) + ✗ add the factory skills to the Pi package - Pi's flat namespace would collide `scope`, `build`, and `ship` with dev's (validate.sh P01 keeps the package dev-only) + ✗ a Pi subprocess transport for phases - `run-pi-agents.sh` disables write tools by design, so a build phase cannot run through it + ⊘ not doing - no phase transport exists for Pi; reopen when Pi gains a native subagent tool or a writable subprocess transport is wanted +D-invoker-invariant: The repo says skills never invoke each other; does the orchestrator break it? (2026-09-20, factory-plugin/spec.md) + ✓ the orchestrator is the one sanctioned invoker - it launches phase skills by path, and phase skills still never invoke each other or the next phase; `docs/architecture.md` and `CLAUDE.md` state the exception (auto-applied at Confidence: 85%) + ✗ keep the invariant literal and make the human launch each phase - that is the dev workflow, not a factory +D-unattended-wording-check: How is "no human after scope" enforced in the copies? (2026-09-20, factory-plugin/spec.md) + ✓ a validate.sh check (revived old F03) - it fails any line under `factory/skills/{scope-review,build,ship,run}` (SKILL.md and their references) and `factory/references/factory-run.md` that routes a decision to a person, with `` blocks exempt; the regex self-tests against four sample phrases and a failed self-test fails validation, as the removed F03 already did - P-deterministic-guards-over-prose (auto-applied at Confidence: 85%) ⚠ the other copied references (`ci-parity.md`, `contracts.md`, `jira.md`) keep their dev wording about people because phases read them for notation, not for who decides + ✗ prose in the preamble only - a rule in prose softens as the context grows (CLAUDE.md); the check's exit code does not +D-protocol-check: How is the result protocol kept present in every phase copy? (2026-09-20, factory-plugin/spec.md) + ✓ a validate.sh check (revived old F02) requires each phase SKILL.md to name `factory-run.json` and the result path pattern `results/{phase}-{attempt}.json`, and the `run` SKILL.md to name every phase - a copy that drifts off the protocol fails validation + ✗ trust the copies - a dropped line would only show up in a paid run +D-handoff-seal: Is the approved scope protected after the human leaves? (2026-09-20, factory-plugin/spec.md) + ✓ recorded, not sealed - at the go, the orchestrator stores the spec's sha256, every `⊘` line, and every `tests:` scenario text per change set (the plan format has no scenario ids; `check-tests.py` counts scenarios per change set) in the run state; before each judgment it diffs them against the current spec and treats a dropped or reworded approved scenario as evidence for its decision (auto-applied at Confidence: 75%) + ✗ the old sealed intent with a gate - a reworded scenario parked the run even when the rewording was right + ✗ nothing - later phases could quietly drop scenarios the human approved +D-orchestrator-context: How does the orchestrator stay small across four phases? (2026-09-20, factory-plugin/spec.md) + ✓ it reads result files, check outputs, `git log`, and the state file - it never reads a subagent transcript, a full spec, or implementation notes unless a judgment needs a specific section (auto-applied at Confidence: 75%) ⚠ hands-on repairs and the scope interview still accumulate; clearing and re-invoking is the escape + ✗ summaries pasted by subagents into their final message - the final message is not the channel +D-scope-review-deferrals: What does the unattended scope-review do with escalations? (2026-09-20, factory-plugin/spec.md) + ✓ it answers them itself under the unattended policy and records each as `auto-decided` - a finding that invalidates the premise or opens a new effort is reported as `failed` with reason `rescope`, and the orchestrator ends the run naming what scope must revisit (auto-applied at Confidence: 80%) + ✗ APPROVED WITH DEFERRALS carried into build - build would implement around an unanswered question +D-metrics: How is a run measured? (2026-09-20, factory-plugin/spec.md) + ✓ each phase copy keeps its `skill-metrics.py start/end` calls - the orchestrator records wall-clock per attempt in the run state (auto-applied at Confidence: 75%) ⚠ `skill-metrics.py` reads Claude transcripts, so Codex rows are zeros and subagent runs may anchor on the wrong transcript; the run state timings are the portable numbers and metrics rows are never judgment evidence + ✗ a new metrics script for the orchestrator - more code for numbers nobody has asked for yet +D-model-selection: Can the orchestrator pick a model per phase? (2026-09-20, factory-plugin/spec.md) + ✗ a per-phase model table - Claude's Agent tool takes a model, Codex's spawn does not document one; a table would be host-specific + ⊘ not doing - phases inherit the host session's model; reopen when a real run shows one phase needs a different model than the session +D-parallel-phases: Can phases overlap? (2026-09-20, factory-plugin/spec.md) + ✗ start build while scope-review's panel runs - `spec.md` has one writer at a time (plan-layout.md) + ⊘ not doing - phases run strictly in order; reopen if a run's wall clock is dominated by a phase that could safely overlap +D-plan-slug: Who names the plan directory? (2026-09-20, factory-plugin/spec.md) + ✓ the `run` skill derives the kebab-case slug from the request and creates `.dev/{plan}/` with `request.md` in it before scope starts, and the scope copy's Factory context says to use that directory rather than name one - `init` already takes the plan, and the handoff spec hash, the `results/` paths, and every judgment path key off it + ✗ the scope copy names the directory as dev scope does (`plan-layout.md`: pick a kebab-case slug for the outcome) - a slug chosen during the interview can differ from the one `factory-run.json` and `results/` already use, putting `spec.md` in a second directory; scope running inline in the orchestrator's own context makes that unlikely, not impossible + ✗ derive it from the branch as plan-layout.md's Locating section does - D-checkout creates `factory/{plan}` at handoff, after scope has already written files +D-nested-validate: How does the harness run `validate.sh` without recursing into itself? (2026-09-20, factory-plugin/spec.md) + ✓ `check_factory_script` discovers the tests under `factory/evals/tests/` with `test_factory_checks.py` excluded by name, with the reason at the call site - that one file copies the tree and runs the copy's `scripts/validate.sh`, whose ROOT is `dirname $0`, so without the exclusion the gate re-enters itself once per level and never terminates (a minimal reproduction recursed seven levels, one full tree copy each); with it, the copy's gate runs `test_run_state.py` only and the nesting stops at one level ⚠ a second harness-shaped test would have to be excluded too, so the exclusion is one line naming the files that drive `validate.sh` + ✗ a guard environment variable the harness exports so the nested `check_factory_script` skips - it also makes change set 5's "an unmutated copy passes all three checks" unassertable, because the third check never runs in the copy + ✗ the harness runs a stub instead of the real `scripts/validate.sh` - the thing under test is shell in `validate.sh`, and a stub would only prove the stub +D-jira-source: Can a run start from a Jira ticket? (2026-09-20, factory-plugin/spec.md) + ✗ an `acli` fetch of the ticket into `request.md` - the old `--jira` mode needed runner plumbing, and the dev Jira sync in `.dev/config.json` still applies inside phases + ⊘ not doing - text and file cover the first slice; reopen when a run starts from a ticket more often than from text +D-verification: How is the factory itself verified? (2026-09-20, factory-plugin/spec.md) + ✓ three levels - validate.sh structural checks and unit tests for the one script, both free; one real end-to-end run per host on a fixture repository, paid and manual, with its procedure and results recorded under `factory/evals/`; races with a user editing during a run and host-side subagent crashes are not testable and are recorded as such; six stated behaviors are knowingly unproven because no scenario, script, or check covers them: the one-writer-per-file invariant, the three-attempt cap, the two-repairs-per-attempt cap, the no-subagent-tool preflight, the `.dev/`-ignored preflight with its `.gitignore` repair (the fixture's `setup.sh` writes the entry, so no free scenario ever reaches the missing-entry branch), and the third-failure report (auto-applied at Confidence: 80%) ⚠ those six stay orchestrator prose by choice: counting them or checking them in validate.sh would put a deterministic gate above the model's judgment, which D-completion-authority rejects as what produced the oscillating build attempts and the parks on codes nobody could act on (user (2026-09-20)); this is the one place the spec sets P-deterministic-guards-over-prose aside, and deliberately: the principle buys a guard where a guard may hold, and a count that refuses a fourth attempt is exactly the gate above the model's judgment this design removes, while the guards the principle does buy here (the two validate.sh checks, `run-state.py`) all sit below it; the run state still records attempts and repairs, so the accepted cost is only that a miscount silently burns a fourth attempt, and runaway attempts are the pattern that cost 18 hours in the old runner + ✗ an offline benchmark with stub hosts - the old bench never ran in real mode and proved nothing about parking + ✗ paid runs in validate.sh - paid model calls never run in the gate (docs/architecture.md) +D-run-state-script: Is the run state handled by prose or code? (2026-09-20, factory-plugin/spec.md) + ✓ one script, `factory/skills/run/scripts/run-state.py`, with `init`, `handoff`, `record`, `show`, and `check-result` subcommands - they validate and update `factory-run.json` and parse a result file; the orchestrator loops on its exit code, so a malformed state never reaches a judgment (auto-applied at Confidence: 80%) + ✗ the orchestrator edits JSON by hand - one typo breaks resume; P-deterministic-guards-over-prose +D-fixture-repo: What do the end-to-end runs run against? (2026-09-20, factory-plugin/spec.md) + ✓ a tiny Python webhook service under `factory/evals/fixture/`, written new (the old bench app is a 7-line class with an empty test package and no server, so only its name is reused): an `http.server` endpoint, two unittest tests, and `python3 -m webhook` as the launch command, turned into a fresh git repository by `factory/evals/fixture/setup.sh` at run time, with a bare repository beside the copy set as `origin` so `git push` works with no network and no GitHub account - small enough that a run costs minutes, real enough for build's unit and e2e layers, and free to run offline (auto-applied at Confidence: 75%; the bare origin is user (2026-09-20)) ⚠ `gh pr create` has no GitHub repository behind a bare origin, so ship fails there with the exact text "none of the git remotes configured for this repository point to a known GitHub host", raised before any network call with `gh` installed and authenticated, and the fixture never proves the pull-request step + ✗ a real project - a first run on real code hides factory faults behind project faults +D-scratch-root: Where do subagent batches write scratch files? (2026-09-20, factory-plugin/spec.md) + ✓ `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/` - includes the plan so two runs on one machine never collide, and follows reporting.md's `/tmp/{project-slug}/` root + ✗ `/tmp/ship-{session-id}/` as in dev - session ids are not shared across the orchestrator and its subagents +D-codex-gh: Does ship on Codex reach GitHub? (2026-09-20, factory-plugin/spec.md) + ✓ documented prerequisites in `factory/README.md` - `GH_TOKEN` (or `GITHUB_TOKEN`) exported before launching Codex because the workspace-write sandbox blocks the macOS keychain `gh` uses, `gh auth setup-git` so `git push` over HTTPS uses the same token, and `sandbox_workspace_write.network_access=true` because the sandbox has no network by default (the old `hosts.py` set all three); the orchestrator tells three faults apart by the error text - a 401, a network refusal, and a remote no GitHub host backs ("none of the git remotes configured for this repository point to a known GitHub host", whose trailing `gh auth login` hint names no missing credential and must not be read as one) - and ends the run naming the missing prerequisite or the unbacked remote ? verify: still open after the fixture runs - the fixture's bare origin exercises `git push` but has no GitHub repository behind it, so `GH_TOKEN`, `gh auth setup-git`, and `sandbox_workspace_write.network_access=true` stay waiting on a Codex run against a real GitHub remote + ✗ inject the token from the orchestrator - it has no way to reach outside the sandbox either +D-evals: Are there skill evals for the new plugin? (2026-09-20, factory-plugin/spec.md) + ✗ a comprehension eval per factory skill now - the copies differ from dev only in the rewritten human lines and the result file, and dev's evals cover the rest + ⊘ not doing - `factory/evals/` holds the fixture and the e2e procedure only; reopen when a factory copy diverges from its dev source in behavior +D-tests-location: Where do the run-state script's unit tests live? (2026-09-20, factory-plugin/spec.md) + ✓ under `factory/evals/tests/` - `evals/` is never copied into a distribution, matching how dev keeps its evals out of `plugins/` + ✗ next to the script under `skills/run/scripts/tests/` - shipped to every install by the generator's `copied_dirs` +D-skill-length: How do the copies stay under the repo's SKILL.md length rule? (2026-09-20, factory-plugin/spec.md) + ✓ within five lines of the dev source - the Factory context section is short and points at `factory-run.md`, and each rewritten human line replaces its dev line instead of adding one; `scope` is already 160 lines, so the rule of roughly 150 is met the same way dev meets it + ✗ a full Factory context block per skill as before - the old copies grew past the rule and R00 only fails at 200 diff --git a/factory/.claude-plugin/plugin.json b/factory/.claude-plugin/plugin.json new file mode 100644 index 0000000..23e7118 --- /dev/null +++ b/factory/.claude-plugin/plugin.json @@ -0,0 +1,8 @@ +{ + "name": "factory", + "version": "0.1.0", + "description": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", + "author": { + "name": "Tobrun" + } +} diff --git a/factory/README.md b/factory/README.md new file mode 100644 index 0000000..faa3e8b --- /dev/null +++ b/factory/README.md @@ -0,0 +1,25 @@ +# factory + +An orchestrator skill that takes a request through an interactive `scope`, then runs `scope-review`, `build`, and `ship` as unattended phases on the current checkout - judging each phase's completion itself, repairing or relaunching it on failure, and ending in a pull request or a report. No runner process to install: the whole thing is one skill, `run`, that launches subagents through the host's own subagent tool. + +## Install + +- **Claude Code**: `/plugin marketplace add tobrun/workflow` then `/plugin install factory@nurbot`, or locally `/plugin marketplace add ~/ws/workflow` then `/plugin install factory@nurbot`. +- **Codex**: `codex plugin marketplace add .` then `codex plugin add factory@nurbot`; set `agents.max_depth` to at least 2, since scope-review's and ship's own panels run one level inside the phase subagent. + +## The run flow + +Invoke `/factory:run "a quoted request"` or `/factory:run path/to/request.md`. `run` checks the transport, scopes the change with you inline (the only interactive phase), then ends with one go question. From there it launches `scope-review`, `build`, and `ship` in order, each as a fresh-context subagent, judging every result against the phase's own deterministic checks before advancing. A found secret or a destructive action outside the run branch stops the run immediately; anything else gets repaired, relaunched, or - after three attempts of one phase - ends the run with a report naming what a person could do. + +Invoke `/factory:run` with no request to resume: the plan on the checked-out branch, or the single unfinished state file under `.dev/`. + +## Prerequisites + +- Every host: a secret scanner ship's gauntlet can run offline - `gitleaks`, or the repo-fitted grep `references/tools.md` allows - since the gauntlet's acquire-a-missing-tool step needs a network a factory run may not have. +- Codex additionally: `GH_TOKEN` (or `GITHUB_TOKEN`) exported before launching Codex, because the workspace-write sandbox blocks the macOS keychain `gh` normally uses; `gh auth setup-git` so `git push` over HTTPS uses the same token; `sandbox_workspace_write.network_access=true`, since the sandbox has no network by default. + +Ignoring `.dev/` in the consuming repository is not a prerequisite: the run's own preflight checks `git check-ignore -q .dev/` and appends the entry to `.gitignore` itself, recorded as a repair, when it is missing. + +## Clear and resume + +The run lives as long as the session; `.dev/{plan}/factory-run.json` is what survives a closed one. Clear the session and invoke `/factory:run` again with no request to pick up where it left off, with a fresh context. diff --git a/factory/evals/tests/__init__.py b/factory/evals/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/factory/evals/tests/test_factory_checks.py b/factory/evals/tests/test_factory_checks.py new file mode 100644 index 0000000..f3dba60 --- /dev/null +++ b/factory/evals/tests/test_factory_checks.py @@ -0,0 +1,190 @@ +"""Unit tests for the factory-specific validate.sh checks (F02, F03). + +Each scenario copies the repo tree to a temp directory, mutates the copy, +rebuilds the Codex distributions there (any mutation under factory/ makes +plugins/factory stale, so validate.sh would fail on [C01] instead of on the +check under test without a rebuild), then runs the copy's own +scripts/validate.sh - never the real repo's. + +This file is excluded by name from check_factory_script's own discovery +(D-nested-validate): it runs validate.sh itself, so without the exclusion +the gate would re-enter itself once per nesting level and never terminate. +Run from the repo root: python3 -m unittest discover -s factory/evals/tests -t . +""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +IGNORE = shutil.ignore_patterns( + ".git", "__pycache__", ".pytest_cache", ".ruff_cache", "node_modules", ".DS_Store" +) + + +def copy_repo() -> Path: + tmp = Path(tempfile.mkdtemp(prefix="factory-checks-")) + dest = tmp / "repo" + shutil.copytree(REPO_ROOT, dest, ignore=IGNORE) + return dest + + +def rebuild(copy_root: Path) -> None: + subprocess.run( + [sys.executable, "scripts/build_codex_plugin.py"], + cwd=copy_root, + check=True, + capture_output=True, + text=True, + ) + + +def run_validate(copy_root: Path) -> subprocess.CompletedProcess: + full_env = dict(os.environ) + full_env["VALIDATE_LOGS"] = str(copy_root / "validate-logs") + return subprocess.run( + ["bash", "scripts/validate.sh"], + cwd=copy_root, + capture_output=True, + text=True, + env=full_env, + ) + + +class FactoryChecksTest(unittest.TestCase): + def setUp(self) -> None: + self.copy_root = copy_repo() + + def tearDown(self) -> None: + shutil.rmtree(self.copy_root.parent, ignore_errors=True) + + def append_to(self, relative: str, text: str) -> Path: + target = self.copy_root / relative + with target.open("a", encoding="utf-8") as handle: + handle.write(text) + return target + + def test_unmutated_copy_passes_and_excludes_the_harness(self) -> None: + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + log = self.copy_root / "validate-logs" / "factory-script-tests.log" + if log.is_file(): + contents = log.read_text(encoding="utf-8") + self.assertIn("test_run_state", contents) + self.assertNotIn("test_factory_checks", contents) + + def test_decision_routed_to_a_person_fails_f03(self) -> None: + target = self.append_to( + "factory/skills/build/SKILL.md", "\n\nAsk the user which one to pick.\n" + ) + rebuild(self.copy_root) + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 1) + self.assertIn("[F03]", result.stdout) + self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) + self.assertNotIn("[C01]", result.stdout) + + def test_same_line_inside_interactive_only_passes(self) -> None: + self.append_to( + "factory/skills/build/SKILL.md", + "\n\n\nAsk the user which one to pick.\n\n", + ) + rebuild(self.copy_root) + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + + def test_broken_self_test_regex_fails_f03(self) -> None: + validate_sh = self.copy_root / "scripts" / "validate.sh" + text = validate_sh.read_text(encoding="utf-8") + # Break the F03 pattern so it no longer matches its own self-test samples. + broken = text.replace( + r"ask(s|ed|ing)?|confirm(s|ed|ing)?\s+with", + r"askXXX(s|ed|ing)?|confirm(s|ed|ing)?\s+withXXX", + ) + self.assertNotEqual(text, broken, "expected the F03 pattern text to be present") + validate_sh.write_text(broken, encoding="utf-8") + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 1) + self.assertIn("[F03]", result.stdout) + self.assertIn("pattern no longer matches", result.stdout) + + def test_missing_result_protocol_literal_fails_f02(self) -> None: + target = self.copy_root / "factory" / "skills" / "build" / "SKILL.md" + text = target.read_text(encoding="utf-8") + text = text.replace("results/{phase}-{attempt}.json", "some-other-path.json") + target.write_text(text, encoding="utf-8") + rebuild(self.copy_root) + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 1) + self.assertIn("[F02]", result.stdout) + self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) + self.assertNotIn("[C01]", result.stdout) + + def test_invoking_another_phase_fails_f02(self) -> None: + target = self.append_to( + "factory/skills/build/SKILL.md", "\n\nOn success, invoke $factory:ship.\n" + ) + rebuild(self.copy_root) + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 1) + self.assertIn("[F02]", result.stdout) + self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) + + def test_unattended_wording_check_is_clean_over_the_four_phase_copies(self) -> None: + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertNotIn("[F03]", result.stdout) + + def test_protocol_check_is_clean_over_every_phase_skill(self) -> None: + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertNotIn("[F02]", result.stdout) + + def test_build_codex_plugin_check_reports_up_to_date(self) -> None: + rebuild(self.copy_root) + result = subprocess.run( + [sys.executable, "scripts/build_codex_plugin.py", "--check", "--plugin", "factory"], + cwd=self.copy_root, + capture_output=True, + text=True, + ) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertIn("up to date", result.stdout) + + def test_architecture_check_is_clean_on_the_updated_overview(self) -> None: + result = subprocess.run( + [sys.executable, "dev/scripts/architecture-check.py", "docs/architecture.md", "--root", "."], + cwd=self.copy_root, + capture_output=True, + text=True, + ) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + + def test_phase_copies_stay_within_five_lines_of_dev(self) -> None: + for skill in ("scope", "scope-review", "build", "ship"): + dev_lines = len( + (REPO_ROOT / "dev" / "skills" / skill / "SKILL.md") + .read_text(encoding="utf-8") + .splitlines() + ) + factory_lines = len( + (REPO_ROOT / "factory" / "skills" / skill / "SKILL.md") + .read_text(encoding="utf-8") + .splitlines() + ) + with self.subTest(skill=skill): + self.assertLessEqual( + abs(factory_lines - dev_lines), + 5, + f"{skill}: factory copy is {factory_lines} lines, dev is {dev_lines}", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/factory/.codex-plugin/plugin.json b/plugins/factory/.codex-plugin/plugin.json new file mode 100644 index 0000000..1aa8b81 --- /dev/null +++ b/plugins/factory/.codex-plugin/plugin.json @@ -0,0 +1,24 @@ +{ + "name": "factory", + "version": "0.1.0", + "description": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", + "author": { + "name": "Tobrun" + }, + "repository": "https://github.com/tobrun/workflow", + "skills": "./skills/", + "interface": { + "displayName": "Factory", + "shortDescription": "Run scope, scope-review, build, and ship unattended.", + "longDescription": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", + "developerName": "Tobrun", + "category": "Developer Tools", + "capabilities": [ + "Interactive", + "Write" + ], + "defaultPrompt": [ + "Take this request through scope, then run it unattended to a pull request." + ] + } +} diff --git a/plugins/factory/README.md b/plugins/factory/README.md new file mode 100644 index 0000000..abd15b8 --- /dev/null +++ b/plugins/factory/README.md @@ -0,0 +1,3 @@ +# factory for Codex + +Generated from `factory/` by `scripts/build_codex_plugin.py`. Do not edit this directory directly. diff --git a/plugins/factory/references/architecture.md b/plugins/factory/references/architecture.md new file mode 100644 index 0000000..e1f6fb6 --- /dev/null +++ b/plugins/factory/references/architecture.md @@ -0,0 +1,92 @@ +# Architecture Overview + +`docs/architecture.md` in the consuming project: one repo-tracked file describing the system at high level - what it is for, which components make it up, how they talk, and where its boundaries with the outside world lie. +It is the page a new engineer reads on day one and an agent reads before touching anything, and it exists because every run otherwise rebuilds that picture from the file tree, gets it subtly wrong, and never writes it down. + +It is not a registry and not a ledger: the decisions, contracts, and dependency rules live in their own files with their own notation and consumers. +This file is plain prose and tables at the altitude of the whole system, and nothing in it is machine-enforced beyond staying current with the code it describes. + +## Notation + +``` +# Architecture + +Purpose: one paragraph - what the system does, for whom, and the shape of a request through it. +Captured: 2026-09-15 (full, ship) - Updated: 2026-09-18 (commit 4f2a1c9) + +## Components + +| Component | Responsibility | Lives at | Talks to | +| --------- | -------------- | -------- | -------- | +| api | HTTP surface, auth, request validation | `src/api/` | app | +| app | use cases, orchestration, transactions | `src/app/` | domain, infra | +| domain | pure business rules and entities | `src/domain/` | nothing | +| infra | Postgres, payment-svc client, mail | `src/infra/` | domain (types only) | +| worker | queue consumers for async jobs | `src/worker/` | app | + +## Flows + +### Checkout +1. `POST /checkout` (api) validates the cart and calls `placeOrder` (app). +2. app reserves stock in domain, writes the order via infra, enqueues `charge-order`. +3. worker charges through payment-svc and emails the receipt. + +## Boundaries + +| Boundary | Kind | Owned by | Notes | +| -------- | ---- | -------- | ----- | +| Postgres | store | infra | single database, migrations in `db/migrations/` | +| payment-svc | external HTTP | infra | webhooks arrive at `POST /webhooks/payment` (api) | +| Redis queue | queue | worker | at-least-once delivery, consumers are idempotent | + +## Cross-cutting + +Auth: JWT checked in api middleware, `requireUser`; config: `src/config.ts` from env only; observability: structured logs, OpenTelemetry traces from api and worker. + +## Entry points + +`src/api/main.ts` (HTTP server), `src/worker/main.ts` (queue consumer), `scripts/migrate.ts`. +``` + +Five sections in this order - Components, Flows, Boundaries, Cross-cutting, Entry points - under a Purpose paragraph and a dated `Captured` / `Updated` line. +Every component names the path it lives at, in backticks - that is what lets the checker tell a stale overview from a current one. +Flows are the three to six journeys that explain why the components exist, as numbered steps naming the component at each hop; not every endpoint, not every function. +Boundaries are everything the system does not own: stores, queues, external services, files, clocks worth naming. + +## What belongs here + +The level a new engineer needs on day one and a reviewer needs before reading a diff: components and their responsibilities, the main flows, the boundaries, the cross-cutting mechanisms, the entry points. +Not here: function-level detail, per-endpoint lists, configuration values, decisions and their alternatives, boundary guarantees, allowed dependency edges - those have their own files. +Keep it under roughly 150 lines; an overview that needs more is describing more than one system, or describing it too closely. + +## The checker + +`architecture-check.py` (in this toolkit's shared `scripts/`) only checks that the overview is still true of the code, nothing more: + +```bash +python3 {skill-root}/../../scripts/architecture-check.py docs/architecture.md [--touched file ...] +``` + +- Exit 2: the file is missing - the caller runs an initial capture, never a fix agent. +- Exit 1: violations, one per line - a required section missing, a backticked path that no longer exists, or a `--touched` file no component's path covers (`uncharted`). +- Exit 0: the overview is current for what was checked. + +## Initial capture + +When a skill needs the overview and the file is absent, capture the whole system in one pass before continuing - a partial overview that only covers the current change misleads the next reader more than none. +Dispatch read-only explorers in a single message, one per top-level source area (the directories under the source root, or the packages of a monorepo), each reporting for its area: components with paths and responsibilities, what they call and what calls them, external boundaries they touch, entry points, and the cross-cutting mechanisms they participate in. +Merge their results into the notation above, resolving each flow end to end across areas, then loop the checker until it exits 0. +Write the `Captured` line with the date and the skill, tell the user the overview is new and where it is, and commit it as its own `docs(architecture): capture the system overview` commit. +Never fabricate a component or flow to make the overview look complete; an area the explorers could not characterize is listed with `unverified` in its responsibility cell, so the next reader knows to look. + +## Continuous updates + +The overview changes when the system changes, in the same commit or batch: + +- **build** updates it as part of the change set that adds, removes, renames, or moves a component, changes a flow, or adds a boundary, and runs the checker before that change set's commit. +- **commit** runs the checker over every batch (its sync-checks step 4d): a touched file that is uncharted, a stale path, or a flow or boundary the diff visibly changed is a targeted edit committed as `docs(architecture): ...`, never a rewrite. +- **scope** starts its prior-art explorers from the overview, and names the components and flows a change will add or reshape in the spec's scope section so build knows what to update. +- **ship**'s architecture lens reads it for orientation - where things live and how they talk - and treats a diff that reshapes a component or flow without updating the overview as a finding. + +Edits are minimal and factual: change the row, step, or cell that is now wrong, refresh the `Updated` line, leave the rest alone. +A section rewritten in a run that touched one component is a sign the run drifted into documentation work it was not asked for. diff --git a/plugins/factory/references/ci-parity.md b/plugins/factory/references/ci-parity.md new file mode 100644 index 0000000..382ac06 --- /dev/null +++ b/plugins/factory/references/ci-parity.md @@ -0,0 +1,52 @@ +# CI Parity and PR Follow-through + +The local quality loop is not complete while a required pull-request check is +known red. Spec validation and feature E2E prove the promised change; CI parity +proves the repository's merge gate against the exact checkout being proposed. + +## Discover the merge gate + +Read the repository's pull-request workflows under `.github/workflows/` and +any scripts they call. Build a list of project-owned commands from required +jobs: tests, lint/type/build, generated-artifact checks, screenshot/report +suites, packaging, and repository validation. + +Do not attempt to reproduce GitHub-owned setup actions locally. Reproduce the +project command after performing its documented local setup. Prefer a +repository-provided aggregate target when it covers the same jobs. + +Record the commands and outcomes in the plan's `implementation-notes.md`. + +## Run before proposing a PR + +Run every reproducible project-owned required-check command against the final +checkout, after feature E2E and after any hardening edits. + +- A failure is work to fix, including a test described as flaky, unrelated, or + pre-existing. Diagnose and remove its nondeterminism; do not add retries, + sleeps, or looser assertions. +- "Pre-existing" is not a waiver. Prove it by running the same command on the + merge base in an isolated worktree. If the base also fails and fixing it is + materially outside scope, present the evidence as a human call. Do not call + the branch PR-ready while the required check remains red. +- A command that cannot run locally because it needs GitHub-only credentials or + infrastructure is marked `remote-only`, with the reason. It is verified by + PR follow-through rather than silently skipped. + +Only offer or create the PR once all reproducible required checks are green and +all remote-only checks are identified. + +## Follow the PR to green + +After the user authorizes a push/PR and the PR exists: + +1. Watch required checks to a terminal state. +2. For each failure, fetch the failing job log, reproduce its project command + locally when possible, fix the root cause, run the narrow regression and the + full CI-parity command, commit, and push. +3. Repeat until every required check is green or a genuine human call is + reached. + +Do not stop at "CI restarted" when the user asked to finish or ship the change. +Do not rerun a failed job unchanged unless the log proves an external service +failure; deterministic failures require a code or test fix. diff --git a/plugins/factory/references/contracts.md b/plugins/factory/references/contracts.md new file mode 100644 index 0000000..8e5c3c2 --- /dev/null +++ b/plugins/factory/references/contracts.md @@ -0,0 +1,67 @@ +# Contract Registry + +`docs/contracts.md` in the consuming project: one repo-tracked file recording facts about behavior at boundaries - what one side guarantees, what the other side relies on, what can never happen. +It exists because reviewers and fixers invent boundary premises when none are recorded ("someone retries upstream", "this endpoint might return null"), and a wrong premise produces a wrong finding. + +Contracts are not decisions, and the difference is when consumers may read them. +A decision is a *justification* - it pre-forgives findings, so `docs/decisions.md` enters a run only after judgment (see the recommender contract in [decision-ledger.md](decision-ledger.md)). +A contract is a *premise* - a fact about what the code does at a boundary, the same class of context as a code map's intent and structure - so contracts enter *before* the walk. +That timing difference is why the two live in separate files: a walker retrieving contracts must not pull in recorded conclusions. + +## Notation + +``` +# Contracts + +## Payments + +C-webhook-retry: payment-svc delivers each webhook once; it never retries - callers own retry + guaranteed by: deliverWebhook (single attempt, no loop) + relied on by: billing worker, reconciliation job + (2026-08-21, webhook-delivery/spec.md) + +C-order-status: GET /orders never returns a status outside {pending, paid, failed} + guaranteed by: OrderStatus enum at serializeOrder ? verify: migration path for legacy rows + relied on by: mobile order screen + (2026-08-21, commit a1b2c3d) +``` + +Each entry: a `C-` kebab slug, one sentence stating the guarantee or invariant, a `guaranteed by:` line naming the enforcing code, a `relied on by:` line naming the sites that assume it, and a date + source. +When a recorded decision created the guarantee, the decision's slug is the source - `(2026-08-21, D-retry-ownership)` - which makes breaking the contract traceable to the choice behind it; a contract with no parent decision cites its spec, audit, or commit as usual. +Evidence marks and their maintenance follow [decision-ledger.md](decision-ledger.md) Notation exactly. + +Entries are filed under the area that owns the *guarantee* side - that's where the enforcing code lives, so that's where a change breaks it. +The `relied on by:` line makes the entry findable from the other side; a consumer searching by either module's name retrieves it. +One global file, one entry per contract: splitting per module would force either an arbitrary owner or a duplicate that drifts. + +## What qualifies + +A fact qualifies when it crosses a boundary and someone could plausibly assume it differently: retry and idempotency ownership, value domains and nullability of responses, delivery and ordering semantics, which side validates, which errors can actually surface. +Facts the type system already states at the boundary don't qualify - the compiler is their registry. +Module-internal behavior doesn't qualify - it has no second side to mislead. + +## How consumers use contracts + +- **Premise, pre-walk.** + Reviewers, verifiers, and fixers working near a recorded boundary read the matching entries before forming claims about the other side. + A claim that contradicts a cited contract needs to explain why the contract is wrong, not assume it away. +- **Refutation strength follows the evidence mark.** + A contract whose guarantee carries a citation can refute a finding outright. + A contract carrying `? verify:` can only weaken one - it records that somebody asserted the guarantee, not that anybody checked it. + Never refute on a `?`. +- **Violations are findings.** + A change to a guarantee site that breaks the stated guarantee, while reliance sites still assume it, is a first-class finding - the contract names exactly who gets hurt. + +## Who writes entries + +Contract writes are candidates until the user approves them, like every other durable write in this toolkit. + +- **Commit capture and maintenance.** + The commit skill captures new guarantees a commit establishes and re-checks recorded contracts whose guarantee sites the diff touches, with typed verdicts per the ledger-capture step of its sync-checks reference. + This check is what keeps citations trustworthy enough to refute anything. +- **Review refutations.** + When verification refutes a finding by discovering a boundary fact recorded nowhere - the guard, the value domain, the retry owner - that fact becomes a `C-` write candidate, so the next run inherits the premise instead of re-assuming. +- **Spec invariants.** + The spec's scope section (written by `scope`) records inputs, outputs, and invariants per change; the invariants that cross a boundary promote here at spec promotion time. +- **Audit recovery.** + An audit that observes an undocumented reliance ("billing assumes exactly-once but nothing guarantees it") records the gap as a `C-` candidate with the `?` on whichever side is unverified. diff --git a/plugins/factory/references/decision-ledger.md b/plugins/factory/references/decision-ledger.md new file mode 100644 index 0000000..2386611 --- /dev/null +++ b/plugins/factory/references/decision-ledger.md @@ -0,0 +1,120 @@ +# Decision Ledger + +`docs/decisions.md` in the consuming project: one repo-tracked file that accumulates design decisions across specs and audits. +Spec files are per-change and live in `.dev/{plan-name}/`; the ledger holds what outlives a change - decisions a future spec or audit could collide with. +It is what stops a settled question from being re-litigated every run. + +## Notation + +The marks, used identically here and in a spec's research section: `✓` chosen, `✗` rejected, `?` open, `⚠` accepted downside, `⊘` not doing (with the condition that would reopen it), each line with a short "because" clause. +This file is the canonical definition; consumers cite it rather than restating it. +Every entry keeps its `D-` slug and adds a date and a source - the spec file (`{plan-name}/spec.md`) or audit that produced it. + +### Evidence marks + +A because clause often leans on a claim about the world - "no abuse observed", "matches user expectations". +When the claim is an observed fact that would flip the decision if false, it carries one of two marks: + +- **A citation, in parentheses** - the claim was checked when written. + Code is cited by function name, never line number (line numbers rot on every edit; a function name survives until a rename). + Data is cited by its source - the log, query, metric, or dashboard. + A user statement is cited as `user ()`. + The entry's date scopes the citation: it means *verified then*, and consumers judge staleness at read time like everything else here. +- **`? verify: `** - nobody checked, even at write time. + Inline `?` after a claim is distinct from a line-leading `?` (an open alternative) by position. + +Judgment clauses - "operational burden", "adds a dependency for one call site" - are arguments, not evidence, and take no mark. +The mark is only signal while it's rare; marking every clause drowns it. + +Two maintenance rules keep the marks honest. +A flow that re-reads an entry and can check its `?` right now resolves it - the mark becomes the citation (a ledger write, subject to the consumer's candidate flow). +A citation that no longer resolves - the function renamed away, the dashboard gone - downgrades to `? verify:`; it is never silently removed. + +One `⊘` recorded about this notation itself: typed gap kinds (code-checkable vs. needs-runtime-data vs. needs-user) are not doing - no consumer dispatches on the distinction, so `verify:` stays free text; reopen when three flows would route differently on it. + +## Layout + +Principles at the top, then one section per codebase area. + +``` +# Decisions + +## Principles + +P-no-new-infra: no new infrastructure for an unproven need + promoted 2026-08-08 - recurred in D-response-caching, D-job-queue, D-metrics-store + +## Auth + +risk: high - domains: security, data (updated 2026-08-02, review oauth2-providers) + +D-session-length: How long do sessions last? (2026-07-14, login-sessions/spec.md) + ✓ 30 days sliding - matches user expectations ? verify: no support-ticket data checked ⚠ stolen-token window is long + ✗ 24h fixed - support burden from daily re-login + +D-rate-limit-login: Rate limit the login endpoint? (2026-08-02, security audit) + ⊘ not doing - no abuse observed (checkLoginAttempts audit log, none flagged); reopen if failed-login volume exceeds 100/day +``` + +## Area headers + +Two dated lines open each area section. +Both are priors, and a prior biases whoever holds it - so consumers read them only at the edges of a run: before dispatch (routing, gating how much autonomy a fix gets) or after findings exist (presentation, escalation). +Never during the walk itself, where they would shape what gets found - a walker who knows "high risk" starts seeing danger everywhere, and one who knows "the user knows this area cold" starts deferring. +Headers change how findings are *treated*, never whether things get *found*. + +`risk: - domains: (updated , )`. +Maintained mechanically from commit `Severity:`/`Risk:` trailers by the commit skill and by review runs, and its two halves age differently. +**Domains accumulate** - they're sticky facts about the area (auth touches tokens forever), so new `Risk:` values union in and are never removed mechanically; only a human prunes one, when the area genuinely sheds it. +**The level overwrites** - it's an observation about the most recent trailer-carrying changes (max of their `Severity:` values), not an all-time high-water mark. +Never ratchet: an area whose risky code was removed must be able to come back down, and the date tells consumers how stale the observation is. +Commits without severity trailers leave the level untouched. +A section without a risk line carries no signal - treat it as unassessed, not as safe. + +## Promoting principles + +A `P-` entry names a rationale that has recurred. +When the same because-clause logic picks or rejects alternatives in a third decision, promote it: kebab slug, one-line statement, the decisions it recurred in. +Never author a principle ahead of recurrence - three citations is the bar. +Once named, cite it by ID (`✗ Redis - P-no-new-infra`); "violates P-no-new-infra" replaces re-deriving the argument in reviews and audits. + +## What gets promoted + +From a finished spec or audit, copy the entries that pass this test: **would a future spec or audit in this area need to know this was decided?** +Approach choices, `⊘` lines with reopen conditions, and accepted-`⚠` tradeoffs usually pass; change-local trivia (a variable name, an internal function split) doesn't. +Copy entries verbatim with date + source; don't rewrite them. + +## Recommender contract + +For any flow that emits recommendations. +The order is the point: + +1. **The contract starts after judgment.** + A consumer skill brings the ledger into scope only at its reporting step - findings formed, not yet presented. + Analysis that reads recorded conclusions before forming its own inherits them instead of testing them; a fresh pass that collides with a recorded decision is signal, not waste. + When wiring a new consumer, sequence it so the ledger's first mention is its first use - earlier phases never name it, and never carry a "don't read it yet" guard, which only advertises the file where it must stay out of scope. +2. **Reconcile.** + Grep the ledger sections for the touched areas and classify every finding that collides with an entry: + - `still-holds` - a `⊘` or accepted `⚠` covers the finding and its reopen condition isn't met. + Suppress the finding but report the check ("D-rate-limit-login still holds - failed logins ~20/day"). + A silent suppression is indistinguishable from never having checked. + If the reopen condition can't be verified this run, the report carries the gap instead of implying a check - "still holds `? verify:` current failed-login volume" - a suppression resting on unverified evidence is still a suppression, but says so. + - `reopened` - the entry's reopen condition now holds. + Raise it citing the condition and the evidence that tripped it, not as a fresh recommendation. + - `diverged` - the fresh analysis reached a different conclusion than a recorded `✓` or `⊘`, and no reopen condition explains it. + Surface the disagreement as its own item: either the old decision missed something or the new analysis lacks its context. + Never silently suppress, never silently override. + + Reconcile also maintains evidence marks on the entries it touched, per Notation; those are ledger writes and wait for step 4 like any other. + + Findings with no collision pass through unchanged. + The reconcile report also states whether the pass was `clean` or `contaminated` - contaminated meaning recorded conclusions entered context before findings were formed (an unlucky grep, a file read that pulled them in). + A suppression from a contaminated pass proves nothing; say which findings were exposed. +3. **Recover.** + The walk saw decisions nobody recorded - timeouts, retry counts, validation gaps, structural choices, defaults of any kind. + For each one worth remembering (promotion test), ask what problem it was solving and record a ledger entry with the answer - or `no known problem - unexamined default`, which marks a decision nobody made and invites ratification. + Recovery rides along with reporting; it is not a separate pass over the code. +4. **Write back.** + After the user responds to findings: a declined recommendation becomes a `⊘` line with a concrete reopen condition, dated, sourced to this audit - declined always gets written; that's what makes the next run stateful. + An accepted one becomes a `✓` entry if it passes the promotion test, or warrants a recommended scope run if it carries enough decisions to need one. + A recommendation that's real but needs a call the run can't make becomes an `[open]` entry carrying the alternatives - escalations get a durable home instead of dying in a run directory. diff --git a/plugins/factory/references/dependency-rules.md b/plugins/factory/references/dependency-rules.md new file mode 100644 index 0000000..0e0ca84 --- /dev/null +++ b/plugins/factory/references/dependency-rules.md @@ -0,0 +1,47 @@ +# Dependency Rules + +`docs/dependencies.md` in the consuming project: one repo-tracked file declaring which modules may depend on which, in a shape a checker can parse and agents cannot argue with. +It exists because prose architecture rules soften into guidelines inside a long context; a checker's exit code does not. +`scope` drafts and updates the rules when a change touches module boundaries; `ship` enforces them in its gauntlet phase with a generated checker. + +## Notation + +A human-readable header, then one fenced `rules` block the checker parses: + +```` +# Dependencies + +Domain logic stays pure; the UI reaches it only through the app layer. + +```rules +[modules] +domain = src/domain/** +app = src/app/** +ui = src/ui/** +infra = src/infra/** + +[allowed] +app -> domain +ui -> app +infra -> domain +``` +```` + +Semantics, kept deliberately small: + +- `[modules]` names each module and binds it to one or more path globs (comma-separated). Every source file the checker scans must match exactly one module; a file matching none or several is itself a violation, so the map stays honest. +- `[allowed]` lists the permitted dependency edges as `from -> to, to`. **Anything not listed is forbidden** - an allowlist, not a denylist, so a new dependency is a deliberate edit to this file, never a drive-by import. +- Dependencies within a module are always allowed. Edges are not transitive: `ui -> app` and `app -> domain` do not grant `ui -> domain`; write the edge if it is wanted. +- A dependency is any static reference the language makes checkable: imports, includes, requires. Runtime indirection (dependency injection, events) is invisible to the checker by design - that is what makes inverting a dependency the standard fix. + +## Who does what + +- **scope** drafts the rules. A change that adds a module, adds an edge, or is blocked by an existing edge lands in the spec as a decision (`D-` entry with the alternatives: add the edge, invert the dependency, insert an interface, split the module). The chosen resolution edits this file as part of a change set - the edit is visible in review, never implicit. +- **ship** enforces them in its gauntlet phase. Its checker parses the `rules` block, maps changed files to modules, extracts their static dependencies, and fails on any edge not in `[allowed]`. Fix agents resolve violations by changing the code (invert, interface, split), never by editing this file - a rule change is a decision that belongs to a spec, not to a fix loop. +- **ship**'s architecture lens treats this file as settled context: a diff that conforms needs no boundary debate; a diff that edits the rules is reviewed as the decision it is. + +## What belongs here + +Module-level edges only. +Function-level or file-level rules drown the signal and rot fast; the compiler and `docs/contracts.md` cover finer grain. +A project without meaningful module boundaries yet does not need this file - `ship` skips the check when the file is absent, and says so rather than inventing rules. diff --git a/plugins/factory/references/factory-run.md b/plugins/factory/references/factory-run.md new file mode 100644 index 0000000..db3ecb6 --- /dev/null +++ b/plugins/factory/references/factory-run.md @@ -0,0 +1,59 @@ +# Factory Run Protocol + +Owned by the `run` skill. Every phase copy (`scope`, `scope-review`, `build`, `ship`) reads this file for the result envelope and the unattended policy; `run/SKILL.md` also reads it for the run state schema and the launch prompt shape. + +## Run state + +`.dev/{plan}/factory-run.json`, written only by the `run` skill, via `run-state.py`. It exists from `init` on, before any branch. Fields: + +- `plan`, `request`, `base` (the default branch at `init`). +- `branch`, `spec_sha256`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `dirty_files` - all written at `handoff`. +- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. +- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. +- `repairs`: an ordered list of `{phase, attempt, description, files: [], evidence: []}`. + +## Result envelope: `factory.result/1` + +Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its last action, one file per attempt. The orchestrator reads it leniently: a missing or unparseable file is a failed attempt with that reason, and an unrecognized extra key is ignored rather than rejected. + +```json +{ + "schema": "factory.result/1", + "phase": "build", + "skill_path": "/abs/path/to/factory/skills/build/SKILL.md", + "status": "done", + "reason": "", + "artifacts": ["path/or/description"], + "counts": {}, + "auto_decided": ["escalation text -> option chosen"], + "stop": { "kind": "secret.found", "action": "what a person must do" } +} +``` + +- `schema` - always the literal `factory.result/1`. +- `phase` - the phase name. +- `skill_path` - the exact path the launch prompt passed for `{phase}-skill-root`'s `SKILL.md`, echoed back so a subagent having opened the file it was given is visible in a file the orchestrator already reads (D-skill-locating). +- `status` - `done`, `failed`, or `stopped`. +- `reason` - required when not `done`; the failure or stop reason in plain text. +- `artifacts` - paths or short descriptions of what the phase produced. +- `counts` - phase-specific numbers (scenario counts, decisions made), never judgment evidence on their own. +- `auto_decided` - escalations the phase answered itself under the unattended policy. +- `stop` - present only when `status` is `stopped`: `kind` is `secret.found` or `action.destructive`, `action` is the exact step a person must take. + +## The unattended policy + +After the go, no phase skill asks a person anything. An escalation that dev's version would ask a human about is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. + +`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. Every other phase copy, and `run/SKILL.md`'s own person-routing lines, live inside `` / `` blocks or outside the scan root entirely. + +## Launch prompt shape + +Each phase subagent's prompt carries, in order: + +1. The phase's skill path: the absolute path to `factory/skills/{phase}/SKILL.md`, with the instruction to read `{phase}-skill-root` as that path's parent directory (a subagent reading a file cannot resolve `{phase}-skill-root}`-style placeholders on its own). +2. The plan name and the plan directory's absolute path. +3. The attempt number. +4. On a relaunch: the previous attempt's reason and explicit guidance naming what it did and what is required instead - never "try again". +5. The scratch root for this attempt: `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/`. +6. The result path this attempt must write: `.dev/{plan}/results/{phase}-{attempt}.json`. +7. The instruction that this phase never launches another phase and, after the go, never asks a person anything - decide under the unattended policy above and record it. diff --git a/plugins/factory/references/jira.md b/plugins/factory/references/jira.md new file mode 100644 index 0000000..9ccda17 --- /dev/null +++ b/plugins/factory/references/jira.md @@ -0,0 +1,115 @@ +# Jira mirror reference + +This reference is used only when the consuming repository opts into Jira in +`.dev/config.json`. The `.dev/` files remain the source of truth; Jira mirrors +the records identified by the keys persisted in those files. + +## Configuration and activation + +Read `.dev/config.json` before doing any Jira work. The supported shape is: + +```json +{ + "jira": { + "enabled": true, + "site": "acme.atlassian.net", + "project": "PROJ" + } +} +``` + +`jira.project` is required when enabled. `jira.site` is optional and should be +passed to `acli` when present. If the file is absent, malformed, Jira is +missing, or `jira.enabled` is false, the workflow is pure-local. In the +disabled case do not mention Jira, ask a Jira question, or invoke `acli`. + +When enabled, check that `acli` is installed and authenticated before the first +operation. A useful check is: + +```text +acli jira auth status [--site SITE] +``` + +If the check fails, stop and ask the user to install or authenticate `acli`, or +disable Jira sync, per the failure protocol below. + +## Command shapes + +The exact flags may vary slightly with the installed `acli` release. Preserve +the intent and verify every created issue with `workitem view`. Treat a +non-zero exit, invalid JSON, missing key, or ambiguous result as a failure. + +List open Initiatives for the configured project: + +```text +acli jira workitem search --project PROJ --type Initiative --status open --json [--site SITE] +``` + +Ask the user to choose one returned Initiative. If none is returned, stop with +a clear message that an existing open Initiative is required; never create one. + +Create the spec Epic linked to the chosen Initiative, then verify it: + +```text +acli jira workitem create --project PROJ --type Epic --summary SUMMARY --parent INITIATIVE-1 --json [--site SITE] +acli jira workitem view EPIC-1 --json [--site SITE] +``` + +Create a task issue for a change set under the Epic, then verify it: + +```text +acli jira workitem create --project PROJ --type Task --summary SUMMARY --parent EPIC-1 --json [--site SITE] +acli jira workitem view TASK-1 --json [--site SITE] +``` + +Discover the project's available transitions once per issue type per run, from +a representative issue, and reuse the names; re-discover only after a failed or +ambiguous transition. Match the displayed names exactly and require one +unambiguous transition for the requested state: + +```text +acli jira workitem transition-list --issue EPIC-1 --json [--site SITE] +acli jira workitem transition --issue EPIC-1 --transition "In Progress" --json [--site SITE] +``` + +Use the same discovery and transition sequence for task issues. + +When a change set is superseded by a spec revision, transition the old issue +to its available closed state and add the required comment: + +```text +acli jira workitem transition-list --issue TASK-1 --json [--site SITE] +acli jira workitem transition --issue TASK-1 --transition "Done" --json [--site SITE] +acli jira workitem comment --issue TASK-1 --body "superseded by change set 3" --json [--site SITE] +``` + +## Persistence and timing + +After the Epic is created, write `Jira: PROJ-1` directly under the spec title +in `spec.md`. After each task issue is created, write its key directly under +the matching change set in the spec's change plan. + +`scope` lists Initiatives during its interview and creates the Epic only +after the change plan is final. It then creates one Task per change set; a +re-run creates Tasks only for change sets that don't carry a key yet. On +supersede, close and comment the old issue before creating and persisting the +replacement issue. + +build transitions the persisted Epic to In Progress at the start. The +orchestrator transitions each persisted change-set issue to In Progress +immediately when its wave is dispatched, and to Done only after the change set +is verified and its commit lands. + +## Failure protocol + +Jira is first-class when enabled. Any failed auth check, command, JSON parse, +verification, missing key, missing Initiative, unknown transition, or +ambiguous transition stops the skill immediately. Explain the failed +operation and ask the user how to proceed. Never silently skip Jira, continue +with divergent local-only state, invent an issue key, or mark a transition +successful without verification. + +The Epic is not polled or closed by the skills. When build creates the work +branch and when ship opens the PR, include the Epic key in the branch name and +at the start of the PR title. A documented Jira automation rule closes the Epic +after the linked PR is merged. diff --git a/plugins/factory/references/plan-layout.md b/plugins/factory/references/plan-layout.md new file mode 100644 index 0000000..f1bb69d --- /dev/null +++ b/plugins/factory/references/plan-layout.md @@ -0,0 +1,34 @@ +# Plan Directory Layout + +`.dev/{plan-name}/` is the durable home of one change, named by a kebab-case slug for the outcome. +This reference owns the layout, the locating convention, and the diff scope; skills state only their own role. + +## Files and owners + +| File | Written by | Read by | +| ---- | ---------- | ------- | +| `request.md` | the `run` skill, before `scope` starts | `scope` | +| `spec.md` | `scope`; `scope-review` (verified refinements only) | everyone downstream | +| `spec-review_N.md` | `scope-review` (next free index) | `scope` (remediation), re-reviews | +| `implementation-notes.md` | `build` (append-only) | `ship`, `to-pitch`, `to-quiz` | +| `review_N.md` | `ship` (next free index) | `scope` (remediation), re-reviews | +| `pr.md` | `ship` (phase 3, overwritten per run) | the PR tool via `--body-file`; re-runs | +| `factory-run.json` | the `run` skill only | the `run` skill's judgment loop | +| `results/{phase}-{attempt}.json` | the phase skill, as its last action | the `run` skill's judgment loop | +| `.dev/config.json` | the user | any skill with Jira behavior | + +Each producing skill also renders an HTML companion under `/tmp/{project-slug}/reports/` per [reporting.md](reporting.md), named by that skill. +One writer per file; every other skill only reads. + +## Locating the plan directory + +Match the current branch name to a `.dev/{plan-name}/` slug; fall back to commit messages, then to the only directory in a plausible state for the skill (e.g. the only spec whose change sets aren't done). +Ask which one only if more than one is a plausible match; otherwise proceed without waiting. + +## Diff scope + +Skills that operate on "the change" (`ship`) scope to the local branch against the default branch (`git merge-base HEAD origin/{default}`), diffed with local git only (never `gh pr diff`), excluding lockfiles, build output, minified files, binaries, fonts, and snapshots: + +``` +':!*lock*' ':!go.sum' ':!dist/' ':!build/' ':!*.min.*' ':!*.map' ':!*.png' ':!*.jpg' ':!*.gif' ':!*.webp' ':!*.woff*' ':!*.ttf' ':!**/__snapshots__/' +``` diff --git a/plugins/factory/references/reporting.md b/plugins/factory/references/reporting.md new file mode 100644 index 0000000..5c14c18 --- /dev/null +++ b/plugins/factory/references/reporting.md @@ -0,0 +1,23 @@ +# Report Artifacts + +Every producing skill renders its output as self-contained HTML under `/tmp/{project-slug}/reports/`. +This reference owns the shared etiquette; each skill states only its own output filename and data shape. + +## Rendering + +- Copy the skill's template to the output path, replacing **only the data block** between its `*_DATA_START` / `*_DATA_END` markers. + The rendering engine below the markers is generic and reads only that shape - never touch it on a data refresh. +- Before first authoring or restyling a template, load an installed artifact or frontend design skill; a plain data refresh on an existing template doesn't need it again. +- Open the rendered file with the host's browser integration when available; otherwise give the user a clickable local path. + Do not fail solely because GUI launch is unavailable. + +## Publishing + +The local file is the deliverable. +Publish with an artifact-publishing tool only when the user asks for a shareable link, using a stable per-skill favicon and a title and description naming the artifact's subject. +Never publish unprompted, and if the host has no publisher, the local HTML remains the deliverable - say so instead of apologizing. + +## Reading another skill's report + +Consumers of a rendered report read its data block, not the whole file, and extract only the fields they need. +In particular, `E2E_DATA` embeds one base64 `dataUri` per screenshot step: never ingest those payloads - scenario ids, titles, statuses, and the `summary` are the useful content. diff --git a/plugins/factory/scripts/architecture-check.py b/plugins/factory/scripts/architecture-check.py new file mode 100755 index 0000000..3f2417a --- /dev/null +++ b/plugins/factory/scripts/architecture-check.py @@ -0,0 +1,135 @@ +#!/usr/bin/env python3 +"""Check that docs/architecture.md is present, well-formed, and current. + +Usage: + python3 architecture-check.py docs/architecture.md [--root DIR] [--touched FILE ...] + +Exit 2 when the file is missing (run an initial capture), 1 with one violation +per line when the overview is stale or malformed, 0 when it is current for what was +checked. Violations: a required section missing, a backticked path that does +not exist under --root, and a --touched file that no component's "Lives at" +path covers ("uncharted"). +""" + +from __future__ import annotations + +import argparse +import glob +import re +import sys +from pathlib import Path + +SECTIONS = ["Components", "Flows", "Boundaries", "Cross-cutting", "Entry points"] +HEADER = re.compile(r"^##\s+(.+?)\s*$", re.M) +BACKTICK = re.compile(r"`([^`\n]+)`") +PATH_LIKE = re.compile(r"^[^\s]+(/[^\s]*|\.[A-Za-z0-9]{1,6})$") +TABLE_ROW = re.compile(r"^\|(.+)\|\s*$") + + +def sections(text: str) -> dict[str, str]: + found: dict[str, str] = {} + matches = list(HEADER.finditer(text)) + for index, match in enumerate(matches): + end = matches[index + 1].start() if index + 1 < len(matches) else len(text) + found[match.group(1)] = text[match.end():end] + return found + + +def path_exists(root: Path, raw: str) -> bool: + candidate = raw.strip().rstrip("/") + if any(ch in candidate for ch in "*?["): + return bool(glob.glob(str(root / candidate), recursive=True)) + return (root / candidate).exists() + + +def table_rows(section: str) -> list[list[str]]: + rows = [] + for line in section.splitlines(): + match = TABLE_ROW.match(line) + if not match: + continue + cells = [cell.strip() for cell in match.group(1).split("|")] + if all(set(cell) <= set("-: ") for cell in cells): + continue + rows.append(cells) + return rows[1:] if rows else [] + + +def component_paths(section: str) -> dict[str, list[str]]: + paths: dict[str, list[str]] = {} + for cells in table_rows(section): + if not cells: + continue + paths[cells[0]] = [p.strip().rstrip("/") for p in BACKTICK.findall(" ".join(cells[1:])) if PATH_LIKE.match(p.strip())] + return paths + + +def covered(touched: str, patterns: list[str]) -> bool: + for pattern in patterns: + if any(ch in pattern for ch in "*?["): + base = pattern.split("*", 1)[0].rstrip("/") + if base and touched.startswith(base + "/"): + return True + elif touched == pattern or touched.startswith(pattern + "/"): + return True + return False + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("map") + parser.add_argument("--root", default=".") + parser.add_argument("--touched", nargs="*", default=[]) + args = parser.parse_args() + + root = Path(args.root).resolve() + map_path = Path(args.map) + if not map_path.exists(): + print(f"missing: {map_path} - run an initial capture") + return 2 + text = map_path.read_text(encoding="utf-8", errors="replace") + violations: list[str] = [] + + if not re.search(r"^Purpose:", text, re.M): + violations.append("section: no 'Purpose:' line") + if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.M): + violations.append("section: no dated 'Captured:' line") + found = sections(text) + order = [name for name in found if name in SECTIONS] + for name in SECTIONS: + if name not in found: + violations.append(f"section: missing '## {name}'") + if order != [name for name in SECTIONS if name in found]: + violations.append(f"section: out of order, expected {', '.join(SECTIONS)}") + + for raw in BACKTICK.findall(text): + candidate = raw.strip() + if PATH_LIKE.match(candidate) and not path_exists(root, candidate): + violations.append(f"stale: `{candidate}` does not exist under {root}") + + components = component_paths(found.get("Components", "")) + if "Components" in found and not components: + violations.append("section: Components table has no rows") + for name, paths in components.items(): + if not paths: + violations.append(f"component: '{name}' names no backticked path in its row") + + all_paths = [p for paths in components.values() for p in paths] + for touched in args.touched: + relative = touched.strip() + try: + relative = str(Path(touched).resolve().relative_to(root)) + except ValueError: + pass + if not covered(relative, all_paths): + violations.append(f"uncharted: {relative} is covered by no component's path") + + if violations: + print("\n".join(violations)) + return 1 + print(f"architecture-check: ok ({len(components)} components, {len(args.touched)} touched files covered)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/factory/scripts/skill-metrics.py b/plugins/factory/scripts/skill-metrics.py new file mode 100755 index 0000000..d9d2b7b --- /dev/null +++ b/plugins/factory/scripts/skill-metrics.py @@ -0,0 +1,333 @@ +#!/usr/bin/env python3 +"""Measure one skill run: time, tokens, agents, tool calls, and git delta. + +Usage: + python3 skill-metrics.py start {skill} + python3 skill-metrics.py end {skill} [--count key=value ...] + +`start` snapshots the git state and locates the session transcript under +$CLAUDE_CONFIG_DIR (default ~/.claude), anchoring on the transcript line that +invoked the skill. `end` sums everything from that anchor across the main +transcript and every subagent transcript the run spawned, diffs git against +the snapshot, prints a markdown table, and appends a row to .dev/metrics.jsonl +so later runs can be compared against earlier ones. Every value is measured; +nothing here is narrated from memory. Missing transcript or git degrade to +"n/a" rather than failing the run. +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import statistics +import subprocess +import sys +import time +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +TEST_FILE = re.compile(r"(^|/)(tests?|specs?|__tests__)(/|$)|[._-](test|spec)s?\.[A-Za-z]+$|^test_.*\.py$") +TOKEN_KEYS = ("input_tokens", "cache_read_input_tokens", "cache_creation_input_tokens", "output_tokens") + + +def now_iso() -> str: + return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.%fZ") + + +def parse_iso(value: str) -> float: + return datetime.strptime(value[:19], "%Y-%m-%dT%H:%M:%S").replace(tzinfo=timezone.utc).timestamp() + + +def run_git(*args: str) -> str | None: + try: + return subprocess.run(["git", *args], capture_output=True, text=True, check=True).stdout + except (subprocess.CalledProcessError, FileNotFoundError): + return None + + +def repo_root() -> Path: + top = run_git("rev-parse", "--show-toplevel") + return Path(top.strip()) if top else Path.cwd() + + +def state_dir() -> Path: + path = Path("/tmp") / repo_root().name / "metrics" + path.mkdir(parents=True, exist_ok=True) + return path + + +# --- transcript ------------------------------------------------------------- + +def find_transcript() -> Path | None: + config = Path(os.environ.get("CLAUDE_CONFIG_DIR", "~/.claude")).expanduser() + slug = re.sub(r"[^A-Za-z0-9]", "-", str(Path.cwd())) + project = config / "projects" / slug + if not project.is_dir(): + return None + files = sorted(project.glob("*.jsonl"), key=lambda p: p.stat().st_mtime, reverse=True) + return files[0] if files else None + + +def iter_lines(path: Path, start: int = 0): + with path.open(encoding="utf-8") as handle: + for index, raw in enumerate(handle): + if index < start: + continue + try: + yield index, json.loads(raw) + except json.JSONDecodeError: + continue + + +def find_anchor(path: Path, skill: str) -> tuple[int, str | None]: + """Line index and timestamp of the last invocation of this skill.""" + marker = re.compile(rf"/(?:[\w-]+:)?{re.escape(skill)}") + anchor, stamp, count = 0, None, 0 + for index, obj in iter_lines(path): + count = index + 1 + if obj.get("type") != "user": + continue + content = (obj.get("message") or {}).get("content") + if isinstance(content, str) and marker.search(content): + anchor, stamp = index, obj.get("timestamp") + if stamp is None: + anchor = count + return anchor, stamp + + +def aggregate(path: Path, start: int, since: float | None) -> dict: + """Token, turn and tool totals for one transcript from a line index.""" + usage: dict[str, dict] = {} + tools: Counter = Counter() + models: Counter = Counter() + first_stamp = None + for _, obj in iter_lines(path, start): + stamp = obj.get("timestamp") + if first_stamp is None and stamp: + first_stamp = stamp + if obj.get("type") != "assistant": + continue + message = obj.get("message") or {} + message_id = message.get("id") or obj.get("uuid") + if message.get("usage"): + usage[message_id] = message["usage"] + if message.get("model"): + models[message["model"]] = 1 + for block in message.get("content") or []: + if isinstance(block, dict) and block.get("type") == "tool_use": + tools[block.get("name", "?")] += 1 + if since is not None and first_stamp and parse_iso(first_stamp) < since: + return {} + tokens = {key: sum(int(u.get(key) or 0) for u in usage.values()) for key in TOKEN_KEYS} + return {"turns": len(usage), "tokens": tokens, "tools": dict(tools), "models": sorted(models)} + + +def subagent_files(transcript: Path) -> list[Path]: + folder = transcript.with_suffix("") / "subagents" + return sorted(folder.glob("*.jsonl")) if folder.is_dir() else [] + + +# --- git --------------------------------------------------------------------- + +def numstat(ref: str | None) -> tuple[int, int, set[str]]: + args = ["diff", "--numstat"] + ([ref] if ref else []) + output = run_git(*args) or "" + added = removed = 0 + files: set[str] = set() + for line in output.splitlines(): + parts = line.split("\t") + if len(parts) != 3: + continue + added += int(parts[0]) if parts[0].isdigit() else 0 + removed += int(parts[1]) if parts[1].isdigit() else 0 + files.add(parts[2]) + untracked = (run_git("ls-files", "--others", "--exclude-standard") or "").split() + for path in untracked: + files.add(path) + try: + added += sum(1 for _ in Path(path).open(encoding="utf-8", errors="ignore")) + except OSError: + pass + return added, removed, files + + +def git_snapshot() -> dict: + head = run_git("rev-parse", "HEAD") + added, removed, files = numstat("HEAD" if head else None) + return {"head": head.strip() if head else None, "added": added, "removed": removed, "files": sorted(files)} + + +def git_delta(start: dict) -> dict | None: + head = start.get("head") + if not head: + return None + added, removed, files = numstat(head) + commits = run_git("rev-list", "--count", f"{head}..HEAD") + return { + "commits": int(commits.strip()) if commits and commits.strip().isdigit() else 0, + "files": len(files), + "test_files": sum(1 for f in files if TEST_FILE.search(f)), + "added": max(0, added - start["added"]), + "removed": max(0, removed - start["removed"]), + "baseline_dirty_files": len(start["files"]), + } + + +# --- commands ---------------------------------------------------------------- + +def cmd_start(skill: str) -> int: + transcript = find_transcript() + anchor, stamp = find_anchor(transcript, skill) if transcript else (0, None) + state = { + "skill": skill, + "started_at": stamp or now_iso(), + "anchored": stamp is not None, + "transcript": str(transcript) if transcript else None, + "anchor_line": anchor, + "git": git_snapshot(), + } + (state_dir() / f"{skill}.json").write_text(json.dumps(state, indent=2), encoding="utf-8") + where = f"transcript line {anchor}" if stamp else "now (no invocation marker found)" + print(f"metrics: {skill} run started, anchored at {where}") + return 0 + + +def fmt_tokens(tokens: dict) -> str: + return "in {} / cache read {} / cache write {} / out {}".format( + *(human(tokens[key]) for key in TOKEN_KEYS) + ) + + +def human(value: float) -> str: + for unit, size in (("M", 1_000_000), ("k", 1_000)): + if value >= size: + return f"{value / size:.1f}{unit}" + return str(int(value)) + + +def duration(seconds: float) -> str: + minutes, secs = divmod(int(seconds), 60) + hours, minutes = divmod(minutes, 60) + return f"{hours}h {minutes:02d}m" if hours else f"{minutes}m {secs:02d}s" + + +def total(tokens: dict) -> int: + return sum(tokens.get(key, 0) for key in TOKEN_KEYS) + + +def cmd_end(skill: str, counts: dict[str, str]) -> int: + state_file = state_dir() / f"{skill}.json" + if not state_file.exists(): + print(f"metrics: no start snapshot for {skill}; run `start {skill}` at invocation", file=sys.stderr) + return 1 + state = json.loads(state_file.read_text(encoding="utf-8")) + since = parse_iso(state["started_at"]) + elapsed = time.time() - since + + main: dict = {} + agents: list[dict] = [] + transcript = Path(state["transcript"]) if state.get("transcript") else None + if transcript and transcript.exists(): + main = aggregate(transcript, state["anchor_line"], None) + agents = [a for a in (aggregate(p, 0, since) for p in subagent_files(transcript)) if a] + agent_tokens = {key: sum(a["tokens"][key] for a in agents) for key in TOKEN_KEYS} + all_tokens = {key: main.get("tokens", {}).get(key, 0) + agent_tokens[key] for key in TOKEN_KEYS} + tools = Counter(main.get("tools", {})) + for agent in agents: + tools.update(agent["tools"]) + delta = git_delta(state["git"]) + + record = { + "skill": skill, + "started_at": state["started_at"], + "ended_at": now_iso(), + "seconds": int(elapsed), + "anchored": state["anchored"], + "turns": main.get("turns", 0), + "agents": len(agents), + "tools": dict(tools), + "tokens_main": main.get("tokens", {}), + "tokens_agents": agent_tokens, + "tokens_total": total(all_tokens), + "git": delta, + "counts": counts, + } + ledger = repo_root() / ".dev" / "metrics.jsonl" + previous = [] + if ledger.exists(): + for line in ledger.read_text(encoding="utf-8").splitlines(): + try: + row = json.loads(line) + except json.JSONDecodeError: + continue + if row.get("skill") == skill: + previous.append(row) + ledger.parent.mkdir(parents=True, exist_ok=True) + with ledger.open("a", encoding="utf-8") as handle: + handle.write(json.dumps(record) + "\n") + + top_tools = ", ".join(f"{name} {n}" for name, n in tools.most_common(4)) or "none" + rows = [ + ("duration", duration(elapsed) + ("" if state["anchored"] else " (from start call, invocation marker not found)")), + ("orchestrator turns / tool calls", f"{main.get('turns', 0)} / {sum(tools.values())} ({top_tools})"), + ("agents dispatched", str(len(agents))), + ("tokens orchestrator", fmt_tokens(main["tokens"]) if main else "n/a (transcript not found)"), + ("tokens agents", fmt_tokens(agent_tokens) if agents else "0"), + ("tokens total", human(total(all_tokens)) if main else "n/a"), + ] + if delta: + rows.append(( + "git since start", + f"{delta['commits']} commits, {delta['files']} files (+{delta['added']}/-{delta['removed']}), " + f"{delta['test_files']} test files" + + (f"; {delta['baseline_dirty_files']} files were already dirty" if delta["baseline_dirty_files"] else ""), + )) + else: + rows.append(("git since start", "n/a (not a git repo)")) + for key, value in counts.items(): + rows.append((key.replace("_", " "), value)) + if previous and main: + med_tokens = statistics.median(r["tokens_total"] for r in previous if r.get("tokens_total")) + med_secs = statistics.median(r["seconds"] for r in previous) + change = (total(all_tokens) - med_tokens) / med_tokens * 100 if med_tokens else 0 + rows.append(( + f"vs previous {skill} runs", + f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " + f"this run {change:+.0f}% tokens", + )) + + width = max(len(name) for name, _ in rows) + print(f"## {skill} run metrics\n") + print(f"| {'metric'.ljust(width)} | value |") + print(f"| {'-' * width} | ----- |") + for name, value in rows: + print(f"| {name.ljust(width)} | {value} |") + print(f"\nledger: {ledger}") + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + sub = parser.add_subparsers(dest="command", required=True) + sub.add_parser("start").add_argument("skill") + end = sub.add_parser("end") + end.add_argument("skill") + end.add_argument("--count", action="append", default=[], metavar="KEY=VALUE", + help="skill-specific measured counter to include, repeatable") + args = parser.parse_args() + if args.command == "start": + return cmd_start(args.skill) + counts = {} + for item in args.count: + if "=" not in item: + parser.error(f"--count expects KEY=VALUE, got {item!r}") + key, value = item.split("=", 1) + counts[key.strip()] = value.strip() + return cmd_end(args.skill, counts) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/factory/skills/build/SKILL.md b/plugins/factory/skills/build/SKILL.md new file mode 100644 index 0000000..abd6cfa --- /dev/null +++ b/plugins/factory/skills/build/SKILL.md @@ -0,0 +1,104 @@ +--- +name: build +description: Execute the change sets of a spec at .dev/{plan-name}/spec.md, proving every scenario with a test at its tagged layer across unit/integration/e2e, running independent change sets in parallel and looping until every change set is done and the e2e suite passes. Use when the user asks to build or implement a spec or its change sets. +--- + +# Build + +Execute every change set of a spec, proving each `tests:` scenario with a real test at its tagged layer, until all change sets are done and the e2e run is green. + +Input: `.dev/{plan-name}/spec.md` (from `scope`), located per [../../references/plan-layout.md](../../references/plan-layout.md). +If there is no `spec.md` or its change plan is empty, report `failed` with reason "no spec"; without one there are no layer-tagged scenarios to implement against. + +Read [references/layers.md](references/layers.md), [references/tests.md](references/tests.md), and [references/mocking.md](references/mocking.md) before implementing anything yourself; [references/parallel.md](references/parallel.md) points subagents at them. + +## Workflow + +1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. +2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. +3. For each wave, run its change sets in parallel per the same reference, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). +4. Move straight to the next wave. Never stop after one change set or wave to ask about review. +5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. +6. Then run the full e2e pass per "The e2e layer" below over the whole spec, and loop on failures until it is green. +7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. +8. After the e2e report and CI-parity gate, close per "Closing message". Build never pushes or opens a PR; that is `ship`'s phase 3. + +## Jira sync + +Read `.dev/config.json`; when `jira.enabled` is true, follow +[../../references/jira.md](../../references/jira.md) from before the first `acli` call - it owns the +command shapes, the transition timing, and the failure protocol. The +orchestrator alone invokes `acli`; subagent prompts and the `parallel.md` +contract do not change. + +With an absent or disabled config, perform no Jira behavior or mention. +If still on the default branch, create the work branch before the first commit, +named per jira.md when Jira is enabled; never push it and never open a PR - +`ship` pushes and opens the PR with evidence after its gauntlet and review. + +## Closing message + +Every build run first prints the measured run metrics, pasting the table verbatim: + +```bash +python3 {build-skill-root}/../../scripts/skill-metrics.py end build --count change_sets=N --count scenarios=N --count e2e_passed=N --count e2e_failed=N +``` + +Then it ends with the same two lines, in this order - a green run, a blocked gate, and a run with open deviations all get both: + +1. `Next step: run ship over this work, pointed at .dev/{plan-name}/implementation-notes.md and the {plan-name}-e2e-report.html.` Recommend it; never launch it yourself. +2. Then any question left for the user - a blocked gate, an unresolved deviation. + +A blocked gate never replaces line 1. +Neither does a failed e2e loop: say what is blocked, then still point at `ship`. + +## The change-set loop + +Each change set, whether you run it yourself or a subagent runs it, follows the same loop: + +- Test at the seams the spec's scope section declares, per [references/tests.md](references/tests.md); if the declared boundary is wrong or missing, follow its fallback and log the change under Deviations - do not stall on it. +- Implement in **vertical slices**: one scenario's behavior at a time, its test written before or right after the code - the enforced outcome is what matters, not the ritual order. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. +- Run the change set's own tests and typecheck continuously; once the change set is green, run the spec's Validation block verbatim - it is the wider suite plus typecheck/lint - and only report done when it passes clean. + +## Rules of the loop + +- **Every test must have been seen red.** A test that has never failed proves nothing: earn its green by writing it before the code, or by briefly breaking the behavior once after. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. +- **One slice at a time.** One seam, one behavior, one test, one minimal implementation per cycle. +- **Refactoring is not part of the loop.** It belongs to `ship`'s review phase. +- **Keep going.** A red test, a failing e2e scenario, or an edge case that contradicts the spec is work to do, not a reason to hand back. Fix it, log the deviation, continue. Stop early only when a blocking question makes further work unsafe or wasted. + +## The e2e layer + +E2E scenarios are proven by running the actual application against the **fully mocked environment** defined in [references/mocking.md](references/mocking.md#the-e2e-environment). Run this once per spec, after all change sets are committed, covering every `[e2e]` scenario across change sets. + +1. **Launch the app.** Invoke an installed `run` skill with the mocked environment configured when the host supports direct skill invocation. Otherwise inspect the repository's documented commands and start the app directly. When no safe launch command can be determined, record the decision (mock the launch, skip the affected `[e2e]` scenarios, or report `failed` naming what a person must supply) and continue. +2. **Drive it and capture evidence**, per scenario: + - `kind: "frontend"` - use available browser automation (the host browser integration or Playwright) to exercise the scenario, one screenshot per meaningful step, embedded as a base64 data URI. + - `kind: "non-frontend"` - capture the entity's real before/after state from the run's own output or fixtures. +3. **Never fabricate a screenshot or a data-model-state entry.** Both come from this actual run. +4. **Loop until green.** A failed scenario is a bug: diagnose it, fix the code (a new red-green cycle at the right layer), re-run and re-capture that scenario. Never flip a status to pass without a fresh capture. If a scenario fails three times on the same root cause, write what you found into Deviations and report the phase `failed` with the root cause as the reason. +5. Map the results onto `E2E_DATA` per [references/e2e-report.md](references/e2e-report.md) and render `templates/e2e-report.html` to `/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`, opening and publishing per [../../references/reporting.md](../../references/reporting.md). + +## Implementation notes + +Maintain `.dev/{plan-name}/implementation-notes.md`, appended after each change set completes, never written once at the end. +It is the shared state across waves - parallel change-set agents can't see each other's conversation, only this file and the code - and the evidence `ship`, `to-pitch`, and `to-quiz` read later. + +```markdown +## Change set {n}: {title} +- What was done: ... +- Seams tested: ... +- Tests added: {path::test name}, ... # or "none - {reason}"; the checker reads this line +- Deviations from spec: {edge case found} -> conservative choice made: {what/why} # only when a deviation occurred +``` + +This file is a short running log, not a rendered report. + +## Anti-patterns + +- **Horizontal slicing** - all tests first, then all implementation. Tests then verify an imagined shape and go insensitive to change. +- The other tells - implementation-coupled tests, tautological assertions, top-heavy testing - are defined in [references/tests.md](references/tests.md) and [references/layers.md](references/layers.md); flag and fix them on sight. + +## Factory context + +Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. diff --git a/plugins/factory/skills/build/agents/openai.yaml b/plugins/factory/skills/build/agents/openai.yaml new file mode 100644 index 0000000..2b1ae55 --- /dev/null +++ b/plugins/factory/skills/build/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "Build" + short_description: "Execute a spec test-first through e2e" + default_prompt: "Use $factory:build to execute the current spec test-first and verify it end to end." +policy: + allow_implicit_invocation: false diff --git a/plugins/factory/skills/build/references/e2e-report.md b/plugins/factory/skills/build/references/e2e-report.md new file mode 100644 index 0000000..b8f00e1 --- /dev/null +++ b/plugins/factory/skills/build/references/e2e-report.md @@ -0,0 +1,50 @@ +# E2E_DATA Schema + +The shape to populate in `templates/e2e-report.html` between the `E2E_DATA_START` / `E2E_DATA_END` markers. +Replace the whole object per the shared etiquette in [../../../references/reporting.md](../../../references/reporting.md). + +```js +const E2E_DATA = { + title: "string - the spec's title", + planName: "string - the .dev/{plan-name} slug", + generatedAt: "string - ISO date of the run this report captures", + + // Decides how each scenario renders: a screenshot gallery, or a data-model-state table. + kind: "frontend" | "non-frontend", + + scenarios: [ + { + id: "string", + title: "string - one line, e.g. 'guest checks out with an expired coupon'", + given: "string", + when: "string", + then: "string", + status: "pass" | "fail", + + // kind === "frontend": one entry per meaningful step, screenshot embedded as a data URI. + screenshots: [ + { step: "string", caption: "string", dataUri: "data:image/png;base64,..." } + ], + + // kind === "non-frontend": one entry per meaningful step. Also a valid supplement for a + // frontend scenario when a step's real effect is a data change a screenshot can't show. + dataModelState: [ + { step: "string", caption: "string", entity: "string", before: "object or string", after: "object or string" } + ], + + logsOrOutput: "string - optional, captured stdout or response body", + durationMs: 0 + } + ], + + summary: { total: 0, passed: 0, failed: 0 } +}; +``` + +## Filling it in honestly + +- **Never fabricate a screenshot or a data-model-state entry.** Both must come from an actual run of the actual application - a screenshot invented to look plausible, or a before/after pair guessed instead of captured, defeats the entire point of this report being the enforceable proof behind an e2e criterion. +- **kind is chosen once per spec**, based on whether the system under test has a UI a screenshot could meaningfully show. A CLI, a backend API, a batch job: `"non-frontend"`. Anything a user clicks through: `"frontend"`. +- **dataModelState is a valid supplement even for a frontend scenario** when a step's real effect is invisible on screen (a queued job, a row written to a table the UI doesn't reflect yet). +- **status must reflect what actually happened.** A scenario that failed and was then fixed gets re-run and re-captured, not silently flipped to pass. +- **summary must match the scenarios array** - recompute it from the real counts, don't hand-write it separately. diff --git a/plugins/factory/skills/build/references/layers.md b/plugins/factory/skills/build/references/layers.md new file mode 100644 index 0000000..d7ecc5d --- /dev/null +++ b/plugins/factory/skills/build/references/layers.md @@ -0,0 +1,50 @@ +# Test Layers + +Every scenario on a change set's `tests:` line in `spec.md` is tagged `[unit]`, `[integration]`, or `[e2e]`. +A scenario is not met until a concrete test exists at its tagged layer - a prose input -> outcome with no test behind it is not done, however confident the implementation looks. +The tag is what keeps tests acting as the spec: it says where the proof has to live, so it can't quietly stay in prose. + +## The pyramid + +Cost, speed, and breadth of failure differ by orders of magnitude between the layers, so their populations should too: many unit tests, fewer integration tests, few e2e scenarios. + +| Layer | Population | Runtime | What it proves | +| ----- | ---------- | ------- | -------------- | +| Unit | most of the suite | milliseconds | one business rule, exactly | +| Integration | a middle band | sub-second to seconds | two owned components really agree | +| E2E | a handful per spec | seconds to minutes | a whole user journey actually works | + +Two rules follow, and they matter more than the ratios: + +- **Push every check down to the cheapest layer that can still fail for the right reason.** If a rule can be wrong in a pure function, test it in a pure function. Proving a rounding rule through a browser click is a slow test that also localizes badly: when it fails, it doesn't tell you where. +- **Push up only what lower layers structurally cannot see.** Wiring, configuration, serialization across a boundary, and the shape of a real user journey are invisible to a unit test no matter how many you write. That is what the upper layers are for, and why they exist at all. + +Two shapes to avoid: the **ice-cream cone**, where the suite is mostly e2e and every change costs a long red-green cycle, and the **hourglass**, where unit and e2e are both fat but nothing tests real collaboration, so integration bugs surface only in production. + +## Unit + +Pure business logic, no I/O. +Fast enough to run on every save. +Mock only at the true architectural boundaries in [mocking.md](mocking.md) - everything else inside the application stays real. +A unit scenario describes a computation or a business rule: "expired coupon is rejected," not "checkout endpoint returns 400." + +## Integration + +Validates real collaboration between two or more components this codebase owns, at the seam between them: a service against a real test database, a module against a real module it calls. +Never reaches the outside world. +Prefer real wiring or a fake over a mock at the boundary, per [mocking.md](mocking.md). +An integration scenario describes a cross-component contract: "the order API persists the order and a repository read returns it back." + +## E2E + +Drives the actual running application - the built artifact, not a test-harness shortcut - through a realistic end-user scenario, against the **mocked environment** of [mocking.md](mocking.md#the-e2e-environment). +It never runs against production or a shared staging environment; the point is a repeatable journey, not a live probe. +This is the only layer allowed to drive a browser. +It produces the HTML scenario report in [e2e-report.md](e2e-report.md): screenshots for frontend systems, before/after data-model state for everything else. +An e2e scenario describes user-observable, end-to-end behavior: "a guest can complete checkout with an expired coupon and sees the correct error." +Keep the count small and reserve it for critical journeys - each scenario is also a piece of evidence someone reads in the report. + +## Where tests.md and mocking.md apply + +[tests.md](tests.md)'s seams, behavior-over-implementation, tautology, and naming rules govern how you write a test at any layer. +[mocking.md](mocking.md) governs what may be replaced: architectural boundaries only at unit and integration, and the environment-only mocking that e2e is allowed. diff --git a/plugins/factory/skills/build/references/mocking.md b/plugins/factory/skills/build/references/mocking.md new file mode 100644 index 0000000..26af070 --- /dev/null +++ b/plugins/factory/skills/build/references/mocking.md @@ -0,0 +1,52 @@ +# Mocking Guidelines + +Mocks exist to make tests deterministic and fast, not to isolate every class from every other class. +Every mock is a place where the test stops verifying reality, so use as few as possible. + +## Mock only at architectural boundaries + +Replace a dependency only when it crosses a boundary you don't control in the test: + +- Network calls to third-party services (payment providers, external APIs) +- The system clock and randomness +- Anything genuinely slow or flaky in a test environment (email delivery, message queues) + +Everything inside the application - services, repositories wired to a test database, domain objects - should be real. +If the internal wiring is wrong, a test full of internal mocks will still pass, which makes it worse than no test. + +## Never mock internal collaborators + +If you feel the need to mock a class your own codebase owns, that is a design signal, not a testing problem. +Either test at a higher seam where the collaborator can just run, or the collaborator itself is a seam worth agreeing on with the user. + +The tell that a mock is wrong: the test asserts *that* a method was called (`toHaveBeenCalledWith`) instead of *what the outcome was*. +Interaction assertions couple the test to the implementation; state and output assertions couple it to behavior. + +## Prefer fakes over mocks at the boundary + +When you do replace a boundary dependency, prefer a small working fake over a per-test stub: + +- An in-memory repository instead of stubbing each query +- A fake clock you can advance instead of freezing time per test +- A recording fake for outbound calls (a fake mail sender with a `sent` list) instead of call-count assertions + +Fakes keep behavior consistent across tests and survive refactors; ad-hoc stubs encode one test's assumptions. + +## The e2e environment + +E2E has no mocks *inside* the application and no real world *outside* it. +The application under test is the real built artifact with its real internal wiring; everything it depends on beyond its own boundary is mocked, seeded, and deterministic: + +- Third-party HTTP replaced by a local stub server or recorded fixtures, with no credentials that could reach a real provider. +- A dedicated database or store, seeded to a known state before the run and torn down after. +- A fixed clock and seeded randomness, so the same scenario produces the same report twice. +- Outbound side effects (mail, payments, webhooks, queues) captured by a recording fake rather than delivered. + +Never point an e2e run at production or a shared staging environment. +A live run makes the scenario unrepeatable, its before/after state unreliable as evidence, and its side effects real. +If the mocked environment for a dependency doesn't exist yet, building it is part of the work, not a reason to fall back to the real one. + +## Determinism + +Inject the clock and randomness at the seam rather than patching globals. +A test that patches `Date.now` globally is fragile and can leak into other tests; a component that accepts a clock is testable by construction. diff --git a/plugins/factory/skills/build/references/parallel.md b/plugins/factory/skills/build/references/parallel.md new file mode 100644 index 0000000..d3b6a05 --- /dev/null +++ b/plugins/factory/skills/build/references/parallel.md @@ -0,0 +1,63 @@ +# Parallel Execution + +`scope` ordered the change plan so each change set builds only on the ones before it, and listed each change set's files. +This file is how to cash that in: you are the orchestrator, subagents are the implementers. + +## Waves + +Group the change sets into waves by consecutive-disjoint batching: + +- Walk change sets in numeric order; the spec author's ordering is the dependency order. +- Grow the current wave with the next change set only when its file list is disjoint from every change set already in the wave AND it consumes nothing a change set in the wave introduces (a module, function, endpoint, or decision outcome - your judgment while reading the spec). +- Any overlap or doubt excludes that change set from the wave; it anchors the next wave. +- A change set past an excluded one may still join the current wave, under a doubled condition: disjoint from, and consuming nothing introduced by, every change set in the wave AND every earlier change set not yet committed. Doubt excludes it - the skip-ahead is the same dependency proxy applied against everything still unfinished, so it cannot produce an order the sequential walk would forbid. + +Sequential-by-default means parallelism is a pure optimization that can never produce a wrong order. +Never start work belonging to the next wave while the current wave is in flight. + +## Launching a wave + +Launch one `Agent` per change set, **all in a single message** so they run concurrently. +A wave of one change set needs no subagent: implement it yourself in the main thread. + +Each agent prompt contains: + +1. The role: `You implement exactly one change set of a spec. Other agents implement sibling change sets concurrently; stay inside your change set's file list.` +2. The absolute path to `spec.md`, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The spec is a self-contained handoff by design; the agent reads all of it, then implements only its own change set. +3. The absolute path to this skill's `SKILL.md`, with the instruction to follow its "The change-set loop" and "Rules of the loop" sections - read like the other references, not pasted into the prompt. +4. Hard constraints: + - Implement only this change set's `[unit]` and `[integration]` scenarios. `[e2e]` scenarios are run once per spec by the orchestrator afterwards - do not launch the app. + - Do not commit, stage, or touch git state. The orchestrator commits. + - Do not edit files outside your change set's file list. If the change set genuinely needs a file another change set owns, stop and report it as a conflict instead of editing it. + - Do not edit `implementation-notes.md`. Report your entry; the orchestrator appends it. + - Run the spec's Validation block and report its real result. A red result is a fact to report, not something to hide or paper over. +5. The output contract below. + +Output contract (the agent's final message must be exactly one fenced JSON block): + +```json +{ + "changeSet": 3, + "status": "done | blocked", + "whatWasDone": "2-4 lines", + "seamsTested": ["..."], + "testsAdded": ["path::test name"], + "deviations": ["edge case found -> conservative choice made: what/why"], + "selfValidation": { "commands": ["..."], "result": "pass | fail", "output": "the tail that matters" }, + "conflicts": ["file another change set owns that this change set needed"] +} +``` + +## After a wave + +1. Verify rather than trust the reports: run the spec's Validation block yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. +2. Commit each change set's work on the current branch, in number order, one commit per change set. A skipped-ahead change set's commit waits until every earlier change set is committed, so history keeps the spec's order. +3. Append each change set's entry to `implementation-notes.md` from `whatWasDone`, `seamsTested`, `testsAdded`, and `deviations`. `testsAdded` becomes the entry's `Tests added:` line, which the scenario checker reads. + +A `status: blocked` change set, a failing validation, or a reported conflict is yours to finish in the main thread before the next wave starts - do not carry a red change set forward and do not relaunch the same agent on the same failure more than once. +If two change sets in a wave edited the same file anyway, reconcile it yourself and log it under Deviations. + +## When not to parallelize + +- The spec has one change set, or every change set's files overlap with the one before it: run them sequentially yourself. +- Two change sets batched into a wave visibly collide anyway: treat that as a `scope` sizing miss, run them sequentially, and note it in `implementation-notes.md`. diff --git a/plugins/factory/skills/build/references/tests.md b/plugins/factory/skills/build/references/tests.md new file mode 100644 index 0000000..78c7939 --- /dev/null +++ b/plugins/factory/skills/build/references/tests.md @@ -0,0 +1,91 @@ +# Test Examples + +Worked examples of the principles in SKILL.md: behavior over implementation, tests as specification, independent expected values. +See [layers.md](layers.md) for how a scenario's tag decides which layer its test lives at. + +## Seams + +A **seam** is the public boundary a test exercises, never internals. +Naming the seam before writing the test is what keeps testing effort on real interfaces instead of spreading over every private helper: "what is the public interface here, and which boundary would a caller actually cross?" + +The spec's scope section declares the public inputs and outputs of the change, so the seam for a scenario - the boundary its input crosses - is normally a decision you inherit, not one to reopen. +Reopen it only when the declared boundary is wrong or missing: pick the nearest real public boundary, use it, and log the change under Deviations in `implementation-notes.md`. +The signal that a seam is wrong is usually a mocking urge - see [mocking.md](mocking.md)'s "never mock internal collaborators." + +## Behavior test vs implementation-coupled test + +The seam here is the public `Cart` interface. + +Good - exercises the seam, asserts observable behavior: + +```ts +test("user can checkout with a valid cart", async () => { + const cart = new Cart(); + cart.add(product("book", 1200), 2); + + const receipt = await cart.checkout(validPayment()); + + expect(receipt.totalCents).toBe(2400); + expect(receipt.status).toBe("paid"); +}); +``` + +Bad - reaches inside and verifies internal wiring: + +```ts +test("checkout calls the tax calculator", async () => { + const taxCalc = jest.spyOn(cart["taxCalculator"], "compute"); + await cart.checkout(validPayment()); + expect(taxCalc).toHaveBeenCalledWith(2400); +}); +``` + +The bad test breaks if tax computation moves, is renamed, or gets inlined, even though checkout behavior is identical. +The good test survives all of those refactors. + +## Side-channel verification + +Bad - asserts through the database instead of the interface: + +```ts +await api.createUser({ name: "Ada" }); +const row = await db.query("SELECT * FROM users WHERE name = 'Ada'"); +expect(row.status).toBe("active"); +``` + +Good - observes the result the way a caller would: + +```ts +await api.createUser({ name: "Ada" }); +const user = await api.getUser("Ada"); +expect(user.status).toBe("active"); +``` + +If the schema changes but the API contract holds, only the bad test breaks. + +## Independent expected values + +Bad - tautological, recomputes the expectation the way the code does: + +```ts +expect(priceWithVat(100)).toBe(100 * 1.21); +``` + +Good - the expected value comes from a worked example or the spec: + +```ts +// Spec section 4.2: 100.00 EUR at 21% VAT is 121.00 EUR +expect(priceWithVat(100)).toBe(121); +``` + +If someone changes the VAT logic incorrectly, the tautological test still passes; the literal catches it. + +## Test names as specification + +Name tests after the capability, not the method under test. + +- Good: `"expired coupon is rejected at checkout"` +- Bad: `"test applyCoupon returns false"` + +A reader should be able to reconstruct what the system does from the test names alone. +Use the vocabulary the code and existing tests already use, so names match how the team talks about the feature. diff --git a/plugins/factory/skills/build/scripts/check-tests.py b/plugins/factory/skills/build/scripts/check-tests.py new file mode 100755 index 0000000..7d90d25 --- /dev/null +++ b/plugins/factory/skills/build/scripts/check-tests.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""Check that a spec's tests: scenarios really landed as tests. + +Usage: python3 check-tests.py .dev/{plan-name} [--repo-root .] +Exit 0 when every change set is accounted for, 1 with one problem per line +otherwise. Every test named in implementation-notes.md must exist on disk with +that name in it, so a claimed test that was never written cannot pass as done. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +NOTE_TESTS = re.compile(r"^\s*-\s*Tests added:\s*(.*)$", re.IGNORECASE) + + +def spec_scenarios(spec: Path, problem) -> dict[int, int]: + """Scenario count per change set, from the spec's change plan.""" + counts: dict[int, int] = {} + in_plan = False + current = None + for number, line in enumerate(spec.read_text(encoding="utf-8").splitlines(), start=1): + heading = HEADING.match(line) + if heading: + in_plan = heading.group(1).lower() == "change plan" + continue + if not in_plan: + continue + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + counts.setdefault(current, 0) + continue + tests = TESTS.match(line) + if not tests or current is None: + continue + value = tests.group(1).strip() + if value.lower().startswith("none"): + continue + counts[current] = len([s for s in value.split(";") if s.strip()]) + if not counts: + problem(f"{spec}: no change sets found under '## Change plan'") + return counts + + +def note_entries(notes: Path) -> dict[int, list[str]]: + """Tests named per change set, from implementation-notes.md.""" + entries: dict[int, list[str]] = {} + current = None + for line in notes.read_text(encoding="utf-8").splitlines(): + entry = NOTE_ENTRY.match(line) + if entry: + current = int(entry.group(1)) + entries.setdefault(current, []) + continue + named = NOTE_TESTS.match(line) + if not named or current is None: + continue + value = named.group(1).strip() + if value.lower().startswith("none"): + continue + entries[current].extend(t.strip() for t in value.split(",") if t.strip()) + return entries + + +def check_named_test(reference: str, repo_root: Path, change_set: int, problem) -> None: + """A named test must be path::name, and that name must be in that file.""" + if "::" not in reference: + problem(f"change set {change_set}: '{reference}' is not in path::test name form") + return + path, _, name = reference.partition("::") + target = repo_root / path.strip() + if not target.is_file(): + problem(f"change set {change_set}: {path.strip()} does not exist") + return + if name.strip() not in target.read_text(encoding="utf-8", errors="replace"): + problem( + f"change set {change_set}: {path.strip()} contains no test " + f"named '{name.strip()}'" + ) + + +def main(argv: list[str]) -> int: + args: list[str] = [] + repo_root = Path(".") + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--repo-root" and rest: + repo_root = Path(rest.pop(0)) + else: + args.append(value) + if len(args) != 1: + print("usage: check-tests.py [--repo-root .]", file=sys.stderr) + return 2 + + plan = Path(args[0]) + spec, notes = plan / "spec.md", plan / "implementation-notes.md" + for path in (spec, notes): + if not path.is_file(): + print(f"no {path.name} at {path}", file=sys.stderr) + return 2 + + problems: list[str] = [] + + def problem(message: str) -> None: + problems.append(message) + + counts = spec_scenarios(spec, problem) + entries = note_entries(notes) + + for change_set, scenarios in sorted(counts.items()): + if change_set not in entries: + problem(f"change set {change_set} has no entry in {notes}") + continue + named = entries[change_set] + if scenarios and len(named) < scenarios: + problem( + f"change set {change_set}: {scenarios} scenario(s) specced, " + f"{len(named)} test(s) named" + ) + for reference in named: + check_named_test(reference, repo_root, change_set, problem) + + for change_set in sorted(set(entries) - set(counts)): + problem(f"change set {change_set} is in {notes} but not in the spec's change plan") + + for message in problems: + print(message) + if problems: + print(f"\n{len(problems)} problem(s); the spec is not fully implemented.") + return 1 + print(f"{plan}: clean - {len(counts)} change sets, every named test found.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/plugins/factory/skills/build/templates/e2e-report.html b/plugins/factory/skills/build/templates/e2e-report.html new file mode 100644 index 0000000..5550d44 --- /dev/null +++ b/plugins/factory/skills/build/templates/e2e-report.html @@ -0,0 +1,248 @@ + + + + + +E2E Report + + + + +
+
+

E2E Report

+
+
+ +
+ +
+
+
+
+ + + + diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md new file mode 100644 index 0000000..05dd6ab --- /dev/null +++ b/plugins/factory/skills/run/SKILL.md @@ -0,0 +1,50 @@ +--- +name: run +description: Take a request through an interactive scope, then run an unattended scope-review, build, and ship as phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use to run the factory end to end on Claude Code or Codex. +--- + +# Run + +You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase invoke another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. + +## 1. Preflight + +Confirm the host has a subagent tool per the transport ladder in [references/launch.md](references/launch.md); with none, stop before the interview so nobody pays for a scope that cannot run. +Check that `.dev/` is git-ignored in the consuming repository (`git check-ignore -q .dev/`); when it is not, append the entry to `.gitignore` yourself, and once the run branch exists at handoff, record the write as a repair with the check output as evidence (D-plan-files-ignored). The two hard stops in section 5 bind this write too. +Resolve your own base directory from the host's injected "Base directory for this skill" line. + + +If two unfinished state files exist under `.dev/` with no branch match, or a recorded run branch no longer exists, list the candidates (plan, branch, last phase) or name the missing branch and ask which to resume - a person is at the keyboard invoking `/factory:run`, so this is not a post-go touchpoint. + + +## 2. Init or resume + +No request and a run branch checked out: resume that plan. +No request and no branch: resume the single state file under `.dev/` not marked done. +A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it, create `.dev/{plan}/` with `request.md` holding the request, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch}` (D-plan-slug, D-request-input). + +## 3. Scope, inline + + +Launch the copied `scope` skill inline, in this session: read its `SKILL.md` at the absolute path `{run-skill-root}/../scope/SKILL.md`, telling it to read `{scope-skill-root}` as that path's parent directory. It keeps its interview and ends with one explicit go question you ask the user. + + +## 4. Handoff + +At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan}` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). + +## 5. Per phase: launch, judge, act + +For each phase in order (`scope-review`, `build`, `ship`): + +1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. +2. Wait for the host's completion signal - a batch in flight is not over. +3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). +4. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). +5. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. +6. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. +7. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. + +## 6. Closing report + +Print: phases, attempts, decisions, repairs, and the PR link or the exact action a person must take. Read only result files, check outputs, `git log`, and the state file for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). diff --git a/plugins/factory/skills/run/agents/openai.yaml b/plugins/factory/skills/run/agents/openai.yaml new file mode 100644 index 0000000..317e24b --- /dev/null +++ b/plugins/factory/skills/run/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "Run" + short_description: "Take a request from scope to a shipped change unattended" + default_prompt: "Use $factory:run to take a request through scope, then run scope-review, build, and ship unattended." +policy: + allow_implicit_invocation: false diff --git a/plugins/factory/skills/run/references/judgment.md b/plugins/factory/skills/run/references/judgment.md new file mode 100644 index 0000000..a800b8a --- /dev/null +++ b/plugins/factory/skills/run/references/judgment.md @@ -0,0 +1,40 @@ +# Judgment + +Per phase, the evidence commands the orchestrator re-runs as proof and what a `done` looks like. `run-state.py check-result` gives the result file's own claim; these checks are the independent evidence read before trusting it. + +## scope + +`done`: `lint-spec.py .dev/{plan}/spec.md` reports clean, and the go was recorded (the interview happened inline in this session, so the orchestrator already knows). + +## scope-review + +`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a `Verdict: APPROVED` line with a `Rounds:` line. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. + +## build + +`done`: `check-tests.py .dev/{plan}` clean, the spec's Validation block green, commits since handoff match the change plan's numbering, and (when the spec has `[e2e]` scenarios) the e2e report's data block shows every scenario passed. + +## ship + +`done` has two endings, decided by what `origin` resolves to: + +- **A GitHub remote**: `pr-evidence.py check` clean, `review_N.md` carries a verdict for HEAD, and `gh pr view` shows the PR open with head equal to HEAD. +- **A remote no GitHub host backs** (the fixture's bare origin): `gh pr create` cannot succeed, so ship cannot reach `done` at all. The evidence is `git ls-remote origin` showing the branch at HEAD and `pr.md` written to disk. The orchestrator ends the run with a report naming the pull request as the one unfinished step - this is the fixture's expected ending, not a factory fault, and is judged under the third `gh` fault category below, never as a park. + +## The failure ladder + +Cheapest sufficient step, recorded before it is acted on: + +1. **Repair.** Fix it yourself and re-run the phase's checks - at most two repairs per attempt; the third fix is a relaunch and counts as a new attempt. A repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches. +2. **Relaunch.** A fresh subagent, guidance naming what the previous attempt did and what is required instead - never "try again". +3. **End.** Three attempts of one phase without a `done` the orchestrator accepts: end the run with a report naming the last reason, the artifacts, and what a person could do. + +Every repair is committed on the run branch and bound by the two hard stops: never rewrite history, force-push, or delete outside the run branch. + +## Fault categories during ship + +Three, told apart by the error text, never guessed: + +1. **`gh` unauthenticated or GitHub unreachable** (token, credential helper, or sandbox network): an environment fault, the run ends naming the missing prerequisite. +2. **No GitHub remote behind `origin`**: `gh pr create` refuses with "none of the git remotes configured for this repository point to a known GitHub host" before any network call, even with `gh` installed and authenticated. Match this text first - its trailing `gh auth login` hint is never read as a credentials fault. The branch stays pushed, `pr.md` stays written, the run ends with a report naming the origin and the pull request as the one unfinished step. +3. **A secret found after build committed it**: the run ends without a pull request, the branch is left unpushed, the report names the commit to purge. diff --git a/plugins/factory/skills/run/references/launch.md b/plugins/factory/skills/run/references/launch.md new file mode 100644 index 0000000..6dbb186 --- /dev/null +++ b/plugins/factory/skills/run/references/launch.md @@ -0,0 +1,32 @@ +# Launch + +The host transport ladder for one phase agent, and the exact prompt template. Mirrors ship's `orchestration.md` native-transport rung, one agent at a time instead of a batch. + +## Transport ladder + +1. **Claude Code**: the Agent tool. +2. **Codex**: `spawn_agent`. +3. **opencode**: the `task` tool. +4. **Unavailable**: stop before the interview - nobody pays for a scope that cannot run. + +## Prompt template + +``` +You implement exactly one factory phase. Read {phase}-skill-root/SKILL.md at +the absolute path {skill_path}; treat {skill_path}'s parent directory as +{phase}-skill-root when the file uses that placeholder. Follow it exactly. + +Plan: {plan}, at .dev/{plan}/ (absolute: {plan_dir}). +Attempt: {attempt}. +{On a relaunch only: The previous attempt failed with reason "{reason}". +Guidance: {what to do differently, never "try again"}.} + +Scratch root for this attempt: /tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/. +Write your result to .dev/{plan}/results/{phase}-{attempt}.json as your last +action, per the schema in factory-run.md, echoing skill_path back exactly as +given above. + +This phase never launches another phase, and after the go it never asks a +person anything - decide under the unattended policy in factory-run.md and +record the decision in auto_decided. +``` diff --git a/plugins/factory/skills/run/scripts/run-state.py b/plugins/factory/skills/run/scripts/run-state.py new file mode 100644 index 0000000..fc9502a --- /dev/null +++ b/plugins/factory/skills/run/scripts/run-state.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +"""Own factory-run.json: init, handoff, diff-spec, record, check-result, show. + +Usage: + run-state.py init --request --base + run-state.py handoff --branch [--spec ] [--dirty ...] + run-state.py diff-spec [--spec ] + run-state.py record # reads one JSON decision object on stdin + run-state.py check-result + run-state.py show + +The orchestrator loops on the exit code of every subcommand, so a malformed +state file or a bad call never reaches a judgment. Every function here stays +at or under cyclomatic complexity 10 (D-complexity-threshold). +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import subprocess +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) +NOT_DOING = re.compile(r"⊘") + +ACTIONS = {"advance", "repair", "relaunch", "end"} +STATE_NAME = "factory-run.json" + + +def state_path(plan: str) -> Path: + return Path(".dev") / plan / STATE_NAME + + +def spec_path(plan: str, override: str | None) -> Path: + return Path(override) if override else Path(".dev") / plan / "spec.md" + + +def load_state(plan: str) -> dict: + path = state_path(plan) + return json.loads(path.read_text(encoding="utf-8")) + + +def save_state(plan: str, state: dict) -> None: + state_path(plan).write_text(json.dumps(state, indent=2) + "\n", encoding="utf-8") + + +def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list[str]]: + """Scenario texts per change set, and every ⊘ line, from the current spec.""" + texts: dict[str, list[str]] = {} + not_doing: list[str] = [] + in_plan = False + current: str | None = None + for line in spec.read_text(encoding="utf-8").splitlines(): + heading = HEADING.match(line) + if heading: + in_plan = heading.group(1).lower() == "change plan" + if NOT_DOING.search(line): + not_doing.append(line.strip()) + if not in_plan: + continue + change_set = CHANGE_SET.match(line) + if change_set: + current = change_set.group(1) + texts.setdefault(current, []) + continue + tests = TESTS.match(line) + if not tests or current is None: + continue + value = tests.group(1).strip() + if value.lower().startswith("none"): + continue + texts[current] = [s.strip() for s in value.split(";") if s.strip()] + return texts, not_doing + + +def git_dirty_files() -> list[str]: + try: + out = subprocess.run( + ["git", "status", "--porcelain"], capture_output=True, text=True, check=True + ).stdout + except (OSError, subprocess.CalledProcessError): + return [] + return [line[3:].strip() for line in out.splitlines() if line.strip()] + + +def cmd_init(args: argparse.Namespace) -> int: + path = state_path(args.plan) + if path.exists(): + print(f"factory-run.json already exists at {path}") + return 1 + path.parent.mkdir(parents=True, exist_ok=True) + state = { + "plan": args.plan, + "request": args.request, + "base": args.base, + "branch": None, + "spec_sha256": None, + "scenario_texts": {}, + "not_doing_lines": [], + "dirty_files": [], + "phases": {"scope": [], "scope-review": [], "build": [], "ship": []}, + "decisions": [], + "repairs": [], + } + save_state(args.plan, state) + print(f"initialized {path}") + return 0 + + +def cmd_handoff(args: argparse.Namespace) -> int: + spec = spec_path(args.plan, args.spec) + if not spec.is_file(): + print(f"no spec.md at {spec}") + return 1 + state = load_state(args.plan) + digest = hashlib.sha256(spec.read_bytes()).hexdigest() + texts, not_doing = scenario_texts_and_not_doing(spec) + state["branch"] = args.branch + state["spec_sha256"] = digest + state["scenario_texts"] = texts + state["not_doing_lines"] = not_doing + state["dirty_files"] = args.dirty if args.dirty else git_dirty_files() + save_state(args.plan, state) + print(f"handoff recorded: {sum(len(v) for v in texts.values())} scenario(s), {len(not_doing)} not-doing line(s)") + return 0 + + +def cmd_diff_spec(args: argparse.Namespace) -> int: + state = load_state(args.plan) + spec = spec_path(args.plan, args.spec) + texts, not_doing = scenario_texts_and_not_doing(spec) + problems: list[str] = [] + stored_texts: dict[str, list[str]] = state.get("scenario_texts", {}) + for change_set, before in stored_texts.items(): + after = texts.get(change_set, []) + dropped = [t for t in before if t not in after] + added = [t for t in after if t not in before] + if dropped: + problems.append( + f"change set {change_set}: scenario dropped or reworded: before {dropped!r}, now {added!r}" + ) + stored_not_doing = state.get("not_doing_lines", []) + for line in stored_not_doing: + if line not in not_doing: + problems.append(f"⊘ line dropped: {line!r}") + for message in problems: + print(message) + return 1 if problems else 0 + + +def cmd_record(args: argparse.Namespace) -> int: + raw = sys.stdin.read() + try: + decision = json.loads(raw) + except json.JSONDecodeError as exc: + print(f"invalid JSON on stdin: {exc}") + return 1 + action = decision.get("action") + if action not in ACTIONS: + print(f"action {action!r} is not one of {sorted(ACTIONS)}") + return 1 + state = load_state(args.plan) + entry = { + "phase": decision.get("phase"), + "attempt": decision.get("attempt"), + "action": action, + "rationale": decision.get("rationale", ""), + "evidence": decision.get("evidence", []), + } + state.setdefault("decisions", []).append(entry) + if action == "repair": + repair = decision.get("repair", {}) + state.setdefault("repairs", []).append( + { + "phase": entry["phase"], + "attempt": entry["attempt"], + "description": repair.get("description", ""), + "files": repair.get("files", []), + "evidence": entry["evidence"], + } + ) + save_state(args.plan, state) + print(f"recorded {action} for {entry['phase']} attempt {entry['attempt']}") + return 0 + + +def cmd_check_result(args: argparse.Namespace) -> int: + path = Path(args.path) + if not path.is_file(): + print(f"missing result file: {path}") + return 3 + try: + data = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError as exc: + print(f"unparseable result file {path}: {exc}") + return 3 + status = data.get("status") + if status == "done": + print("done") + return 0 + if status == "failed": + print(f"failed: {data.get('reason', '')}") + return 1 + if status == "stopped": + kind = data.get("stop", {}).get("kind", "") + print(f"stopped: {kind}") + return 2 + print(f"failed: unrecognized status {status!r}") + return 1 + + +def cmd_show(args: argparse.Namespace) -> int: + state = load_state(args.plan) + for phase, attempts in state.get("phases", {}).items(): + if not attempts: + print(f"{phase}: no attempts") + continue + for index, attempt in enumerate(attempts, start=1): + print(f"{phase} attempt {index}: {attempt.get('status', 'unknown')}") + for repair in state.get("repairs", []): + print(f"repair on {repair['phase']} attempt {repair['attempt']}: {repair['description']}") + return 0 + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + + init_p = sub.add_parser("init") + init_p.add_argument("plan") + init_p.add_argument("--request", required=True) + init_p.add_argument("--base", required=True) + + handoff_p = sub.add_parser("handoff") + handoff_p.add_argument("plan") + handoff_p.add_argument("--branch", required=True) + handoff_p.add_argument("--spec") + handoff_p.add_argument("--dirty", action="append") + + diff_p = sub.add_parser("diff-spec") + diff_p.add_argument("plan") + diff_p.add_argument("--spec") + + record_p = sub.add_parser("record") + record_p.add_argument("plan") + + check_p = sub.add_parser("check-result") + check_p.add_argument("path") + + show_p = sub.add_parser("show") + show_p.add_argument("plan") + + return parser + + +def main(argv: list[str]) -> int: + args = build_parser().parse_args(argv[1:]) + handlers = { + "init": cmd_init, + "handoff": cmd_handoff, + "diff-spec": cmd_diff_spec, + "record": cmd_record, + "check-result": cmd_check_result, + "show": cmd_show, + } + return handlers[args.command](args) + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/plugins/factory/skills/scope-review/SKILL.md b/plugins/factory/skills/scope-review/SKILL.md new file mode 100644 index 0000000..3a16fa0 --- /dev/null +++ b/plugins/factory/skills/scope-review/SKILL.md @@ -0,0 +1,131 @@ +--- +name: scope-review +description: Review and auto-refine a settled spec before build starts - a fresh-context agent panel checks the plan against the actual repo for infeasible change sets, missing failure paths, semantic contradictions, and untestable scenarios, then verified findings are applied to spec.md (and, where they touch settled decisions or cross-boundary invariants, to docs/decisions.md and docs/contracts.md) by refine agents and the panel re-runs; findings only the user can decide are asked as questions at the end and their answers applied and promoted the same way, so a finished run hands build a spec - and ledger - ready to implement, with no separate scope pass needed to promote the changes. Use after scope settles a spec and before build implements it. +--- + +# Scope Review + +Review the spec with agents that did not write it, then refine it in place - before any implementation exists, and without a human in the loop. +A defect caught here costs a spec edit; the same defect after build costs a re-implementation, so this loop runs to completion on its own and ends with a spec build can start on - not a findings list to triage, and not a handoff back to `scope`. +The few findings only the user can decide are asked as questions at the end of the run, and the answers are applied before it finishes. +The panel judges the plan against the actual repo, not against the conversation that produced it. +You are the orchestrator: run tools, dispatch agents, apply the loop, report - your own reading of the spec is not a lens, and findings reach the spec only through verification. +This skill edits `spec.md`, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. + +`scope`'s own phase-5 reviewer hunts while the spec is still being drafted, from the spec file alone. +This skill is the standalone deeper pass: fresh agents with repo access, adversarial verification, and automatic refinement - worth running when the change is large or risky, or when build will run in a different session. + +Invoking this skill is the task - locate the spec yourself and start immediately; do not ask what to review. +First run `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py start scope-review` so the wrap-up can measure this run. + +## 1. Locate and gate + +- Locate the plan directory per [../../references/plan-layout.md](../../references/plan-layout.md) and read its `spec.md`; no spec: say so and stop - there is nothing to review. +- Run `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` once. If it reports anything, stop and recommend finishing the `scope` run: refinement here presumes a mechanically settled spec, and repairing an unfinished draft is `scope`'s job, not this loop's. +- Read the target repo's `docs/decisions.md`, `docs/contracts.md`, and `docs/dependencies.md` where they exist; their entries are premises the lenses cite. +- `spec-review_N.md` files present -> unresolved escalations from the highest-numbered one become verification items for round 1, and new reports continue the numbering. + +## 2. The review-refine loop + +Run up to two rounds; each round is panel -> verify -> refine. +A third panel means the refinements are churning, not converging - stop and escalate what remains. + +1. **Panel.** Run batch 1 (all four lenses) and batch 2 (verifiers) per the mechanics below. + From round 2 on, brief the lenses on what round 1 refined: their job is regressions in the refined material and their own unresolved findings, not a fresh full-spectrum hunt - a round that raises no BLOCK is convergence, and its CONCERNs go verified into the report rather than through another refine cycle. +2. **Aggregate.** Apply verdicts with the script below; REFUTED findings drop, a BLOCK survives only when CONFIRMED. +3. **Split.** Sort surviving findings into refinable and escalations per the authority rules below. +4. **Refine.** No refinable findings: exit the loop. Otherwise dispatch fresh-context refine agents - one per independent group of findings, launched in a single message - each given only its findings, the spec path, and the refinement rules below. They edit `spec.md` only. +5. **Re-gate.** Loop `lint-spec.py` until clean, fixing mechanical fallout with the same refine agents; then start the next round so fresh eyes judge the refined spec. + +Exit the loop when a panel raises nothing refinable; two rounds of the same finding surviving refinement is itself an escalation. +Escalations collected across the rounds go to the interview below, not to a handoff. + +## 3. Authority and refinement rules + +Refine agents resolve conflicts by this order - each level beats everything below it: + +1. The user's recorded intent: the spec's stated problem, scope, and interview outcomes. +2. The repo's reality: what the code actually contains beats what the spec claims about it. +3. A settled `✓` decision: a change set contradicting its linked decision is rewritten to match the decision, not the other way around. +4. The spec's prose. + +Refinable: false premises about the repo (rewrite the entry against the real code, including the extra work that reveals), change sets contradicting their linked decisions, missing test scenarios for stated invariants and failure paths, untestable scenarios (replace with one provable at that layer), and gaps whose resolution is forced once the repo is consulted. +Escalations - never auto-applied, queued for the interview instead: anything that would flip a `✓` decision to a rejected alternative, change the user-visible scope or behavior, add or drop a dependency, or contradict the user's recorded intent. +A refine agent that cannot fix its finding without crossing that line marks it escalated and leaves the spec alone. +Refinements follow `scope`'s notation: decision entries keep their slugs and marks, change sets keep their numbering, new scenarios carry layer tags. +A refinement that adds or rewrites a decision entry, or a cross-boundary invariant, is promoted to the ledger immediately per step 5 - it does not wait for the interview. + +## 4. Resolve escalations under factory policy + +After the loop, answer each escalation yourself under the unattended policy in [../../references/factory-run.md](../../references/factory-run.md): pick the recommended option when one is defensible, else the safest alternative, and record it in `auto_decided`. +Apply each answer immediately with a refine pass: update the decision entry's marks and because clauses, rewrite the affected change sets and `tests:` lines, keep `scope`'s notation, and loop `lint-spec.py` until clean. +An answer that resolves cleanly in place ends that escalation; verify the applied refinement yourself against the repo rather than re-running a panel for it. +When the answer settles or flips a decision, or changes a cross-boundary invariant, promote it to the ledger per step 5 as soon as the refine pass lands - do not wait for the run to finish. + +One outcome cannot resolve under policy: an escalation that invalidates the change's premise or opens a genuinely new effort - a new sub-effort with its own decision tree - exceeds what factory policy may decide. +Report the phase `failed` with reason `rescope`, naming what a future `scope` run must revisit; never guess at a new effort's shape. + +Verdict: `Verdict: APPROVED` always - a run that cannot settle every escalation under policy is reported `failed` with `rescope` instead of a partial verdict. + +## 5. Promote to the ledger + +Every settled decision entry and cross-boundary invariant that this run added or changed in `spec.md` - from refinement or from an answered escalation - gets promoted the same run, so build never has to wait on a separate `scope` pass for it. +Apply the promotion test in [../../references/decision-ledger.md](../../references/decision-ledger.md): copy qualifying decisions into `docs/decisions.md` verbatim, dated, sourced to this spec, evidence marks included, and promote recurring rationales to `P-` principles under the same bar. +Promote cross-boundary invariants to `docs/contracts.md` under the same test, phrased for the relying side - same as `scope` phase 8. +A decision or invariant that fails the promotion test, or a `? verify:` mark still open, stays in `spec.md` only and is not forced into the ledger. + +## 6. Panel mechanics + +Lenses live in [references/lenses.md](references/lenses.md): `feasibility`, `completeness`, `consistency`, `testability` - all four, every round. +Run the batches on the transport selected by [../ship/references/orchestration.md](../ship/references/orchestration.md), which owns transport choice, result-file delivery, and the batch mechanics, with these substitutions: + +- The artifact under review is `spec.md`, not a diff: hand each agent the spec path and the repo root instead of a diff file, and drop the diff-location line from the prompt contract. +- Lens definitions and shared rules come from this skill's [references/lenses.md](references/lenses.md). +- Verifiers refute findings against the spec file and the actual repo; `file`/`line` in a finding points into `spec.md` unless a repo path is named. +- Scratch root: `/tmp/scope-review-{session-id}/round-{R}/`. + +If no transport is available, stop and explain that this panel-based skill cannot preserve its verification contract. +Between the batches, number the findings the verifiers get, and after batch 2 aggregate: + +```bash +python3 {ship-skill-root}/scripts/aggregate-findings.py plan {batch-1 results} +python3 {ship-skill-root}/scripts/aggregate-findings.py aggregate {batch-1 results} {batch-2 results} --expected {lenses} +``` + +## 7. Write the report + +Write `.dev/{plan-name}/spec-review_N.md` at the next free index, one per run, covering all rounds: + +```markdown +# Spec review N - {plan-name} - {date} + +Verdict: APPROVED | APPROVED WITH DEFERRALS +Rounds: {R} - {finding counts per round} + +## Refinements applied +### R1 - {lens} - {one-line title} +{spec.md:line} - {what was wrong} -> {what the spec says now} + +## Escalations resolved +### E1 - {lens} - {one-line title} +Asked: {the question} - Answered: {the user's decision} -> {what the spec says now} + +## Promoted to the ledger +{decision slug or contract entry} -> `docs/decisions.md` | `docs/contracts.md` + +## Deferred +### D1 - {lens} - {one-line title} +{the question still open, and why it exceeded this run: premise invalidated, new effort, or unanswered} + +## Strengths +{the good notes worth keeping, deduplicated} +``` + +## Wrap up + +Open the chat summary with the table from `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py end scope-review --count findings_verified=N --count findings_refuted=N --count refinements_applied=N --count escalated=N`, pasted verbatim. +Then summarize in the same message: the verdict, what was refined and what factory policy decided (so the loop's edits stay auditable after the fact), what was promoted to `docs/decisions.md` and `docs/contracts.md`, and a link to the report. + +## Factory context + +Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. diff --git a/plugins/factory/skills/scope-review/agents/openai.yaml b/plugins/factory/skills/scope-review/agents/openai.yaml new file mode 100644 index 0000000..c2a624a --- /dev/null +++ b/plugins/factory/skills/scope-review/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "Scope Review" + short_description: "Review and auto-refine a settled spec" + default_prompt: "Use $factory:scope-review to review the settled spec with a verified agent panel and refine it in place before building." +policy: + allow_implicit_invocation: false diff --git a/plugins/factory/skills/scope-review/references/lenses.md b/plugins/factory/skills/scope-review/references/lenses.md new file mode 100644 index 0000000..0e5fa5f --- /dev/null +++ b/plugins/factory/skills/scope-review/references/lenses.md @@ -0,0 +1,55 @@ +# Spec Review Lenses + +Each lens is one agent in the panel. +The artifact is `spec.md`; the repo is the evidence. +Every lens reads the whole spec but stays in its lane; prompt assembly and delivery follow ship's [orchestration.md](../../ship/references/orchestration.md) with the substitutions in this skill's SKILL.md. + +## feasibility + +Does the change plan survive contact with the repo? + +- Every file a change set names exists, or the set says it is new; the described edit is possible at that site - the function, hook, or config it assumes is really there. +- Prior art and idioms the spec cites exist where it says they do. +- Premises about current behavior are checked against the code, never trusted: an "X already handles Y" claim that is false is a BLOCK naming the site. +- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on. + +## completeness + +What will the implementer hit that the spec never mentions? + +- Failure paths and edge cases with no decision, no scope entry, and no test scenario. +- Second-order work the change implies but no change set carries: migrations, config, deploy steps, doc updates. +- The medium-decision trap: choices that look small but swing how much work is done, silently defaulted instead of argued. +- Call sites and consumers affected by the change but outside every change set's file list; Grep for them. + +## consistency + +The spec against itself and the project's ledgers, semantically - `lint-spec.py` already enforced the mechanics. + +- A change set that contradicts its linked `✓` decision, or quietly implements a `✗` rejected alternative. +- A scope invariant or error-handling entry that no change set enforces. +- A decision colliding with `docs/decisions.md` without acknowledging it, a plan that would break a `docs/contracts.md` guarantee while reliance sites still assume it, or a new dependency edge `docs/dependencies.md` forbids. + +## testability + +Will the `tests:` lines produce real proof? + +- Each scenario is provable at its tagged layer, and concrete enough that whoever writes it invents nothing - a vague scenario is a finding even when the feature is right. +- Scenarios cover the scope's invariants and failure paths, not just happy paths. +- A `tests: none - {reason}` whose reason does not hold is a finding. +- `[e2e]` scenarios are runnable in the mocked environment build uses; a scenario needing a live third party will fail there. + +## Shared rules (include in every lens prompt) + +Severity ladder: + +- **BLOCK**: implementation following this spec would fail or build the wrong thing; name the concrete site or scenario. +- **CONCERN**: a gap worth a conversation before build; include what you checked. +- **NIT**: minor polish. + +`file`/`line` point into `spec.md` unless a repo path is named. +Stay in your lane; other lenses cover the rest. +A finding based on a guess about unseen code is not a finding - read the repo first. +Every finding needs a location, a one-line title, and a 2-4 line detail naming the evidence. +Also report up to 5 genuinely good things your lens noticed. +Output contract: the same JSON as ship's lens agents, defined in [orchestration.md](../../ship/references/orchestration.md). diff --git a/plugins/factory/skills/scope/SKILL.md b/plugins/factory/skills/scope/SKILL.md new file mode 100644 index 0000000..4dc4446 --- /dev/null +++ b/plugins/factory/skills/scope/SKILL.md @@ -0,0 +1,162 @@ +--- +name: scope +description: Spec a change by interviewing for the real problem, arguing every design decision against alternatives, and writing a self-contained spec with a change plan at .dev/{plan-name}/spec.md. Use to plan work before implementation, to turn accepted review findings into new change sets, or to audit the decisions already embedded in existing code or commit history. +--- + +# Scope + +The spec is the deliverable: do not implement anything. +It is a fresh-context handoff - planning happens here, implementation happens in another session (often another model) that reads only the spec. + +The numbered sections are an addressing scheme, not a march. Four orderings are load-bearing: catalog decisions before arguing them, argue them before reading the project's ledger, settle the spec before promoting anything out of it, and never let the change plan outrun an open decision. The rest is judgment - work it in whatever order the change calls for. +Before the interview, run `python3 {scope-skill-root}/../../scripts/skill-metrics.py start scope` so the retrospective can measure the run. + +## 1. Interview + +Interview the user until you understand what they want, one consequential question at a time, using the host's structured user-input tool when available. Then write down what you understood. +Whenever you recommend an option - here, in the decision talk-through, or in follow-up questions - put a confidence score next to it (`Confidence: 70%`) with the one fact that would most change it. +Score how likely it is the right solution given the evidence, not how strongly you prefer it: a taste call, an unverified premise, or an option chosen without having read the relevant code scores low. +A recommendation at `Confidence: 75%` or above is not a question: apply it and keep going, and write `auto-applied at Confidence: NN%` into the chosen line's because clause so the ledger shows who decided. +Ask only below 75%, and, whatever the score, when the recommendation changes what was asked for: a different problem than the request names, a requirement dropped, or a new non-goal. +At the end of the interview, list every auto-applied recommendation once, in one block, so a single reply can overturn any of them before the spec is written down. + +The request often arrives one level too low: a solution ("add rate limiting with Redis") hides the problem it solves, a symptom hides the cause that picks the fix. +Climb up before interviewing about the change itself - what led to this, what they observed, what would look different if it worked - then pressure-test the premise: what data shows the problem is real, and does it point where they think? Committing to a fix before the cause means speccing the wrong change well. +Once the problem holds up, brainstorm alternatives at the problem level, including other layers entirely (rate limit at the ALB instead of in the app); the user's original ask becomes one alternative among them, and the whole question lands in the research section as the first decision. +If the premise doesn't survive, that resolves as a `⊘` not-doing line, not as a failed interview. + +The interview is latency-bound by the user, so hide machine work inside it: once the change's rough area is clear, launch background read-only subagents on phase 2's prior-art hunt - existing idioms, helpers, earlier attempts, the areas the change touches - so their results are waiting when the interview closes. +They start from `docs/architecture.md` ([../../references/architecture.md](../../references/architecture.md)), the high-level overview of the system; when the project has none, the same explorers run its initial capture first, so the spec is argued against an overview instead of a file tree. +These explorers never read `docs/decisions.md`: the ledger must stay out of context until the phase 5 reconcile, and phase 9 treats an early leak as contamination. + +Then size the change: **small** - a handful of decisions, an area the user knows, limited blast radius - or **full**, everything else. +Tell the user which you picked and why; they can override. Small changes run the lighter variants where marked below - the spec file, its argued decisions, scope, and change plan always happen. + +**Iterate small.** Bias toward the smallest spec that delivers observable behavior - a slice implemented and looked at teaches more than a bigger plan. +A big request becomes a first slice specced now, with later slices as `⊘` lines naming what reopens them; reopen the spec per "Starting from review findings" once build and ship land. The ledger carries the decisions across slices, so going small loses nothing argued. + +## 2. Catalog the decisions + +Look at the code, starting from the phase 1 explorers' results - prior art first: existing idioms, helpers, and earlier attempts - so decisions get argued against what the codebase already does, not from scratch. Identify all the design decisions involved: + +- **Big** - overall approach. +- **Medium** - things that seem simple but can cause big shifts in how much work is done (store a file on S3 or locally, retry strategy). The trickiest tier - make sure you catch all of them. +- **Small** - timeout lengths, shapes of data structures. + +Pay special attention to edge cases and error handling. How the change gets verified is a decision too: what level to test at, what needs a real dependency versus a fake, what can't be tested and why. + +**Blind spot pass** (full-size only): don't grade your own catalog - you'll reread it the way you wrote it. Get a second one from something that hasn't seen your reasoning, reliably a subagent handed only the user's original request and the relevant code - not the conversation, not your catalog. +Both of its inputs are settled when the interview ends, so launch it before you start cataloging; it hunts while you do, and the fold-in is the sync point. +Fold the diff in: what it found and you didn't are blind spots, what you found and it didn't deserves a second look. +Then tell the user what their framing didn't account for: constraints already in the code, behavior the change would break, second-order work, and what a mature solution handles in this domain that they wouldn't know to ask about. + +Between the catalog and the talk, gather each big and medium decision's evidence - call sites, existing constraints, what the code already does - in one parallel subagent batch, one agent per decision area, so each discussion starts armed instead of pausing to grep. +Talk through the big and medium decisions with the user, highest-impact first; when a decision's criteria are taste-driven, sketch the alternatives concretely instead of describing them in prose. + +## 3. Research section + +Pick a kebab-case `{plan-name}` naming the outcome and write `spec.md` in the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)), dated under its title, in three sections: `## Research`, `## Scope`, `## Change plan`. +Research comes first, one entry per decision, ordered by impact. +Every decision gets a stable ID: `D-` plus a short kebab slug (`D-file-storage`); keep a slug once assigned. + +Each entry: the decision phrased as a question, then one line per alternative in the mark and because-clause notation of [../../references/decision-ledger.md](../../references/decision-ledger.md), which owns it. +Evidence marks apply from phase 1 onward, so the premise pressure-test data lands here cited or marked. +A decision still being argued is marked `[open]`; one that resolves to not doing it gets `✗` on every alternative and a closing `⊘`. +`⚑` marks a line waiting on the user. It may stay open in research and scope, but the change plan never links a decision that is `[open]` or carries one. + +``` +D-input-validation: How should input be validated? + ✓ custom validation - rules fit in ~20 lines, no new dependency ⚠ we maintain edge cases ourselves + ✗ validation library - adds a dependency for one call site + +D-file-storage: Where do uploaded files live? [open] + ? S3 - survives redeploys, needs bucket + IAM work + ⚑ ask: expected file sizes and retention? + +D-response-caching: Should responses be cached? + ✗ Redis - operational burden for an unproven need + ⊘ not doing - no measured latency problem ? verify: p95 from prod metrics; reopen if p95 exceeds 500ms +``` + +When a later section references a decision, echo the resolution in parentheses - `D-file-storage (✓ S3)`, `(open)`, `(⊘ not doing)` - so the reader only jumps back for the why. + +## 4. Scope section + +Record the scope: inputs, outputs, invariants, and error handling. +An invariant that crosses a boundary - another module will rely on it without seeing the enforcing code - gets drafted in contract notation ([../../references/contracts.md](../../references/contracts.md)) so phase 8 can promote it. +A change that adds a module or dependency edge, or collides with `docs/dependencies.md`, is a decision with the alternatives named in [../../references/dependency-rules.md](../../references/dependency-rules.md); the chosen resolution edits that file inside a change set, never implicitly. +Name the components and flows the change adds, removes, or reshapes, in the overview's terms, so build knows what in `docs/architecture.md` its change sets must update. +Efforts have second-order effects - capture them as nested sub-efforts, each carrying its own decisions back into the research section (rate limiting in scope means Redis setup, which carries config and deploy decisions). +Record considered non-goals as `⊘` lines with a because clause - things someone weighed and cut, not mere omissions. +End the scope with a `### Validation` block listing the repo's real typecheck/test/lint/build commands, discovered from `package.json`, a `Makefile`, CI config, or equivalent - never guess `npm test` into a `pytest` repo; ask if you cannot determine them. +Writing style for the spec: ELI12, no similes or metaphors. + +## 5. Review and research + +1. Spawn a subagent reviewer with the spec file only - not the conversation, not your reasoning - before starting any of the work below, so it hunts while you research. Give it a hunting job, not a checklist: find the decisions this spec makes without realizing it, the alternatives rejected without a stated reason, and the chosen options whose downsides the spec doesn't admit. It succeeds by finding problems; "looks complete" is a failed review. For small changes, run this hunt yourself against the spec file. +2. While it runs, research how to implement everything: exact call sites, APIs, the idioms from the phase 2 prior-art hunt. Ask the user when you hit weird stuff. +3. Also while it runs, reconcile the drafted decisions against the project's `docs/decisions.md`, if it keeps one, per the recommender contract in [../../references/decision-ledger.md](../../references/decision-ledger.md). The decisions were argued fresh, so this diff means something; a `diverged` classification is raised with the user and marked `⚑` until resolved. +4. When the reviewer returns, ask remaining questions and update the doc. Anything still unanswerable stays a `⚑` line. + +## 6. Change plan + +Short fragmented sentences. Link decisions by ID wherever one applies, echoing the choice. +Each change set ends with one `;`-separated `tests:` line - concrete scenarios as input -> expected outcome, each tagged `[unit]`, `[integration]`, or `[e2e]`, covering happy path, edge cases, and failure paths; a set with nothing to test says `tests: none - {reason}`. +Specific enough that whoever writes the tests invents nothing; the author tags layers here because a fresh implementation session can't recover that intent. +Order change sets so each builds only on the ones before it; keep file lists disjoint where possible - `build` parallelizes consecutive change sets whose files don't overlap. + +``` +1. Change set 1 + a. file 1 - describe changes - decisions: D-input-validation (✓ custom) + b. file 2 - describe changes + tests: [unit] payload missing name -> 400 naming the field; [integration] valid payload -> 200 and row written; [e2e] 11MB upload -> rejected before the transfer starts + +2. Change set 2 + a. file 3 - describe changes + tests: none - deploy config only, verified by the deploy itself +``` + +Then loop `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` until it exits clean. +It owns the mechanics above; the spec is not final while it reports anything. + +## 7. Visualize + +Full-size changes only. Map the settled spec onto `SPEC_DATA` per [references/data-schema.md](references/data-schema.md) and render [templates/spec.html](templates/spec.html) to `/tmp/{project-slug}/reports/{plan-name}-spec.html`, opening and publishing per [../../references/reporting.md](../../references/reporting.md). +Once the spec is settled, this phase, the phase 8 promotion, and the Jira sync have no ordering between them - overlap them rather than running a march. + +## 8. Promote to the ledger + +Copy into `docs/decisions.md` every decision that passes the promotion test in [../../references/decision-ledger.md](../../references/decision-ledger.md). +Entries go in verbatim, dated, sourced to this spec, evidence marks included; a `? verify:` never gets silently dropped, and promotion is the cheapest moment to check what's checkable now. +Promote recurring rationales to `P-` principles per the ledger's bar. +Cross-boundary invariants from phase 4 promote to `docs/contracts.md` under the same test, phrased for the relying side. + +## 9. Run retrospective + +Audit the run itself. Three checks, each reported as a typed line - silence reads as "never ran": + +- **Contamination** - `clean | contaminated`. Did the ledger enter context before the phase 5 reconcile? If so, name the decisions drafted after exposure: their reconcile outcomes are inherited, and a `still-holds` on them proves nothing. +- **Sizing** - `held | mis-sized`. Did the small/full call survive? Name the evidence when it didn't. +- **Catalog gaps** - `none | gap`. What did the reviewer or blind-spot pass find that phase 2 should have caught? A missed *category* is a proposed edit to the phase 2 list - propose it to the user, never apply it silently. + +Then print the measured run metrics with `python3 {scope-skill-root}/../../scripts/skill-metrics.py end scope --count decisions=N --count change_sets=N --count scenarios=N`, pasting its table verbatim. + +## Factory context + +Read `request.md` in the plan directory as the request; the `run` skill already created the directory and named the plan, so use it rather than naming one. This phase keeps its interview - it is the one interactive phase and is exempt from `check_factory_unattended`'s scan - and still ends with one explicit go question. Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. + +## Jira sync + +Read `.dev/config.json`; when `jira.enabled` is true, follow [../../references/jira.md](../../references/jira.md) from before the first `acli` call - it owns the command shapes, the Initiative/Epic/Task timing, and the failure protocol. +With an absent or disabled config, no Jira behavior or mention. + +## Starting from review findings + +When the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)) holds `review_N.md` or `spec-review_N.md` files, the highest-numbered review's accepted Blockers and Concerns are the interview's opening agenda: each becomes an open decision, and rejected or deferred findings land as `⊘` lines so they are visibly not dropped. +Append new change sets to the existing `spec.md` with continued numbering - never renumber - and apply the review's Decision reconciliation section to `docs/decisions.md`. +A defect change set includes the review's triggering scenario as its reproduction. + +## Reverse mode + +When the user points at existing code instead of a planned change ("audit the decisions in the sync layer"), follow [references/reverse-mode.md](references/reverse-mode.md): inventory the decisions already embedded in the code, reconcile them against the ledger, and write ratified entries and unexamined defaults out - no spec file is produced. +To seed a ledger from commit history instead of code, follow [references/bootstrap.md](references/bootstrap.md). diff --git a/plugins/factory/skills/scope/agents/openai.yaml b/plugins/factory/skills/scope/agents/openai.yaml new file mode 100644 index 0000000..2d9fa40 --- /dev/null +++ b/plugins/factory/skills/scope/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "Scope" + short_description: "Spec a change by arguing its decisions" + default_prompt: "Use $factory:scope to spec this change with argued decisions and a change plan." +policy: + allow_implicit_invocation: false diff --git a/plugins/factory/skills/scope/references/bootstrap.md b/plugins/factory/skills/scope/references/bootstrap.md new file mode 100644 index 0000000..4a1b3d8 --- /dev/null +++ b/plugins/factory/skills/scope/references/bootstrap.md @@ -0,0 +1,72 @@ +# Ledger Bootstrap + +Build `docs/decisions.md` in one pass from commit history. +The commit skill's message format records `Why:` / `Considered:` / `Constraint:` / `Directive:` bodies and `Severity:` / `Risk:` trailers - in a repo that has used it for a while, the decisions are already written down, scattered across hundreds of commits. +Bootstrap lifts the durable ones into the ledger; after that, commit-time capture (the ledger-capture step of the commit skill's sync-checks reference) keeps it current and this never runs again. + +Run for the whole repo or scoped to an area the user names. + +## 1. Harvest + +Inventory candidate commits - catalog only, no qualifying yet: + +```bash +# Commits that argued an alternative - the strongest decision signal +git log --format='%H %s' --grep='Considered:' --all-match + +# Commits whose Why: defends a tradeoff (harvested in step 2's read) +git log --format='%H %s' --grep='Why:' + +# Trailer-carrying commits, for risk lines +git log --format='%H %s' --grep='Severity:\|Risk:' +``` + +Group the candidates by area using the commit scope slug (`feat(auth): ...` -> area `Auth`). +Scope slugs are the area vocabulary - don't invent a second taxonomy. + +## 2. Extract + +For each area, read the candidate commits' full bodies (`git show --format=fuller --no-patch `) and draft `D-` entries in ledger notation: + +- The decision as a question, from `What:` + `Why:` context. +- `✓` the chosen approach, because clause distilled from `Why:`. +- `✗` each rejected alternative from `Considered:`, with its stated reason. +- `⚠` any downside the body admits. +- Dated with the commit date, sourced to the hash. + +Qualification bar is the ledger-capture step's: a choice qualifies when someone could plausibly argue it differently later - `Considered:` exists, or the `Why:` defends a tradeoff. +Routine implementation narration doesn't qualify no matter how detailed. + +Evidence marks ([../../../references/decision-ledger.md](../../../references/decision-ledger.md) Notation): a claim the body makes from observation - "we saw timeouts in prod" - is a dated claim sourced to the hash; that's a citation, and its staleness is judged at read time. +Reserve `? verify:` for evidence the body asserts but nothing ever demonstrated. +Don't blanket-mark extracted entries - a ledger seeded from history where every clause carries a `?` has buried the mark. + +The same read harvests contracts ([../../../references/contracts.md](../../../references/contracts.md)): a body stating a boundary guarantee other code relies on - most often in `Directive:` or `Constraint:` fields ("callers must not retry", "never returns partial results") - drafts a `C-` entry for `docs/contracts.md`. +Before recording, confirm the guarantee against the current code and cite the enforcing function; a guarantee only history asserts gets `? verify:`. +History is a thin source for contracts - the boundaries themselves are the real one, and an audit walk seeds the registry better than commits do; harvest opportunistically here, don't sweep for them. + +For a large history, spawn one subagent per area in parallel, each returning drafted entries for its area. +Subagents get the commit bodies as their source - not the ledger, not each other's drafts. + +## 3. Reconcile + +Each drafted entry gets a typed disposition - report the counts per area: + +- `recorded` - enters the ledger. +- `superseded` - a later commit revisited the same decision; the latest resolution is recorded, this entry is folded into it (earlier `✗` alternatives are kept - they're the argument history). +- `stale` - the code the decision governed no longer exists (verify: the touched files or the chosen mechanism are gone). Not recorded; a decision about deleted code is trivia. +- `unqualified` - didn't pass the bar on a closer read. + +Dedupe by slug across areas before writing. + +## 4. Risk lines + +Mechanical, from the trailer harvest, per the area-header semantics in [../../../references/decision-ledger.md](../../../references/decision-ledger.md): domains are the all-time union of `Risk:` values per area (sticky facts); the level is the max `Severity:` across the **recent window only** - last 90 days of trailer commits - because levels observe recent change, they don't ratchet from history. +Date each line `(updated , bootstrap)`. + +## 5. Write and present + +Assemble `docs/decisions.md` from the ledger layout: principles (only if a rationale already recurs across 3+ extracted decisions - the bar doesn't lower for bootstrap), then area sections with risk lines and entries ordered by date. +Present the per-area summary - entries recorded, superseded, stale, unqualified - and let the user prune before committing the file. +Commit as `docs(decisions): bootstrap ledger from commit history`. +Harvested contracts assemble into `docs/contracts.md` the same way - presented for pruning with the rest, committed separately as `docs(contracts): bootstrap from commit history`. diff --git a/plugins/factory/skills/scope/references/data-schema.md b/plugins/factory/skills/scope/references/data-schema.md new file mode 100644 index 0000000..ee5aaf9 --- /dev/null +++ b/plugins/factory/skills/scope/references/data-schema.md @@ -0,0 +1,64 @@ +# SPEC_DATA Schema + +The shape to populate in `templates/spec.html` between the `SPEC_DATA_START` / `SPEC_DATA_END` markers. +Replace the whole object per the shared etiquette in [../../../references/reporting.md](../../../references/reporting.md). + +```js +const SPEC_DATA = { + title: "string - the spec's title", + planName: "string - the .dev/{plan-name} slug", + date: "string - YYYY-MM-DD, the spec header date", + summary: "string - one or two sentences: the problem and the chosen direction", + + // One entry per decision, in the research section's impact order. + decisions: [ + { + id: "string - the D- slug, e.g. D-file-storage", + question: "string - the decision phrased as a question", + status: "decided | open | not-doing", + alternatives: [ + { + mark: "chosen | rejected | open", // renders as ✓ / ✗ / ? + text: "string - the alternative", + because: "string - the because clause, evidence marks included", + downside: "string - the ⚠ clause; omit when none" + } + ], + notDoing: "string - the ⊘ line with its reopen condition; only when status is not-doing", + flags: ["string - each ⚑ line still waiting on the user"] // omit or [] when none + } + ], + + scope: { + overview: "string - the one-sentence overall task", + efforts: [ + { + title: "string - effort name", + description: "string - one sentence", + decisions: ["D-slug (✓ choice)", "..."], // echoes, omit when none + children: [ /* same shape, nested sub-efforts */ ] + } + ], + nonGoals: ["string - each ⊘ line with its because clause"], + invariants: ["string - inputs/outputs/invariants/error handling worth surfacing"], + validation: ["string - each real repo command from the Validation block"] + }, + + changeSets: [ + { + n: 1, // matches the change plan numbering + title: "string - one line", + items: [ { file: "string - path", change: "string - what happens there", decisions: ["D-slug (✓ choice)"] } ], + tests: [ { layer: "unit | integration | e2e | none", scenario: "string - input -> expected outcome, or the none-reason" } ] + } + ] +}; +``` + +## Filling it in honestly + +- **This object mirrors `spec.md`, section for section** - decisions, scope, and change sets come straight from the file you just wrote. A section left empty here reads as a gap in the spec. +- **Marks are the content** - the because clauses, ⚠ downsides, and ⊘ reopen conditions are why the artifact is worth opening; never flatten them into bare labels. +- **flags belong to open decisions only** - a decision the change plan links never carries one. Surface any that remain loudly rather than hiding them. +- **Don't invent a decision that wasn't argued** - a decision with one alternative and no because clause is a statement, not a decision; leave it out or argue it first. +- **tests entries are the acceptance criteria** - keep layer tags accurate; `build` and `ship` treat them as the enforceable spec. diff --git a/plugins/factory/skills/scope/references/reverse-mode.md b/plugins/factory/skills/scope/references/reverse-mode.md new file mode 100644 index 0000000..22bce4e --- /dev/null +++ b/plugins/factory/skills/scope/references/reverse-mode.md @@ -0,0 +1,20 @@ +# Reverse Mode: Recover Implicit Decisions + +When the user points at existing code instead of a planned change - "audit the decisions in the sync layer", "what did the AI decide here?" - the change-speccing phases invert into an inventory of decisions already made but never argued. + +There's no change to understand, so there is no interview. + +1. **Catalog in two passes.** + First pass: inventory every embedded decision without judging any - timeouts, retry counts, limits, page and buffer sizes, validation rules and their gaps, storage and serialization choices, concurrency and ordering assumptions, error-handling policies, defaults of any kind. + Include values that look fine; judging while cataloging skips them. + Second pass: for each entry, ask what problem the value was solving. + Record the answer - or `no known problem - unexamined default`. + That line is the discriminator: those decisions were never made by anyone and are up for grabs. +2. **Reconcile** against the project's decision ledger, if it keeps one - `docs/decisions.md`, format in [../../../references/decision-ledger.md](../../../references/decision-ledger.md). + Catalog entries already recorded -> verify the code still matches the recorded choice (a mismatch is `diverged` - raise it). + The rest are new. +3. **Talk through the new ones worth deciding**, ordered by blast radius if the value is wrong. + A decision the user ratifies gets `✓` with the real because clause. + One worth changing gets a recommended normal scope run; never launch it yourself. +4. **Write out.** + Reverse mode produces no spec file - everything lands in the ledger: ratified decisions and unexamined defaults not worth deciding, recorded with their `no known problem` line if they pass the promotion test so the next audit doesn't re-litigate them. diff --git a/plugins/factory/skills/scope/scripts/lint-spec.py b/plugins/factory/skills/scope/scripts/lint-spec.py new file mode 100755 index 0000000..2056f78 --- /dev/null +++ b/plugins/factory/skills/scope/scripts/lint-spec.py @@ -0,0 +1,208 @@ +#!/usr/bin/env python3 +"""Check a spec.md against the mechanical rules of the scope skill. + +Usage: python3 lint-spec.py .dev/{plan-name}/spec.md +Exit 0 when the spec is clean, 1 with one problem per line otherwise. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +SLUG = re.compile(r"^(D-[a-z0-9]+(?:-[a-z0-9]+)*):\s*(.+)$") +BAD_SLUG = re.compile(r"^(D-\S+):") +ALTERNATIVE = re.compile(r"^\s+([✓✗?⊘])\s+(.*)$") +FLAG = re.compile(r"^\s+⚑") +ECHO = re.compile(r"(D-[a-z0-9-]+)\s*\(([^)]*)\)") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) +LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") +HEADING = re.compile(r"^##\s+(.*?)\s*$") + +CHOSEN, REJECTED, OPEN, NOT_DOING = "✓", "✗", "?", "⊘" + + +def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: + """Split numbered lines into research / scope / change plan buckets.""" + found: dict[str, list[tuple[int, str]]] = {} + current = None + for number, line in enumerate(lines, start=1): + heading = HEADING.match(line) + if heading: + name = heading.group(1).lower() + current = name if name in ("research", "scope", "change plan") else None + if current: + found.setdefault(current, []) + continue + if current: + found[current].append((number, line)) + return found + + +def read_decisions(research: list[tuple[int, str]], problem) -> dict[str, dict]: + decisions: dict[str, dict] = {} + current = None + for number, line in research: + header = SLUG.match(line) + if not header and BAD_SLUG.match(line): + problem(number, f"{BAD_SLUG.match(line).group(1)} is not a kebab-case D- slug") + current = None + continue + if header: + slug, question = header.group(1), header.group(2) + if slug in decisions: + problem(number, f"{slug} is defined twice; slugs are unique and stable") + current = decisions.setdefault( + slug, {"line": number, "open": "[open]" in question, "marks": [], "flagged": False} + ) + continue + if current is None: + continue + if FLAG.match(line): + current["flagged"] = True + continue + alternative = ALTERNATIVE.match(line) + if not alternative: + continue + mark, text = alternative.group(1), alternative.group(2) + current["marks"].append(mark) + if " - " not in text: + problem(number, "alternative has no ' - because' clause") + if mark == NOT_DOING: + current["not_doing"] = True + if "reopen" not in text.lower(): + problem(number, "not-doing line states no condition that would reopen it") + if mark == CHOSEN: + current["choice"] = text.split(" - ")[0].strip().lower() + return decisions + + +def check_decisions(decisions: dict[str, dict], problem) -> None: + for slug, decision in decisions.items(): + chosen = decision["marks"].count(CHOSEN) + # An [open] decision is still being argued and may not have its alternatives yet. + if not decision["open"] and len(decision["marks"]) < 2: + problem(decision["line"], f"{slug} argues fewer than two alternatives") + if decision.get("not_doing"): + if chosen: + problem(decision["line"], f"{slug} is not-doing yet marks an alternative chosen") + elif decision["open"]: + if chosen: + problem(decision["line"], f"{slug} is [open] yet marks an alternative chosen") + if OPEN not in decision["marks"]: + problem(decision["line"], f"{slug} is [open] with no ? alternative") + elif chosen != 1: + problem(decision["line"], f"{slug} has {chosen} chosen alternatives; expected exactly one") + + +def check_echoes( + body: list[tuple[int, str]], decisions: dict[str, dict], in_change_plan: bool, problem +) -> None: + for number, line in body: + for slug, echo in ECHO.findall(line): + decision = decisions.get(slug) + if decision is None: + problem(number, f"{slug} is echoed but never argued in the research section") + continue + resolution = echo.strip().lower() + if decision.get("not_doing"): + expected = resolution.startswith(NOT_DOING) + elif decision["open"]: + expected = resolution == "open" + elif resolution.startswith(CHOSEN): + shorthand = resolution[1:].strip() + expected = not shorthand or shorthand in decision.get("choice", "") + else: + expected = False + if not expected: + problem(number, f"{slug} echo '({echo})' does not match its resolution") + if in_change_plan and (decision["open"] or decision["flagged"]): + problem(number, f"the change plan links {slug}, which is still open or flagged") + + +def check_change_plan(body: list[tuple[int, str]], problem) -> None: + counts: dict[int, int] = {} + current = None + for number, line in body: + if "⚑" in line: + problem(number, "the change plan carries a ⚑; resolve it before the plan is final") + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + counts.setdefault(current, 0) + continue + tests = TESTS.match(line) + if not tests: + continue + if current is None: + problem(number, "tests: line outside any change set") + continue + counts[current] += 1 + value = tests.group(1).strip() + if value.lower().startswith("none"): + if " - " not in value: + problem(number, "'tests: none' states no reason") + continue + if not value: + problem(number, "tests: line is empty") + continue + for scenario in value.split(";"): + scenario = scenario.strip() + if scenario and not LAYER.match(scenario): + problem(number, f"scenario '{scenario[:40]}' carries no [unit]/[integration]/[e2e] tag") + for change_set, seen in sorted(counts.items()): + if seen != 1: + problem(None, f"change set {change_set} has {seen} tests: lines; expected exactly one") + + +def check_validation(lines: list[str], problem) -> None: + for index, line in enumerate(lines): + if line.strip().lower() == "### validation": + if any(rest.strip() and not rest.startswith("#") for rest in lines[index + 1 : index + 12]): + return + problem(index + 1, "the Validation block lists no commands") + return + problem(None, "no '### Validation' block in the scope section") + + +def main(argv: list[str]) -> int: + if len(argv) != 2: + print("usage: lint-spec.py ", file=sys.stderr) + return 2 + path = Path(argv[1]) + if not path.is_file(): + print(f"no spec at {path}", file=sys.stderr) + return 2 + lines = path.read_text(encoding="utf-8").splitlines() + + problems: set[tuple[int, str]] = set() + + def problem(number: int | None, message: str) -> None: + where = f"{path}:{number}" if number else str(path) + problems.add((number or 0, f"{where}: {message}")) + + found = sections(lines) + for name in ("research", "scope", "change plan"): + if name not in found: + problem(None, f"no '## {name.title()}' section") + + decisions = read_decisions(found.get("research", []), problem) + check_decisions(decisions, problem) + check_echoes(found.get("scope", []), decisions, False, problem) + check_echoes(found.get("change plan", []), decisions, True, problem) + check_change_plan(found.get("change plan", []), problem) + check_validation(lines, problem) + + for _, message in sorted(problems): + print(message) + if problems: + print(f"\n{len(problems)} problem(s); the spec is not final.") + return 1 + print(f"{path}: clean - {len(decisions)} decisions argued.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/plugins/factory/skills/scope/templates/spec.html b/plugins/factory/skills/scope/templates/spec.html new file mode 100644 index 0000000..75329a0 --- /dev/null +++ b/plugins/factory/skills/scope/templates/spec.html @@ -0,0 +1,151 @@ + + + + + +Decision Spec + + + +
+ + + diff --git a/plugins/factory/skills/ship/SKILL.md b/plugins/factory/skills/ship/SKILL.md new file mode 100644 index 0000000..7ac4945 --- /dev/null +++ b/plugins/factory/skills/ship/SKILL.md @@ -0,0 +1,142 @@ +--- +name: ship +description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every check passes, phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. +--- + +# Ship + +Ship a change in three phases: a deterministic gauntlet that fixes mechanics, a review panel that judges meaning, then a pull request that carries proof the change works. +Prompted quality rules soften into guidelines as a context grows; a checker's exit code does not, and judgment belongs to a verified panel, not to one context. +You are the orchestrator for both phases: run tools, dispatch agents, aggregate, report. +Never weaken a check to make it pass, and never stand in for the panel - your own reading of the code is not a lens. + +Invoking this skill is the task - detect the diff yourself and start immediately; do not ask what to ship. +First run `python3 {ship-skill-root}/../../scripts/skill-metrics.py start ship` so the wrap-up can measure this run instead of recalling it. + +## Phase selection + +Default is phase 1 (gauntlet), phase 2 (review), then phase 3 (pull request): the panel reviews the diff as it stands after the gauntlet's fixes, so its findings are about meaning, not mechanics already settled, and the PR carries both verdicts. +On request, run a single phase: "gauntlet only" (or "harden only") runs phase 1 alone; "review only" runs phase 2 alone - the right mode when mutating fixes are unwanted, such as on someone else's PR. "No PR" or "local only" runs the default flow without phase 3. +The phases differ in contract: phase 1 mutates the repo (fix agents edit code, tools get committed, accepted thresholds land in `docs/decisions.md`); phase 2 is strictly read-only and never edits code or the ledger, but a BLOCK verdict hands the diff to the remediation loop below, which mutates like phase 1; phase 3 commits, pushes, and opens or updates the PR, never force-pushing. + +## Scope + +Both phases share one scope. + +- PR number given -> target that PR; else the host's PR tooling, when it has any, for an open PR; else the local branch against the default branch. +- Gather the diff per the diff-scope rules in [../../references/plan-layout.md](../../references/plan-layout.md): local git only, standard exclusions. +- Locate the plan directory by the same reference's convention and read its `spec.md`; degrade gracefully without one. +- Full-repo runs only when the user asks - they are expensive, and the loops are the same. +- No reviewable files: report `failed` with that reason and stop. Diff is tiny (1-2 files) or huge (>25k lines): proceed and note it in `auto_decided` rather than spending extra agents confirming. + +## Phase 1: the gauntlet + +Eight checks, cheapest first - the five read-only analyzers scan as one parallel batch, the three test-contending ones run sequentially, per the loop reference; [references/tools.md](references/tools.md) owns their definitions and the acquisition ladder. +Never weaken or skip a check because acquiring its tool is work. + +1. **Project static analysis** - every linter, type checker, and format checker the repo already configures, run over the in-scope files. +2. **Security scan** - secrets, vulnerable dependencies, and static security rules; a found secret escalates immediately, it is never quiet fix-agent work. +3. **Dead code** - symbols the diff added that nothing references, and symbols it orphaned by removing their last caller. +4. **Duplication** - token-level clones the diff introduced against the rest of the repo. +5. **Dependency rules** against `docs/dependencies.md` ([../../references/dependency-rules.md](../../references/dependency-rules.md) owns the checker semantics; absent file: skip with a clear note, never invent rules). +6. **Coverage-weighted complexity** per in-scope function. +7. **Flakiness** - the tests the diff added or touched, repeated and shuffled until trusted. +8. **Mutation testing** over the in-scope source. + +Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). +When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. +Before phase 2, always run the required pull-request commands per +[../../references/ci-parity.md](../../references/ci-parity.md), even when no +fix agent fired. Feature E2E is not a substitute for a repository screenshot, +packaging, or report job that CI requires. + +## Phase 2: the review panel + +Review the diff with a panel of concern-focused agents, verify every finding against the repo, and write the report as a plan artifact. + +### 1. Load intent + +The review checks the diff against what was specced, not just general quality. + +- Write the diff to a scratch file; agents read it from there, not inline. +- Read the target repo's `docs/contracts.md` first, if it exists ([../../references/contracts.md](../../references/contracts.md)): its boundary guarantees are premises. Excerpt the entries whose guarantee or reliance sites the diff touches into the brief handed to every lens; a diff that breaks a cited guarantee while reliance sites still assume it is finding material. +- From the spec read in Scope: the scope section for invariants, boundaries, and error handling, the change plan for per-set files and layer-tagged `tests:` scenarios, and the research section for the decision rationale used to judge deviations. +- Also read `.dev/{plan-name}/implementation-notes.md` if present (deviations logged during implementation) and the latest `/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`'s data block, per the reading rules in [../../references/reporting.md](../../references/reporting.md). Both are additional evidence for the `spec-conformance` and `tests` lenses, not just the diff itself. +- No spec: derive `{plan-name}` from the branch name and fall back to the PR body and commits for intent. +- Build a short brief: what the change does, what's on the critical path, what was specced. + +### 2. Check for previous reviews + +`review_N.md` files present -> this is a re-review: extract unresolved findings from the highest-numbered one as verification items (fixed or not?), and the new report gets the next index. + +### 3. Select the panel + +Read [references/lenses.md](references/lenses.md); select only lenses with surface in this diff (a docs-only change skips performance). +Include `spec-conformance` whenever a spec was found, and always include `simplify` - every diff has simplification surface. +At most one diff-specific custom lens (migrations, concurrency, i18n) when clearly warranted, defined in the same shape as the built-ins. +Tell the user which lenses you selected and why before launching. + +### 4. Run the review panel + +Run the two batches - all lens agents first, then all verifiers - on the transport selected by [references/orchestration.md](references/orchestration.md), which owns transport choice, result-file delivery, and the prompt contracts. +If no transport is available, stop and explain that this panel-based phase cannot preserve its verification contract. +Between the batches, number the findings the verifiers get: + +```bash +python3 {ship-skill-root}/scripts/aggregate-findings.py plan {batch-1 results} +``` + +Every BLOCK and CONCERN it lists is adversarially verified - the verifier's only job is to refute it against the actual repo. + +### 5. Aggregate + +The same script applies the verdicts, so these rules live in one place instead of softening across a long context: + +```bash +python3 {ship-skill-root}/scripts/aggregate-findings.py aggregate {batch-1 results} {batch-2 results} --expected {lenses you selected} +``` + +It drops REFUTED findings, keeps a BLOCK only when CONFIRMED and carries every other surviving finding as a CONCERN, merges duplicates by file:line while crediting each lens that found them, names the lenses that failed or returned nothing parseable, and sets the verdict. +A failed lens doesn't abort the review - the report names it, so the verdict is honestly partial. + +Then, with findings verified, read the target repo's `docs/decisions.md` if it keeps one, and classify each colliding finding per the recommender contract in [../../references/decision-ledger.md](../../references/decision-ledger.md). Read-only - this phase never edits the ledger; classifications land in the report's Decision reconciliation section, and ledger writes belong to the follow-up `scope` run. + +### 6. Write the report + +Write `.dev/{plan-name}/review_N.md` in the shape of [references/report-format.md](references/report-format.md), at the next free index. + +### 7. Publish the report + +Map the report onto `REVIEW_DATA` per [references/data-schema.md](references/data-schema.md) and render [templates/review.html](templates/review.html) to `/tmp/{project-slug}/reports/review_N.html`, opening and publishing per [../../references/reporting.md](../../references/reporting.md) (stable review favicon; title names the plan and review number). + +## Blocker remediation + +A BLOCK verdict is work before it is a question: run up to two rounds per [references/remediation.md](references/remediation.md), each a fresh-context fix pass over the confirmed blockers, a re-harden of the touched files, and a re-review at the next index. +PASS or CONCERNS ends the loop; a blocker still standing after round 2, or one an agent escalated as needing a spec or decision change, is recorded as a decision and stays in the PR body per [references/pull-request.md](references/pull-request.md), never a question. +Skipped in review-only mode. + +## Phase 3: the pull request + +Open the PR automatically, with evidence a reviewer can see before reading the diff, per [references/pull-request.md](references/pull-request.md). + +1. Commit what the gauntlet left uncommitted, branch off the default branch if still on it, and push. +2. Build the Evidence section from the e2e report with `python3 {ship-skill-root}/scripts/pr-evidence.py extract`, publishing frontend screenshots to the `pr-evidence` branch with its `publish` command; without an e2e report, capture the evidence now per the reference - screenshots for a UI, a labeled before/after pair otherwise, and a red-on-base, green-on-branch reproducing test for every bug fix. +3. Write `.dev/{plan-name}/pr.md` in the reference's body shape and loop `pr-evidence.py check` on it until it passes; the check, not your judgment, decides whether the proof is real enough. +4. Create the PR (draft when a real blocker survived remediation) or update the one that already exists, then follow its required checks to green per [../../references/ci-parity.md](../../references/ci-parity.md). + +## Wrap up + +Summarize whichever phases ran in one chat message, opening with the measured run metrics: + +```bash +python3 {ship-skill-root}/../../scripts/skill-metrics.py end ship --count violations_found=N --count violations_fixed=N --count violations_surviving=N --count findings_verified=N --count findings_refuted=N --count remediation_rounds=N --count blockers_cleared=N --count evidence_items=N +``` + +Pass only counters you tallied from tool output and the aggregate script; the table it prints (time, tokens, agents, tool calls, git delta, trend against earlier runs) is pasted verbatim, never retyped. +For the gauntlet, per tool: violations found, fixed, and surviving (with the recorded decision each survivor is waiting on); name the tools acquired or built this run and where they live; state the scope honestly - "hardened the diff" is not "hardened the repo". +For the review: the final verdict and top findings, linking every `review_N.md` this run wrote, the local HTML report, and the published URL when one was requested and created; per remediation round, which blockers cleared and which survived. +For the pull request: its URL, draft or ready, what the Evidence section shows and where it came from, and the state of its required checks. + +## Factory context + +A found secret is a `stopped` result with kind `secret.found`, naming the commit to purge; a destructive action outside the run branch is `stopped` with kind `action.destructive`. Neither is retried or overridden. `gh pr create` refusing with "none of the git remotes configured for this repository point to a known GitHub host" is reported `failed` with that text as the reason and the pushed branch plus `pr.md` in `artifacts[]` - never `stopped`, never retried as an auth fault. `pr_url` and `draft` go in the result's fields when a pull request opens. Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. diff --git a/plugins/factory/skills/ship/agents/openai.yaml b/plugins/factory/skills/ship/agents/openai.yaml new file mode 100644 index 0000000..249ecff --- /dev/null +++ b/plugins/factory/skills/ship/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "Ship" + short_description: "Harden, review, then open a PR with proof" + default_prompt: "Use $factory:ship to run the quality gauntlet, the verified review, and open the pull request with evidence for this change." +policy: + allow_implicit_invocation: false diff --git a/plugins/factory/skills/ship/references/data-schema.md b/plugins/factory/skills/ship/references/data-schema.md new file mode 100644 index 0000000..e1f7bac --- /dev/null +++ b/plugins/factory/skills/ship/references/data-schema.md @@ -0,0 +1,46 @@ +# REVIEW_DATA Schema + +The shape to populate in `templates/review.html` between the `REVIEW_DATA_START` / `REVIEW_DATA_END` markers, per the shared etiquette in [../../../references/reporting.md](../../../references/reporting.md). +This mirrors `review_N.md`'s structure exactly - it's the same content, published as the artifact a PR reviewer actually opens. + +```js +const REVIEW_DATA = { + title: "string - Review {N}: {title}", + planName: "string - the .dev/{plan-name} slug, or empty if no spec was found", + reviewIndex: 1, + verdict: "PASS" | "CONCERNS" | "BLOCK", + panel: ["lens", "..."], // lenses run + failedLenses: ["lens", "..."], // omit or [] if none failed + base: "string - branch or PR reference", + date: "string - ISO date", + summary: "string - 1-2 sentence summary, markdown-lite", + + specConformance: "string - the spec-conformance summary, markdown-lite; omit the section entirely (set to '') if no spec was found", + + previousFindings: [ + { finding: "string", status: "fixed" | "still open" } + ], // [] unless this is a re-review + + blockers: [ + { lenses: ["lens", "..."], file: "string", line: 0, title: "string", detail: "string - markdown-lite, the triggering scenario as confirmed by verification" } + ], + concerns: [ + { lenses: ["lens", "..."], file: "string", line: 0, title: "string", detail: "string" } + ], + nits: [ + { file: "string", lens: "string", issue: "string" } + ], + + whatsGood: [ + { lens: "string", note: "string" } + ], + + nextStep: "string - what to fix first and why, markdown-lite" +}; +``` + +## Filling it in honestly + +- This is a direct transcription of `review_N.md` into data, not a separate editorial pass - the two must agree. +- `blockers`/`concerns` only include findings that survived adversarial verification (CONFIRMED, or PLAUSIBLE demoted to CONCERN) - never a REFUTED finding. +- `previousFindings` is only non-empty on a re-review; omit the section in the template when empty rather than showing "none." diff --git a/plugins/factory/skills/ship/references/gauntlet.md b/plugins/factory/skills/ship/references/gauntlet.md new file mode 100644 index 0000000..1e7270b --- /dev/null +++ b/plugins/factory/skills/ship/references/gauntlet.md @@ -0,0 +1,56 @@ +# The Gauntlet Loop + +How phase 1 of `ship` runs its eight checks to completion. +The check definitions and their acquisition ladder live in [tools.md](tools.md); this file owns the loop, the fix agents, and the thresholds. + +## The loop + +The five read-only analyzers - static analysis, security, dead code, duplication, dependency rules - scan as one parallel batch: their initial runs mutate nothing, so collect all five violation lists concurrently, dispatch fix agents grouped by independent area across the combined lists, and re-run all five together until clean. +The expensive three - complexity x coverage, flakiness, mutation - stay sequential, cheapest first: they contend for the test runner, and each one's input shifts with every fix the previous one landed. + +For each tool (the batched five count as one), in order: + +1. Run it; collect the violations. +2. Dispatch fixes: one fresh-context agent per independent area, launched in a single message, each given only the violation list for its area, the relevant file paths, and the fix vocabulary below. +3. Re-run the tool until clean, then run the spec's Validation block (or the repo's test suite) to prove the fixes broke nothing; skip that run when the tool dispatched no fixes. +4. A violation that resists two fix rounds on the same root cause, or that the change seems to legitimately require, is recorded as a decision and left in place: it is either a real defect, a threshold worth changing, or a rule the spec should have amended, and the choice made is noted in `auto_decided` with the reason. + +## Exit: refresh the e2e evidence + +Fix agents edit code, so once any tool dispatched fixes, the e2e report build wrote no longer describes the code phase 2 will judge. +Before leaving phase 1, re-run the spec's `[e2e]` scenarios exactly as build does - same mocked environment, data mapping, and template, all owned by [../../build/SKILL.md](../../build/SKILL.md) and its references - overwriting `/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`. +A scenario that fails here is a violation like any other: dispatch fixes and loop under the same two-round rule. +Never write new e2e scenarios in this step; generation belongs to build, this step only re-executes. +Skip it when no tool dispatched fixes (the report is still fresh), when there is no spec or no e2e suite to run (note that in the wrap-up), or in review-only mode, which never reaches phase 1. + +## Exit: prove CI parity + +After the e2e refresh decision, run the required pull-request commands using +[../../../references/ci-parity.md](../../../references/ci-parity.md). This gate +always runs in the default two-phase flow and gauntlet-only mode, even when no +fix agent changed code. + +A red required check is a violation. "Pre-existing" requires merge-base proof, +recorded as a decision with that proof; it is never a note that permits a +PR-ready verdict. When a fix changes behavior or test orchestration, re-run +the affected command and continue until the full CI-parity set is green. + +## Fix vocabulary + +- Resolve a static analysis finding by fixing the code it points at, never by suppressing it inline or loosening the tool's config; a finding worth suppressing is recorded as a decision instead. +- Resolve a vulnerable dependency by upgrading it; an upgrade that breaks the build is recorded as a decision. A found secret is always an immediate `stopped` result with kind `secret.found`, never a quiet fix. +- Delete dead code outright; never comment it out or exclude it from the detector. +- Collapse a clone by extracting one shared helper or calling the one that already exists. +- Cut a complexity-coverage score by splitting the function or covering its paths. +- Resolve a dependency violation by inverting the dependency, inserting an interface, or splitting the module. +- De-flake a test by removing its nondeterminism (time, ordering, shared state, network), never by adding retries, sleeps, or looser assertions. +- Kill a mutant by adding the test that catches it. + +Fix agents never edit thresholds, rules files, or the tools themselves, and never delete a test to make a mutant moot. + +## Thresholds are decisions + +Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. +Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. +When the user accepts a different threshold, record it in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. +Never adjust a threshold silently to make a run pass. diff --git a/plugins/factory/skills/ship/references/lenses.md b/plugins/factory/skills/ship/references/lenses.md new file mode 100644 index 0000000..bb74f3e --- /dev/null +++ b/plugins/factory/skills/ship/references/lenses.md @@ -0,0 +1,102 @@ +# Review Lenses + +Each lens is one agent in the panel. +Select only lenses with surface in the diff; every lens sees the whole diff but stays in its lane. +Prompt assembly and delivery live in [orchestration.md](orchestration.md); this file owns the lens definitions and the shared rules. + +## spec-conformance + +Include whenever a spec directory was found. +Checks the diff against `spec.md` - its scope, change plan, and decisions - not against general quality: + +- Does each in-scope change set's `tests:` scenario exist as a real test, **at the layer it was tagged with** (`[unit|integration|e2e]`)? A scenario met by a test at the wrong layer (a unit test standing in for an e2e scenario) is a finding, not a pass. +- Do the scope section's invariants and error-handling entries hold in the diff? A broken invariant is a BLOCK. +- Does the diff implement each change set's `✓` decisions, without smuggling in a `✗` rejected alternative? A `⚠` accepted downside that manifests is expected, not a finding - the spec already admits it. +- Is a `docs/contracts.md` guarantee cited in the brief broken by the diff while its reliance sites still assume it? That is a BLOCK naming who gets hurt. +- Is every `[e2e]`-tagged scenario backed by a passing entry in `{plan-name}-e2e-report.html`? New user-facing behavior with no matching e2e scenario, or a scenario whose `status` is `fail`, is a BLOCK. +- Check `implementation-notes.md` for logged Deviations: does the conservative choice still satisfy the spec's scope and decisions, or does it need a call-out to the reviewer? An unresolved deviation that changes user-facing behavior from what the spec promised is a CONCERN at minimum. +- If a change set named a README or user-facing doc file, is that change present in the diff? A promised doc update that never happened is a CONCERN. +- Does the diff do significant work no change set describes? Scope creep is a CONCERN, not a crime; name it so the reviewer can decide. + +## correctness + +Bugs the compiler and tests would miss. +Walk the hardest paths first: concurrency and ordering hazards, boundary conditions (empty, one, max, off-by-one), null/absent paths, mutation while iterating, broken invariants, state-machine holes, unsafe retries, error paths, time and timezone handling, resource lifecycle, type-system escapes. +For each suspect path: state the invariant, name the input that breaks it, trace it through the code. +A BLOCK must name the triggering input; "this could race" without a scenario is a CONCERN. + +## security + +Authn/authz gaps, injection, secrets in code or logs, unsafe deserialization, crypto misuse, path traversal, SSRF, supply-chain risk in new dependencies, overly broad permissions. +Rate exploitability in this codebase, not theoretical severity. + +## architecture + +Module boundaries, layering, dependency direction, abstraction quality, coupling introduced by the diff. +Judge coherence against the patterns already established in the surrounding code; deviation from an established convention is a finding, personal style preference is not. +When the repo keeps `docs/dependencies.md`, treat its edges as settled: a conforming diff needs no boundary debate, and a diff that edits the rules file is reviewed as the decision it is, not as incidental churn. +Read `docs/architecture.md` first, when the repo keeps one, for orientation on where things live and how they talk ([../../../references/architecture.md](../../../references/architecture.md)): a diff that reshapes a component, flow, or boundary the overview describes without updating it is a finding. + +## tests + +Test quality per this plugin's philosophy, not raw coverage: + +- Tests must live at public seams; tests that mock internal collaborators, test private methods, or verify through side channels are implementation-coupled findings. +- Tautological tests (assertion recomputes the expected value the way the code does) are findings; expected values need an independent source of truth. +- New behavior in the diff without a test at its seam is a finding. (Missing e2e backing for user-facing behavior belongs to spec-conformance, not here.) +- Brittle patterns: global time/random patching, order-dependent tests, interaction assertions where a state assertion would do. + +## simplify + +Always include this lens; every diff has simplification surface. +The gauntlet's duplication, dead code, and complexity checkers are its deterministic front line, so spend this lens on what they cannot see: semantic duplication below the token threshold, generality nothing asked for, indirection that earns nothing. + +Unnecessary complexity that a simpler version would avoid, with behavior held constant: + +- Reinvented helpers: logic the codebase or standard library already provides; Grep for the existing helper before flagging and name it in the finding. +- Duplication introduced by the diff: the same logic now living in two places that will silently diverge. +- Speculative generality: abstractions, parameters, or config with exactly one caller and no planned second one. +- Needless indirection: layers that only forward calls, wrappers that wrap nothing. +- Wrong altitude: low-level mechanics inlined into high-level flow (or vice versa) that a small extraction would clarify. + +Every finding must sketch the simpler alternative concretely enough that the verifier can check it preserves behavior; "this feels complex" is not a finding. +Mostly CONCERN and NIT; reserve BLOCK for duplicating an existing tested helper. +This lens is about making the new code smaller and clearer, not about bugs (correctness), boundaries (architecture), or deleting unused code (dead-code). + +## performance + +Algorithmic complexity on real data sizes, N+1 queries, hot-path allocations, blocking I/O on async paths, missing pagination or streaming for unbounded data, cache invalidation. +Only flag what is on a path that plausibly matters; micro-optimizations are NITs at best. + +## api-design + +Naming, type signatures, error returns, consistency with the codebase's existing API idioms, backwards compatibility of anything already public. + +## errors-observability + +Swallowed errors, catch-and-continue without logging, missing context in log lines at decision points, errors that will be undebuggable at 3 AM, retry loops without backoff or limits. + +## docs + +Comments and docs touched or needed by the diff: stale or now-misleading comments, missing WHY comments on non-obvious constraints, public API docstrings. +Doc updates promised by the spec belong to spec-conformance, not here. + +## dead-code + +The gauntlet's detector already caught provably unreferenced symbols; this lens hunts what needs reading, not reference counting. +Unreachable branches, commented-out code, leftover debug scaffolding, vestigial parameters, dead flags and config keys, code only its own tests still call. +Verify with Grep across the whole repo before flagging; check tests, dynamic string-keyed lookups, and external API surface. +If you cannot prove it dead, report at most a CONCERN and say what you could not rule out. + +## Shared rules (include in every lens prompt) + +Severity ladder: + +- **BLOCK**: a concrete problem you can name with the scenario that triggers it; would misbehave in production or violates the spec. +- **CONCERN**: a path you cannot convince yourself is safe, or a gap worth a conversation; include what you checked. +- **NIT**: minor polish; skip NITs entirely on diffs over 15 files. + +Stay in your lane; other lenses cover the rest. +A finding based on a guess about unseen code is not a finding. +Every finding needs file, line, a one-line title, and a 2-4 line detail naming the evidence. +Also report up to 5 genuinely good things your lens noticed. diff --git a/plugins/factory/skills/ship/references/orchestration-heavy.md b/plugins/factory/skills/ship/references/orchestration-heavy.md new file mode 100644 index 0000000..3fa9a76 --- /dev/null +++ b/plugins/factory/skills/ship/references/orchestration-heavy.md @@ -0,0 +1,108 @@ +# Heavy Mode: Workflow Tool (explicit opt-in only) + +Use this only when the user has explicitly asked for a workflow or an exhaustive run; it triggers the dynamic-workflow confirmation dialog. +It buys schema-validated outputs, pipelining (findings verify while other lenses still review), and `/workflows` progress. +Pass `diffFile`, `brief`, `planContext`, `lenses` (as `{key, prompt}` with the full assembled lens prompt), and `priorFindings` via `args`. + +```js +export const meta = { + name: 'ship', + description: 'Concern-based diff review with adversarial verification of findings', + phases: [ + { title: 'Review', detail: 'one agent per selected lens' }, + { title: 'Verify', detail: 'refute-or-confirm each finding' }, + { title: 'Recheck', detail: 'prior findings fixed?' }, + ], +} + +const FINDINGS = { + type: 'object', + required: ['verdict', 'findings', 'good'], + properties: { + verdict: { enum: ['PASS', 'CONCERNS', 'BLOCK'] }, + findings: { + type: 'array', + items: { + type: 'object', + required: ['severity', 'file', 'line', 'title', 'detail'], + properties: { + severity: { enum: ['BLOCK', 'CONCERN', 'NIT'] }, + file: { type: 'string' }, + line: { type: 'integer' }, + title: { type: 'string' }, + detail: { type: 'string' }, + scenario: { type: 'string' }, + }, + }, + }, + good: { type: 'array', items: { type: 'string' }, maxItems: 5 }, + }, +} + +const VERDICT = { + type: 'object', + required: ['status', 'reason'], + properties: { + status: { enum: ['CONFIRMED', 'PLAUSIBLE', 'REFUTED'] }, + reason: { type: 'string' }, + }, +} + +const RECHECK = { + type: 'object', + required: ['fixed', 'evidence'], + properties: { + fixed: { type: 'boolean' }, + evidence: { type: 'string' }, + }, +} + +const { diffFile, brief, planContext, lenses, priorFindings } = args + +const lensPrompt = (l) => [ + `You are the "${l.key}" voice on a code-review panel; other lenses cover other concerns.`, + l.prompt, + `Brief: ${brief}`, + planContext ? `Plan context:\n${planContext}` : '', + `Read the full diff from ${diffFile}. Read surrounding source files when the diff is not enough.`, + `Return findings for your lens only.`, +].filter(Boolean).join('\n\n') + +const verifyPrompt = (f, l) => [ + `Adversarially verify this code-review finding from the "${l.key}" lens. Your job is to REFUTE it.`, + JSON.stringify(f), + `From the diff at ${diffFile}, read the hunks for the finding's file (search by path - do not ingest the whole diff), plus the actual files in the repo. Check whether the claimed problem is real:`, + `does the scenario actually trigger, is the code path reachable, does something else already handle it?`, + `CONFIRMED = you reproduced the reasoning against real code and it holds.`, + `PLAUSIBLE = you could not refute it, but could not fully confirm the scenario either.`, + `REFUTED = the finding is wrong, unreachable, or already handled; say exactly why.`, +].join('\n') + +const rechecks = priorFindings.length + ? parallel(priorFindings.map((f) => () => + agent( + `A previous review reported:\n${JSON.stringify(f)}\nRead the current code and the diff at ${diffFile}. Is it fixed now? Cite the fixing change or what is still missing.`, + { label: `recheck:${f.file}`, phase: 'Recheck', schema: RECHECK }, + ).then((r) => r && { ...f, recheck: r }))) + : Promise.resolve([]) + +const results = await pipeline( + lenses, + (l) => agent(lensPrompt(l), { label: `review:${l.key}`, phase: 'Review', schema: FINDINGS }), + (review, l) => { + if (!review) return null + const toVerify = review.findings.filter((f) => f.severity !== 'NIT') + return parallel(toVerify.map((f) => () => + agent(verifyPrompt(f, l), { label: `verify:${f.file}:${f.line}`, phase: 'Verify', schema: VERDICT }) + .then((v) => v && { ...f, lens: l.key, verification: v }))) + .then((verified) => ({ + lens: l.key, + good: review.good, + nits: review.findings.filter((f) => f.severity === 'NIT').map((f) => ({ ...f, lens: l.key })), + findings: verified.filter(Boolean), + })) + }, +) + +return { lenses: results.filter(Boolean), rechecks: (await rechecks).filter(Boolean) } +``` diff --git a/plugins/factory/skills/ship/references/orchestration.md b/plugins/factory/skills/ship/references/orchestration.md new file mode 100644 index 0000000..57cc2ec --- /dev/null +++ b/plugins/factory/skills/ship/references/orchestration.md @@ -0,0 +1,150 @@ +# Orchestration + +The review always uses independent agents in two batches. Select the transport +without changing the panel: + +Both transports use the same scratch root, such as `/tmp/ship-{session-id}/`, +with a `results/` directory per batch. Result files in that directory are the +only channel for agent output. A batch in flight means the run is not over: +never end your turn while launched agents are still outstanding - wait for the +host's completion signal, read the results, and continue. Never retrieve results by messaging an agent, +waiting for relayed replies, or re-spawning an agent that already finished; +a fresh spawn has none of the original context and its output must not be used. + +1. **Native transport (preferred):** when the host provides a multi-agent + facility, launch all agents in a batch in one native parallel call. This is + the existing Claude Code, Codex, and opencode behavior; do not route those + hosts through subprocesses. +2. **Pi transport:** when `PI_CODING_AGENT=true` and `pi` is available, write + one prompt file per agent and run + [scripts/run-pi-agents.sh](../scripts/run-pi-agents.sh). The runner launches + one isolated `pi --print` process per prompt concurrently and inherits the + current provider, model, and reasoning level. +3. **Unavailable:** if neither transport exists, stop. The orchestrator must + not impersonate the panel or review the code itself. + +No dynamic workflows are involved by default. A Workflow-tool variant for +explicitly requested heavyweight runs only lives in +[orchestration-heavy.md](orchestration-heavy.md); do not read it otherwise. + +## Native transport + +Use the host's plain subagent tool exactly as provided (the Agent tool on +Claude Code, `spawn_agent` on Codex, the `task` tool on opencode). Launch all +agents of a batch in a single parallel call: on opencode that means issuing +every `task` call of the batch in one message with the default general +subagent. Native subagents receive the prompt contracts below directly. + +Result delivery is file-based, because on some hosts subagents run in the +background and their final message never reaches the orchestrator: + +1. Before launching a batch, create `{scratch-root}/{batch}/results/`. +2. Each agent's prompt instructs it to write its JSON report to + `{scratch-root}/{batch}/results/{safe-agent-name}.json` as its last action, + then end with a one-line confirmation naming that path. The result file is + the deliverable; the final message is not. +3. When the host signals that the batch's agents have finished, read the + result files. +4. A missing or unparseable file after one re-read marks only that agent as + failed, same as a crashed agent. + +Agents stay read-only in the repository; their only write is their own result +file under the scratch root. + +## Pi transport + +Use the shared scratch root. For each batch: + +1. Create `prompts/` and `results/` directories dedicated to that batch. +2. Write one `{safe-agent-name}.md` prompt per agent. Include the complete + prompt contract from the relevant batch below. Tell the process to inspect + the repository but never modify it. +3. From the repository root, run: + + ```bash + bash {ship-skill-root}/scripts/run-pi-agents.sh \ + {batch-root}/prompts \ + {batch-root}/results + ``` + +4. Read each `{safe-agent-name}.out` as that agent's final response. A matching + `.err` or missing/unparseable `.out` marks only that agent as failed. +5. Keep the scratch files until aggregation is complete so malformed output is + auditable, then remove them. + +The runner disables child skills, extensions, prompt templates, and sessions +to prevent recursion and state leakage. It retains project context files and +read-oriented tools (`read`, `bash`, `grep`, `find`, `ls`) so each process can +inspect the diff, surrounding code, and tests independently. Do not add +`write` or `edit`. + +## Batch 1: lens agents + +Launch one agent per selected lens, all in a single message so they run concurrently. +Each lens prompt is assembled from: + +1. A role line: `You are the "{key}" voice on a code-review panel; other lenses cover other concerns.` +2. The lens definition plus the shared rules from `lenses.md`. +3. The brief, and the spec context when a spec was found. +4. The diff location: `Read the full diff from {diffFile}; the repo working tree holds the post-diff state. Read repo files whenever the diff alone is not enough to judge.` +5. The output contract below. + +Also state: `Do not modify repository files. This is a read-only review.` +On the native transport, add the result-file instruction from the transport +section; on Pi, the runner captures the final message instead, which must be +exactly one fenced JSON block. + +Output contract (the JSON every lens agent must produce): + +```json +{ + "verdict": "PASS | CONCERNS | BLOCK", + "findings": [ + { + "severity": "BLOCK | CONCERN | NIT", + "file": "path/relative/to/repo", + "line": 1, + "title": "one line", + "detail": "2-4 lines naming the evidence", + "scenario": "what triggers it (required for BLOCK)" + } + ], + "good": ["up to 5 bullets"] +} +``` + +If an agent errors or returns something unparseable, record it as a failed lens and continue. +Never retry a failed lens more than once. + +## Batch 2: verifiers + +`scripts/aggregate-findings.py plan` collects every BLOCK and CONCERN from the +lens results and numbers them (`f1`, `f2`, ...) so verdicts map back. NITs skip +verification. Duplicates are not merged yet: two lenses that flagged the same +line get refuted independently, and the merge happens after their verdicts. +Launch one verifier per finding, again all in a single message. +If that would exceed 10 verifiers, group the findings by file into at most 10 +shards and give each verifier its shard; each finding is still verified +independently, one verdict object per finding. +Verifier prompt: + +``` +Adversarially verify each of these code-review findings, reported by the +named lens. Your job is to REFUTE them. +{findings as JSON, each with its id} +From the diff at {diffFile}, read the hunks for your findings' files (search +by path - do not ingest the whole diff), plus the actual files in the repo. +Check whether each claimed problem is real: does the scenario actually +trigger, is the claim accurate against the files, does something else +already handle it? +Produce exactly one JSON array with one object per finding: +[{"id": "f1", "status": "CONFIRMED | PLAUSIBLE | REFUTED", "reason": "..."}] +CONFIRMED = you reproduced the reasoning against real files and it holds. +PLAUSIBLE = you could not refute it, but could not fully confirm it either. +REFUTED = wrong, unreachable, or already handled; say exactly why. +``` + +Delivery follows the transport: on native, the verifier writes the array to +its result file; on Pi, it replies with the array as one fenced JSON block. + +Aggregating the verified results is `aggregate-findings.py aggregate` per the skill's Aggregate step - never another agent. diff --git a/plugins/factory/skills/ship/references/pull-request.md b/plugins/factory/skills/ship/references/pull-request.md new file mode 100644 index 0000000..9c0b6ea --- /dev/null +++ b/plugins/factory/skills/ship/references/pull-request.md @@ -0,0 +1,75 @@ +# The Pull Request + +How phase 3 of `ship` turns a hardened, reviewed branch into an open pull request that carries its own proof. +A PR without evidence asks the reviewer to trust the description; this phase makes the change visibly work before anyone reads the diff. + +## When it runs + +Phase 3 runs in the default flow, after the review report is written and the remediation loop in [remediation.md](remediation.md) has run its course: PASS and CONCERNS open a PR ready for review, a real blocker - one that survived both remediation rounds - opens a draft PR with the blockers listed first, so the work is preserved and CI runs while the recorded decision stands. +It is skipped in "gauntlet only" and "review only" runs and when the user says "no PR" or "local only"; a review-only run on someone else's PR never pushes anything. +Never force-push, never rebase, and never touch a branch other than the work branch and the evidence branch. + +## Commit and push + +1. Fixes the gauntlet left uncommitted are committed now, one commit per tool, in `commit`'s `type(scope): subject` plus What/Why shape. + Name files explicitly; never `git add -A`, never a `Co-Authored-By` line. +2. On the default branch, create the work branch first: `{plan-name}`, or `{EPIC-KEY}-{plan-name}` when [../../../references/jira.md](../../../references/jira.md) is enabled. +3. `git push -u {remote} {branch}`; a rejected push is reported and stops the phase. + +## Evidence + +The Evidence section is the PR's proof of the feature or fix, captured from a real run, never composed from memory. +`scripts/pr-evidence.py check` is the gate: loop on its output until it passes before creating or updating the PR. + +- **An e2e report exists** (`/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`, refreshed by phase 1 when fixes landed): run `pr-evidence.py extract` on it. + Frontend reports yield one screenshot per meaningful step; `pr-evidence.py publish` pushes the PNGs to the `pr-evidence` branch so they render inline, because data URIs do not render on a PR. + Non-frontend reports yield labeled Before/After state per step, plus captured output. +- **No e2e report** (no spec, no e2e suite): capture the evidence in this phase, the same way build would. + A system with a UI: launch it in the mocked environment via the `run` skill or the repo's documented command, drive the changed behavior with browser automation, save one screenshot per meaningful step under the evidence directory, then publish them. + Anything else: the observable effect before and after, as a labeled pair of fenced blocks - a CLI transcript, an API response, a table row, a rendered file. +- **A bug fix** always adds the reproducing test as a labeled pair: `**On merge base**` shows it failing in an isolated worktree at `git merge-base HEAD origin/{default}`, `**On this branch**` shows the same test passing. + Red on base is what proves the bug was real; green on the branch is what proves it is gone. +- A screenshot proves the UI; when the real effect is a data change the screen does not show, add the Before/After pair for it as well. +- Never fabricate, reuse a screenshot from another run, or paste a data URI; a scenario that failed stays marked failed in the evidence. + +Evidence files live under `/tmp/{project-slug}/pr-evidence/`; the `pr-evidence` branch is an orphan branch that only ever receives images, one commit per run. +On a non-GitHub remote pass `--url-template` with wherever the images are hosted. + +## Body + +Write the body to `.dev/{plan-name}/pr.md` and hand it to the PR tool with `--body-file`; the file stays as the record of what was proposed. + +```markdown +## Summary + +{What changed and why, 2-4 sentences from the spec's research and scope sections.} +Plan `.dev/{plan-name}` | Review {N}: {verdict} | Jira {EPIC-KEY when enabled} + +## Evidence + +{pr-evidence.py extract output, or the hand-captured section per the rules above} + +## Quality + +| Check | Found | Fixed | Surviving | +| ----- | ----- | ----- | --------- | +{one row per gauntlet tool} + +Review panel: {lenses}; {blockers} blockers, {concerns} concerns ({review_N.md path}). +CI parity: {each reproducible required command and its result}; remote-only: {checks verified by the PR itself, or "none"}. + +## Open calls + +{Each surviving recorded decision from the gauntlet, each real blocker with its two-round history, and each concern from the review, one line each, or "none".} +``` + +Title: `{EPIC-KEY} ` prefix when Jira is enabled, then the change in imperative mood, under 70 characters. + +## Create or update + +- No PR for the branch: `gh pr create --title ... --body-file .dev/{plan-name}/pr.md` (`--draft` on BLOCK), base = the default branch. +- A PR already exists (build opened it on request, or a re-review): `gh pr edit --body-file ...`; `gh pr ready` when a BLOCK verdict has cleared, never the reverse. +- No `gh` and no equivalent host CLI: push, keep `pr.md`, print the compare URL, and say the PR must be opened by hand. + +Then follow the PR's required checks to a terminal state per [../../../references/ci-parity.md](../../../references/ci-parity.md). +Opening the PR is not the end of the phase; every required check green, or a recorded decision naming why one cannot be, is. diff --git a/plugins/factory/skills/ship/references/remediation.md b/plugins/factory/skills/ship/references/remediation.md new file mode 100644 index 0000000..36f8c47 --- /dev/null +++ b/plugins/factory/skills/ship/references/remediation.md @@ -0,0 +1,42 @@ +# Blocker Remediation + +How `ship` handles a BLOCK verdict from the review panel: two autonomous rounds of fix and re-review before a blocker is real. +A confirmed blocker is a defect with a triggering scenario, and a defect with a triggering scenario is work, not a question - the same reasoning that lets the gauntlet loop its fix agents. +Only a blocker that survives two rounds on the same root cause is recorded as a real blocker. + +## When it runs + +After phase 2 writes `review_N.md` with verdict BLOCK, in the default flow. +Never in "review only" mode: that mode exists for diffs whose owner did not ask for mutations. +CONCERNS and PASS skip straight to phase 3; concerns stay in the report and the PR's Open calls, they never trigger a round. + +## A round + +1. **Fix.** Take every Blocker from the current `review_N.md` - each carries a `file:line`, a title, and the triggering scenario verification confirmed. + Group blockers by independent area and dispatch one fresh-context fix agent per group, launched in a single message. + Each agent gets only its blockers, the relevant file paths, the brief phase 2 built (spec excerpts, contracts, decision rationale), and the fix vocabulary below. + Agents commit their fix on the work branch in `commit`'s message shape, naming the review index and finding in the Why. +2. **Re-harden.** Run phase 1's five read-only analyzers over the files the round touched, plus the flakiness check over the tests it added; dispatch fixes and loop per [gauntlet.md](gauntlet.md) as usual. + Then run the spec's Validation block (or the repo's test suite) and re-run the `[e2e]` scenarios when the round touched anything an e2e scenario exercises, overwriting the e2e report. +3. **Re-review.** Run phase 2 again as a re-review: `review_{N+1}.md`, same lenses, previous findings carried as verification items so the panel states whether each blocker is fixed or still open. + The summary line names the round: `Remediation round {r} of 2`. +4. **Decide.** PASS or CONCERNS ends the loop and phase 3 opens a ready PR. + BLOCK after round 1 starts round 2. + BLOCK after round 2 ends the loop: every surviving blocker is a real blocker. + +Round budget is two per ship run, not per blocker: a blocker the fixes introduced counts against the same budget, and a re-review that finds a new blocker after round 2 is a real blocker too. +A re-run of `ship` after a recorded real blocker starts a fresh budget. + +## Fix vocabulary + +- Fix the root cause the scenario names, never the symptom the lens observed; a fix that only makes the triggering scenario pass is a suppression. +- Add the test that reproduces the triggering scenario at its right layer, seen red before the fix and green after; a blocker with no test guarding it is not fixed. +- Stay inside the blocker's area: no refactors, no touching findings another agent owns, no concerns or nits. +- Never delete or weaken a test, a check, or a contract to make the blocker moot; never edit `docs/decisions.md` or `docs/contracts.md`. +- A blocker that cannot be fixed without changing specced behavior, a settled decision, or a contract the reliance sites still assume is marked `escalated` in the agent's result and left alone; escalations go straight to the real-blocker list without spending the second round. + +## Real blockers + +A real blocker is listed in the PR under Open calls with its history: the finding, the round-1 and round-2 attempts (commit and what each changed), and why it still stands - unfixable in scope, escalated, or a new blocker the fixes introduced. +Phase 3 opens the PR as a draft. +The wrap-up presents each real blocker as the recorded decision it is: a real defect the fixes could not reach, a decision worth reopening, or a spec the change outgrew - the same three outcomes the gauntlet's recorded decisions have. diff --git a/plugins/factory/skills/ship/references/report-format.md b/plugins/factory/skills/ship/references/report-format.md new file mode 100644 index 0000000..2f990bf --- /dev/null +++ b/plugins/factory/skills/ship/references/report-format.md @@ -0,0 +1,50 @@ +# Review Report Format + +The shape of `.dev/{plan-name}/review_N.md`, written by phase 2 at the next free index starting at 1. +`REVIEW_DATA` ([data-schema.md](data-schema.md)) mirrors this file section for section; the aggregator's JSON fills both. + +```markdown +# Review {N}: {title} + +Verdict: {PASS | CONCERNS | BLOCK} +Panel: {lenses run}, {failed lenses if any} +Base: {branch or PR}, {date} + +{1-2 sentence summary} + +## Spec conformance + +Tests: scenarios met/not met, invariants held, seams tested/untested. Omit if no spec. + +## Decision reconciliation + +{only when the repo keeps docs/decisions.md} Each colliding finding: still-holds (with the checked reopen condition), reopened, or diverged. + +## Previous findings + +{re-review only} Each finding from review_{N-1}: fixed or still open. + +## Blockers + +[{lenses}] {file:line} - {title} +{The scenario that triggers it, confirmed by verification.} + +## Concerns + +Same shape as blockers. A finding the verifier could not confirm belongs here, not in Blockers. + +## Nits + +| File | Lens | Issue | + +## What's good + +One line per lens; omit empty ones. + +## Next step + +What to fix first and why. +``` + +`scope` reads this file when the user accepts findings that need real work: its Blockers and Concerns open that run's interview, and its Decision reconciliation section is what gets applied to `docs/decisions.md`. +Write it for that reader - a finding with no triggering scenario cannot become a decision. diff --git a/plugins/factory/skills/ship/references/tools.md b/plugins/factory/skills/ship/references/tools.md new file mode 100644 index 0000000..64985db --- /dev/null +++ b/plugins/factory/skills/ship/references/tools.md @@ -0,0 +1,79 @@ +# Acquiring the Tools + +The gauntlet needs eight deterministic tools. +The acquisition ladder, per tool, is: the repo's existing tooling, else the ecosystem's established tool, else a small repo-fitted script an agent writes. +Never hand-roll what the ecosystem already maintains, and never download a generic harness wholesale - fit the tool to this repo, commit it, and reuse it on every later run. + +## Where tools live + +Repo-fitted scripts and configs go under `tools/harden/` in the consuming repository, committed with a README line saying what each does and how to invoke it. +The first run pays the acquisition cost; every later run - and any other agent - reuses them. +Check `tools/harden/` before acquiring anything. +Acquisitions are independent of each other: when several tools are missing, acquire them in parallel - one agent per tool, launched in a single message - rather than building them one after another. + +## 1. Project static analysis + +Run everything the repo already configures, not a subset you pick: linters, type checkers, format checkers, and any other analyzer wired into the project. +Discover the full set from where the repo declares it - package scripts, Makefile or task-runner targets, CI workflow steps, pre-commit hooks, and tool config files at the root (`.eslintrc*`, `pyproject.toml` tool sections, `clippy.toml`, and the like). +A configured tool that CI runs and the gauntlet skips is a check silently weakened; when unsure whether something counts, run it. +Prefer each tool's changed-files or per-path mode to scope to the diff; pre-existing findings elsewhere are noted, not fixed, unless the run is full-repo. +If the repo configures nothing, wire up the ecosystem's standard linter and type checker with their default configs as part of this step, committed like any other acquired tool. + +## 2. Security scan + +Three sub-scans, each zero-threshold: + +- **Secrets**: gitleaks or the ecosystem equivalent over the in-scope files. A found secret is never fix-agent work - stop the gauntlet and escalate immediately, because it needs rotation and possibly history rewriting, both human calls. +- **Dependency vulnerabilities**: the ecosystem's audit tool (osv-scanner, npm audit, pip-audit, cargo audit) over every manifest the diff touched. +- **Static security rules**: the repo's own SAST config if one exists, else semgrep with the ecosystem's default ruleset, scoped to in-scope files. + +## 3. Dead code detector + +The ecosystem's detector - knip or ts-prune (JS/TS), vulture (Python), staticcheck's unused checks (Go) - else a repo-fitted script that greps each added or touched export for references. +Scope: symbols the diff added that nothing references, plus symbols the diff orphaned by removing their last caller. +Declare entry points and deliberate public API surface as exclusions in committed config; a public export is not dead merely because the repo itself never calls it. + +## 4. Duplication detector + +jscpd or PMD CPD, comparing the in-scope files against the whole repo and reporting only clones that a diff-touched file participates in. +The threshold is token-based (default 50) so trivial similarity does not fire; where the line sits is a recorded decision like any other threshold. +Pre-existing clones between untouched files are noted, not fixed, unless the run is full-repo. + +## 5. Dependency checker + +Almost always a repo-fitted script: parse the `rules` block of `docs/dependencies.md` (format in [../../../references/dependency-rules.md](../../../references/dependency-rules.md)), glob-match the in-scope files to modules, extract static imports with the language's own tooling (a compiler API, an AST module, or a disciplined grep for import statements), and print each forbidden edge as `file -> file (module -> module)`. +Ecosystem tools exist for some stacks (dependency-cruiser for JS/TS, import-linter for Python, ArchUnit for JVM); prefer one when the repo already uses it or adoption is one config file that mirrors `docs/dependencies.md` - but `docs/dependencies.md` stays the single source of truth, so generate the tool's config from it rather than maintaining two rule sets. + +## 6. Coverage-weighted complexity + +The score per function combines cyclomatic complexity with test coverage so that only *uncovered* complexity fails: complexity squared, scaled down by the fraction of the function's paths the tests execute. A fully covered function scores its complexity; an uncovered one scores its complexity squared. + +- Coverage comes from the repo's existing coverage runner (jest/vitest `--coverage`, `coverage.py`, `go test -cover`, JaCoCo, tarpaulin). If the repo has none, wiring the standard one up is part of this step. +- Complexity comes from the ecosystem's standard analyzer (eslint `complexity` rule, `radon`, `gocyclo`, checkstyle) or, failing that, a small AST script. +- A repo-fitted script under `tools/harden/` joins the two reports and prints each function over threshold as `file:function score (complexity N, coverage P%)`. + +Scope the report to functions the diff touched; pre-existing offenders elsewhere are noted, not fixed, unless the run is full-repo. + +## 7. Flakiness detector + +Almost always a repo-fitted script: run the tests the diff added or touched N times (default 5), shuffling order where the runner supports it (vitest `sequence.shuffle`, pytest-randomly, `go test -shuffle`, JUnit's method orderer). +Any run disagreeing with any other marks that test flaky - a violation like a failure, because a test that passes inconsistently proves nothing. +Keep it affordable: only diff-touched tests, reuse build caches between repeats, and run repeats inside one runner invocation where the framework allows. + +## 8. Mutation testing + +Prefer the ecosystem's mutation framework - Stryker (JS/TS), mutmut or cosmic-ray (Python), PIT (JVM), cargo-mutants (Rust), go-mutesting (Go). +These handle mutant generation, test selection, and reporting far better than a hand-rolled loop; write only the thin config that scopes them. +Hand-roll only when the ecosystem has nothing: an agent-written script that applies one mutation at a time (flip `<` to `<=`, `==` to `!=`, `+` to `-`, negate conditions, drop return values), runs the narrowest relevant test command, and records survivors. + +Keeping it affordable: + +- **Scope to the diff**: mutate only in-scope files, run only the tests that cover them (most frameworks do incremental or per-file runs; use that). +- Set a per-mutant test timeout so an infinite-loop mutant cannot hang the run. +- Equivalent mutants (mutations that provably cannot change behavior) are the one legitimate survivor category: mark them as such in the report with the reasoning, don't chase them forever - two fix rounds, then escalate per the loop rule. + +## Fix-agent prompts + +Keep them minimal - the violation list, the file paths, the fix vocabulary, nothing else. +The agents inherit no conversation and need none: a surviving mutant at `cart.ts:41 (< flipped to <=)` plus "write the test that kills it, at the layer the surrounding tests use" is a complete task. +Do not paste tool source, spec prose, or this reference into fix prompts; long prompts are how rules soften. diff --git a/plugins/factory/skills/ship/scripts/aggregate-findings.py b/plugins/factory/skills/ship/scripts/aggregate-findings.py new file mode 100755 index 0000000..81d86d9 --- /dev/null +++ b/plugins/factory/skills/ship/scripts/aggregate-findings.py @@ -0,0 +1,185 @@ +#!/usr/bin/env python3 +"""Number the panel's findings for verification, then apply the verdicts. + + aggregate-findings.py plan + aggregate-findings.py aggregate + [--expected lens,lens] + +`plan` prints the BLOCK and CONCERN findings as one JSON array with stable ids +(f1, f2, ...) to hand to the verifiers. `aggregate` applies their verdicts and +prints the report data: refuted findings dropped, an unconfirmed BLOCK demoted +to CONCERN, duplicates merged by file:line, and the overall verdict. +""" + +from __future__ import annotations + +import json +import re +import sys +from pathlib import Path + +FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S) +RESULT_SUFFIXES = (".json", ".out") + + +def load_json(path: Path): + """Result files are JSON, or a final message wrapping one fenced block.""" + text = path.read_text(encoding="utf-8", errors="replace").strip() + for candidate in [text, *FENCE.findall(text)]: + candidate = candidate.strip() + if not candidate: + continue + try: + return json.loads(candidate) + except json.JSONDecodeError: + continue + return None + + +def read_results(directory: Path) -> tuple[dict[str, object], list[str]]: + """Every parseable result file by agent name, plus the names that failed.""" + parsed: dict[str, object] = {} + unreadable: list[str] = [] + for path in sorted(directory.iterdir()): + if not path.is_file() or path.suffix not in RESULT_SUFFIXES: + continue + payload = load_json(path) + if payload is None: + unreadable.append(path.stem) + else: + parsed[path.stem] = payload + return parsed, unreadable + + +def collect_findings(lens_results: dict[str, object]) -> list[dict]: + findings = [] + for lens, payload in lens_results.items(): + if not isinstance(payload, dict): + continue + for raw in payload.get("findings", []) or []: + finding = dict(raw) + finding["lens"] = lens + findings.append(finding) + return findings + + +def numbered(findings: list[dict]) -> list[dict]: + """BLOCK and CONCERN findings get an id; NITs skip verification.""" + out = [] + for finding in findings: + if str(finding.get("severity", "")).upper() not in ("BLOCK", "CONCERN"): + continue + entry = dict(finding) + entry["id"] = f"f{len(out) + 1}" + out.append(entry) + return out + + +def merge(findings: list[dict]) -> list[dict]: + """One entry per file:line; keep the fullest detail, credit every lens.""" + merged: dict[tuple, dict] = {} + for finding in findings: + key = (finding.get("file"), finding.get("line"), ) + existing = merged.get(key) + if existing is None: + entry = dict(finding) + entry["lenses"] = [finding.get("lens")] if finding.get("lens") else [] + entry.pop("lens", None) + merged[key] = entry + continue + if finding.get("lens") and finding["lens"] not in existing["lenses"]: + existing["lenses"].append(finding["lens"]) + if len(str(finding.get("detail", ""))) > len(str(existing.get("detail", ""))): + existing["detail"] = finding["detail"] + existing["title"] = finding.get("title", existing.get("title")) + return list(merged.values()) + + +def cmd_plan(lens_dir: Path) -> int: + lens_results, unreadable = read_results(lens_dir) + findings = numbered(collect_findings(lens_results)) + print(json.dumps(findings, indent=2, ensure_ascii=False)) + print( + f"\n{len(findings)} finding(s) to verify from {len(lens_results)} lens result(s)" + + (f"; unparseable: {', '.join(unreadable)}" if unreadable else ""), + file=sys.stderr, + ) + return 0 + + +def cmd_aggregate(lens_dir: Path, verifier_dir: Path, expected: list[str]) -> int: + lens_results, unreadable = read_results(lens_dir) + all_findings = collect_findings(lens_results) + to_verify = numbered(all_findings) + + verdicts: dict[str, dict] = {} + verifier_results, verifier_unreadable = read_results(verifier_dir) + for payload in verifier_results.values(): + for entry in payload if isinstance(payload, list) else [payload]: + if isinstance(entry, dict) and entry.get("id"): + verdicts[entry["id"]] = entry + + blockers, concerns = [], [] + for finding in to_verify: + verdict = verdicts.get(finding["id"], {}) + status = str(verdict.get("status", "")).upper() + if status == "REFUTED": + continue + finding = dict(finding) + finding["verification"] = status or "UNVERIFIED" + finding["reportedSeverity"] = str(finding.get("severity", "")).upper() + if finding["reportedSeverity"] == "BLOCK" and status == "CONFIRMED": + blockers.append(finding) + else: + # An unconfirmed BLOCK carries on as a CONCERN, not as a BLOCK. + finding["severity"] = "CONCERN" + concerns.append(finding) + + nits = [f for f in all_findings if str(f.get("severity", "")).upper() == "NIT"] + failed = sorted(set(expected) - set(lens_results)) + unreadable + + report = { + "verdict": "BLOCK" if blockers else "CONCERNS" if concerns else "PASS", + "panel": sorted(lens_results), + "failedLenses": failed, + "blockers": merge(blockers), + "concerns": merge(concerns), + "nits": merge(nits), + "good": {lens: (p.get("good", []) if isinstance(p, dict) else []) + for lens, p in sorted(lens_results.items())}, + } + print(json.dumps(report, indent=2, ensure_ascii=False)) + + unverified = [f["id"] for f in to_verify if f["id"] not in verdicts] + notes = [] + if failed: + notes.append(f"failed lenses: {', '.join(failed)}") + if unverified: + notes.append(f"unverified findings kept as concerns: {', '.join(unverified)}") + if verifier_unreadable: + notes.append(f"unparseable verifier files: {', '.join(verifier_unreadable)}") + if notes: + print("\n" + "; ".join(notes), file=sys.stderr) + return 0 + + +def main(argv: list[str]) -> int: + args, expected = [], [] + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--expected" and rest: + expected = [n.strip() for n in rest.pop(0).split(",") if n.strip()] + else: + args.append(value) + + if len(args) == 2 and args[0] == "plan": + return cmd_plan(Path(args[1])) + if len(args) == 3 and args[0] == "aggregate": + return cmd_aggregate(Path(args[1]), Path(args[2]), expected) + print(__doc__, file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/plugins/factory/skills/ship/scripts/pr-evidence.py b/plugins/factory/skills/ship/scripts/pr-evidence.py new file mode 100755 index 0000000..63a448f --- /dev/null +++ b/plugins/factory/skills/ship/scripts/pr-evidence.py @@ -0,0 +1,311 @@ +#!/usr/bin/env python3 +"""Turn e2e evidence into a PR-ready Evidence section, publish its images, and gate the body. + + pr-evidence.py extract --out [--branch pr-evidence] + [--url-template URL] + pr-evidence.py publish [--branch pr-evidence] [--remote origin] + pr-evidence.py check [--kind frontend|non-frontend] + +`extract` reads the E2E_DATA block of a rendered e2e report, decodes every +screenshot to `/{plan}/{scenario}/{step}.png`, and writes +`/evidence.md`: one "## Evidence" section with the images (frontend) or +before/after state tables (non-frontend) a reviewer can read inline on the PR. +Image links use --url-template, where `{path}` is the file's path under ; +on a GitHub remote the template is derived from --branch when omitted. + +`publish` commits everything under onto the assets branch and pushes it, +using a temporary index and plumbing commands only: the working tree, the +current branch, and the index are never touched. Data URIs do not render on a +PR, so this is how screenshots become visible. + +`check` exits non-zero unless the body has a non-empty "## Evidence" section +holding real proof: an image served over http(s), or a labeled before/after +(or merge-base/branch) pair of fenced blocks. Placeholders fail it. +""" + +from __future__ import annotations + +import argparse +import base64 +import json +import os +import re +import subprocess +import sys +import tempfile +from datetime import datetime, timezone +from pathlib import Path + +DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.S) +IDENT = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*") +PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.I) +IMAGE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)") +PAIR_LABELS = { + "before": "before", "after": "after", + "on merge base": "base", "on the merge base": "base", "merge base": "base", + "on this branch": "branch", "on the branch": "branch", "this branch": "branch", +} +LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.I) +FENCE_OPEN = re.compile(r"^\s*```") +IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".webp"} + + +def fail(message: str) -> None: + print(f"pr-evidence: {message}", file=sys.stderr) + sys.exit(1) + + +def slug(text: str, fallback: str) -> str: + cleaned = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-") + return cleaned[:60] or fallback + + +def load_e2e_data(report: Path) -> dict: + """The data block is a JS object literal; accept JSON, then a lightly normalized form.""" + match = DATA_BLOCK.search(report.read_text(encoding="utf-8", errors="replace")) + if not match: + fail(f"no E2E_DATA block between the markers in {report}") + literal = match.group(1) + for candidate in (literal, node_json(literal), js_to_json(literal)): + if candidate is None: + continue + try: + return json.loads(candidate) + except json.JSONDecodeError: + continue + fail("E2E_DATA is not parseable as JSON, through node, or as a plain JS object literal") + return {} + + +def js_to_json(literal: str) -> str: + """Rewrite a plain JS object literal as JSON without touching string contents. + + Handles bare keys, single-quoted strings, trailing commas, and comments; + string bodies (where prose can contain colons and commas) pass through as-is. + """ + out: list[str] = [] + i, n = 0, len(literal) + while i < n: + ch = literal[i] + if ch in "\"'": + quote, j, body = ch, i + 1, [] + while j < n and literal[j] != quote: + if literal[j] == "\\" and j + 1 < n: + body.append(literal[j:j + 2]) + j += 2 + else: + body.append(literal[j]) + j += 1 + text = "".join(body) + if quote == "'": + text = text.replace("\\'", "'").replace('"', '\\"') + out.append(f'"{text}"') + i = j + 1 + elif literal.startswith("//", i): + i = literal.find("\n", i) if literal.find("\n", i) != -1 else n + elif literal.startswith("/*", i): + end = literal.find("*/", i + 2) + i = n if end == -1 else end + 2 + elif ch == ",": + j = i + 1 + while j < n and literal[j] in " \t\r\n": + j += 1 + if j < n and literal[j] in "}]": + i += 1 + else: + out.append(ch) + i += 1 + else: + ident = IDENT.match(literal, i) + if ident: + j = ident.end() + while j < n and literal[j] in " \t\r\n": + j += 1 + if j < n and literal[j] == ":": + out.append(f'"{ident.group(0)}"') + else: + out.append(ident.group(0)) + i = ident.end() + else: + out.append(ch) + i += 1 + return "".join(out) + + +def node_json(literal: str) -> str | None: + """A JS object literal is exactly what node parses; use it when installed.""" + with tempfile.NamedTemporaryFile("w", suffix=".js", delete=False, encoding="utf-8") as handle: + handle.write(literal) + script = f"process.stdout.write(JSON.stringify(eval('(' + require('fs').readFileSync({handle.name!r}, 'utf8') + ')')))" + try: + result = subprocess.run(["node", "-e", script], capture_output=True, text=True) + return result.stdout if result.returncode == 0 else None + except OSError: + return None + finally: + os.unlink(handle.name) + + +def git(*args: str, env: dict | None = None, check: bool = True) -> str: + result = subprocess.run(["git", *args], capture_output=True, text=True, env=env) + if check and result.returncode != 0: + fail(f"git {' '.join(args)} failed: {result.stderr.strip()}") + return result.stdout.strip() + + +def github_slug(remote: str) -> str | None: + url = git("remote", "get-url", remote, check=False) + match = re.search(r"github\.com[:/]([^/]+)/([^/\s]+?)(?:\.git)?$", url) + return f"{match.group(1)}/{match.group(2)}" if match else None + + +def url_template(args: argparse.Namespace) -> str | None: + if args.url_template: + if "{path}" not in args.url_template: + fail("--url-template must contain {path}") + return args.url_template + repo = github_slug(args.remote) + if repo: + return f"https://github.com/{repo}/blob/{args.branch}/{{path}}?raw=true" + return None + + +def fenced(value: object, language: str = "json") -> str: + text = value if isinstance(value, str) else json.dumps(value, indent=2, ensure_ascii=False) + return f"```{language}\n{text.rstrip()}\n```" + + +def extract(args: argparse.Namespace) -> None: + data = load_e2e_data(Path(args.report)) + kind = data.get("kind", "non-frontend") + plan = data.get("planName") or slug(data.get("title", ""), "plan") + scenarios = data.get("scenarios", []) or [] + summary = data.get("summary") or {} + out = Path(args.out) + template = url_template(args) if kind == "frontend" else None + if kind == "frontend" and template is None: + fail("frontend report but no --url-template and the remote is not GitHub; pass --url-template") + + lines = ["## Evidence", "", + f"Captured from the e2e run of `{plan}` at {data.get('generatedAt', 'unknown time')}: " + f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] + written: list[str] = [] + for index, scenario in enumerate(scenarios, 1): + scenario_id = slug(str(scenario.get("id") or scenario.get("title") or index), f"scenario-{index}") + lines += [f"### {scenario.get('title', scenario_id)} - {scenario.get('status', 'unknown')}", "", + f"Given {scenario.get('given', '?')}, when {scenario.get('when', '?')}, then {scenario.get('then', '?')}.", ""] + for step_index, shot in enumerate(scenario.get("screenshots", []) or [], 1): + uri = shot.get("dataUri", "") + if not uri.startswith("data:image/"): + fail(f"{scenario_id} step {step_index} has no data URI screenshot") + ext = "." + uri[len("data:image/"):uri.index(";")].replace("jpeg", "jpg") + relative = Path(plan) / scenario_id / f"{step_index:02d}-{slug(shot.get('step', ''), 'step')}{ext}" + target = out / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(base64.b64decode(uri.split(",", 1)[1])) + written.append(relative.as_posix()) + url = template.replace("{path}", relative.as_posix()) + lines += [f"**{shot.get('step', 'step')}** - {shot.get('caption', '')}", "", + f"![{shot.get('caption') or shot.get('step', 'screenshot')}]({url})", ""] + for state in scenario.get("dataModelState", []) or []: + lines += [f"**{state.get('step', 'step')}** - {state.get('caption', '')} (`{state.get('entity', '?')}`)", "", + "**Before**", "", fenced(state.get("before", "")), "", + "**After**", "", fenced(state.get("after", "")), ""] + output = scenario.get("logsOrOutput") + if output: + lines += ["
Captured output", "", fenced(output, ""), "", "
", ""] + out.mkdir(parents=True, exist_ok=True) + (out / "evidence.md").write_text("\n".join(lines).rstrip() + "\n", encoding="utf-8") + print(json.dumps({"kind": kind, "plan": plan, "scenarios": len(scenarios), "images": written, + "evidence": str(out / "evidence.md"), "urlTemplate": template}, indent=2)) + + +def publish(args: argparse.Namespace) -> None: + root = Path(args.dir).resolve() + files = sorted(p for p in root.rglob("*") if p.is_file() and p.suffix.lower() in IMAGE_SUFFIXES) + if not files: + fail(f"nothing to publish: no image files under {root}") + ref = f"refs/remotes/{args.remote}/{args.branch}" + subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True) + parent = git("rev-parse", "--verify", "--quiet", ref, check=False) or None + with tempfile.TemporaryDirectory() as tmp: + env = {**os.environ, "GIT_INDEX_FILE": str(Path(tmp) / "index")} + if parent: + git("read-tree", parent, env=env) + for path in files: + blob = git("hash-object", "-w", str(path)) + git("update-index", "--add", "--cacheinfo", f"100644,{blob},{path.relative_to(root).as_posix()}", env=env) + tree = git("write-tree", env=env) + message = f"evidence: {len(files)} screenshots captured {datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')}" + commit = git("commit-tree", tree, *(["-p", parent] if parent else []), "-m", message) + git("push", args.remote, f"{commit}:refs/heads/{args.branch}") + print(json.dumps({"branch": args.branch, "commit": commit, "files": [p.relative_to(root).as_posix() for p in files]}, indent=2)) + + +def evidence_section(body: str) -> str | None: + match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.M | re.S) + return match.group(1) if match else None + + +def labeled_pairs(section: str) -> set[str]: + """Labels (before/after/base/branch) that head a fenced block within the next two lines.""" + found: set[str] = set() + lines = section.splitlines() + for i, line in enumerate(lines): + match = LABEL_LINE.match(line) + if not match: + continue + label = PAIR_LABELS.get(match.group(1).strip().lower()) + if label and any(FENCE_OPEN.match(nxt) for nxt in lines[i + 1:i + 3]): + found.add(label) + return found + + +def check(args: argparse.Namespace) -> None: + section = evidence_section(Path(args.body).read_text(encoding="utf-8")) + problems: list[str] = [] + if section is None or not section.strip(): + problems.append("no non-empty '## Evidence' section") + section = "" + placeholder = PLACEHOLDERS.search(section) + if placeholder: + problems.append(f"placeholder or inline data URI in Evidence: {placeholder.group(0)!r}") + images = IMAGE.findall(section) + labels = labeled_pairs(section) + pair = {"before", "after"} <= labels or {"base", "branch"} <= labels + if args.kind == "frontend" and not images: + problems.append("frontend change but no http(s) image in Evidence") + if args.kind == "non-frontend" and not pair: + problems.append("non-frontend change but no labeled Before/After or merge-base/branch pair in Evidence") + if not images and not pair: + problems.append("Evidence holds neither an http(s) image nor a labeled before/after pair") + if problems: + fail("PR body rejected:\n - " + "\n - ".join(problems)) + print(f"pr-evidence: ok ({len(images)} images, pairs: {', '.join(sorted(labels)) or 'none'})") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + sub = parser.add_subparsers(dest="command", required=True) + ex = sub.add_parser("extract") + ex.add_argument("report") + ex.add_argument("--out", required=True) + ex.add_argument("--branch", default="pr-evidence") + ex.add_argument("--remote", default="origin") + ex.add_argument("--url-template") + ex.set_defaults(run=extract) + pub = sub.add_parser("publish") + pub.add_argument("dir") + pub.add_argument("--branch", default="pr-evidence") + pub.add_argument("--remote", default="origin") + pub.set_defaults(run=publish) + ck = sub.add_parser("check") + ck.add_argument("body") + ck.add_argument("--kind", choices=["frontend", "non-frontend"]) + ck.set_defaults(run=check) + args = parser.parse_args() + args.run(args) + + +if __name__ == "__main__": + main() diff --git a/plugins/factory/skills/ship/scripts/run-pi-agents.sh b/plugins/factory/skills/ship/scripts/run-pi-agents.sh new file mode 100755 index 0000000..6abfd4d --- /dev/null +++ b/plugins/factory/skills/ship/scripts/run-pi-agents.sh @@ -0,0 +1,69 @@ +#!/usr/bin/env bash +set -uo pipefail + +if [ "$#" -ne 2 ]; then + echo "Usage: run-pi-agents.sh " >&2 + exit 2 +fi + +prompt_dir="$1" +output_dir="$2" +pi_bin="${PI_BIN:-pi}" + +if [ ! -d "$prompt_dir" ]; then + echo "Prompt directory not found: $prompt_dir" >&2 + exit 2 +fi +if ! command -v "$pi_bin" >/dev/null 2>&1; then + echo "Pi executable not found: $pi_bin" >&2 + exit 2 +fi + +mkdir -p "$output_dir" +prompts=("$prompt_dir"/*.md) +if [ ! -e "${prompts[0]}" ]; then + echo "No .md prompts found in: $prompt_dir" >&2 + exit 2 +fi + +args=( + --no-session + --no-skills + --no-extensions + --no-prompt-templates + --tools read,bash,grep,find,ls + --print +) +if [ -n "${PI_PROVIDER:-}" ]; then + args+=(--provider "$PI_PROVIDER") +fi +if [ -n "${PI_MODEL:-}" ]; then + args+=(--model "$PI_MODEL") +fi +if [ -n "${PI_REASONING_LEVEL:-}" ]; then + args+=(--thinking "$PI_REASONING_LEVEL") +fi + +pids=() +names=() + +for prompt in "${prompts[@]}"; do + name=$(basename "$prompt" .md) + output="$output_dir/$name.out" + error="$output_dir/$name.err" + "$pi_bin" "${args[@]}" < "$prompt" > "$output" 2> "$error" & + pids+=("$!") + names+=("$name") +done + +status=0 +for index in "${!pids[@]}"; do + if wait "${pids[$index]}"; then + rm -f "$output_dir/${names[$index]}.err" + else + echo "Pi agent failed: ${names[$index]}" >&2 + status=1 + fi +done + +exit "$status" diff --git a/plugins/factory/skills/ship/templates/review.html b/plugins/factory/skills/ship/templates/review.html new file mode 100644 index 0000000..31eb8d9 --- /dev/null +++ b/plugins/factory/skills/ship/templates/review.html @@ -0,0 +1,214 @@ + + + + + +Review + + + +
+ REVIEW + +
+ +
+
+

+ +
+
+
+ +
+

Spec conformance

+
+
+ +
+

Previous findings

+
+
+ +
+

Blockers

+
+
+ +
+

Concerns

+
+
+ +
+

Nits

+
FileLensIssue
+
+ +
+

What's good

+
+
+ +
+

Next step

+
+
+
+ + + + diff --git a/scripts/build_codex_plugin.py b/scripts/build_codex_plugin.py index 1735882..8a0cb0a 100644 --- a/scripts/build_codex_plugin.py +++ b/scripts/build_codex_plugin.py @@ -83,6 +83,42 @@ def destination(self) -> Path: ), }, ), + "factory": PluginConfig( + name="factory", + display_name="Factory", + short_description="Run scope, scope-review, build, and ship unattended.", + capabilities=("Interactive", "Write"), + default_prompts=( + "Take this request through scope, then run it unattended to a pull request.", + ), + skill_ui={ + "run": ( + "Run", + "Take a request from scope to a shipped change unattended", + "Use $factory:run to take a request through scope, then run scope-review, build, and ship unattended.", + ), + "scope": ( + "Scope", + "Spec a change by arguing its decisions", + "Use $factory:scope to spec this change with argued decisions and a change plan.", + ), + "scope-review": ( + "Scope Review", + "Review and auto-refine a settled spec", + "Use $factory:scope-review to review the settled spec with a verified agent panel and refine it in place before building.", + ), + "build": ( + "Build", + "Execute a spec test-first through e2e", + "Use $factory:build to execute the current spec test-first and verify it end to end.", + ), + "ship": ( + "Ship", + "Harden, review, then open a PR with proof", + "Use $factory:ship to run the quality gauntlet, the verified review, and open the pull request with evidence for this change.", + ), + }, + ), } From 6929e393dd936989500bae96ba7644b936f7583c Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 11:14:57 +0200 Subject: [PATCH 02/30] feat(factory): protocol reference and shared copies What: add factory/references/factory-run.md (run state schema, result envelope with skill_path, unattended policy, launch prompt shape), copy the eight dev/references/*.md files with plan-layout.md extended for the new plan files, and copy skill-metrics.py/architecture-check.py into factory/scripts/. Why: every phase copy and the run skill read this shared material rather than duplicating it. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- .../evals/tests/test_protocol_and_copies.py | 64 ++++ factory/references/architecture.md | 92 +++++ factory/references/ci-parity.md | 52 +++ factory/references/contracts.md | 67 ++++ factory/references/decision-ledger.md | 120 +++++++ factory/references/dependency-rules.md | 47 +++ factory/references/factory-run.md | 59 ++++ factory/references/jira.md | 115 ++++++ factory/references/plan-layout.md | 34 ++ factory/references/reporting.md | 23 ++ factory/scripts/architecture-check.py | 135 +++++++ factory/scripts/skill-metrics.py | 333 ++++++++++++++++++ 12 files changed, 1141 insertions(+) create mode 100644 factory/evals/tests/test_protocol_and_copies.py create mode 100644 factory/references/architecture.md create mode 100644 factory/references/ci-parity.md create mode 100644 factory/references/contracts.md create mode 100644 factory/references/decision-ledger.md create mode 100644 factory/references/dependency-rules.md create mode 100644 factory/references/factory-run.md create mode 100644 factory/references/jira.md create mode 100644 factory/references/plan-layout.md create mode 100644 factory/references/reporting.md create mode 100755 factory/scripts/architecture-check.py create mode 100755 factory/scripts/skill-metrics.py diff --git a/factory/evals/tests/test_protocol_and_copies.py b/factory/evals/tests/test_protocol_and_copies.py new file mode 100644 index 0000000..f9d89a1 --- /dev/null +++ b/factory/evals/tests/test_protocol_and_copies.py @@ -0,0 +1,64 @@ +"""Unit tests for change set 2: the copied references, scripts, and their +backticked cross-references. Read-only against the real repo tree - safe for +check_factory_script's own discovery since nothing here invokes validate.sh. +""" + +from __future__ import annotations + +import re +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +FACTORY = REPO_ROOT / "factory" + +SKILL_ROOTS = { + "{scope-skill-root}": FACTORY / "skills" / "scope", + "{scope-review-skill-root}": FACTORY / "skills" / "scope-review", + "{build-skill-root}": FACTORY / "skills" / "build", + "{ship-skill-root}": FACTORY / "skills" / "ship", + "{run-skill-root}": FACTORY / "skills" / "run", +} + +BACKTICK = re.compile(r"`([^`\n]+)`") +# Only a {skill-root}-style cross-reference or an explicit relative path +# (../ or ./) is a repo-relative path claim; a bare "spec.md" or +# "docs/decisions.md" names a plan or consuming-repo file, not a path here. +SKILL_ROOT_PATH = re.compile(r"^\{[a-z-]+-skill-root\}/[\w./-]+\.(py|md|sh)$") +RELATIVE_PATH = re.compile(r"^\.\.?/[\w./-]+\.(py|md|sh)$") + + +def resolve(path_dir: Path, raw: str) -> Path | None: + for placeholder, root in SKILL_ROOTS.items(): + if raw.startswith(placeholder): + return (root / raw[len(placeholder):].lstrip("/")).resolve() + return (path_dir / raw).resolve() + + +class ProtocolAndCopiesTest(unittest.TestCase): + def test_copied_tree_holds_dev_references_and_factory_run(self) -> None: + dev_reference_names = {p.name for p in (REPO_ROOT / "dev" / "references").glob("*.md")} + factory_reference_names = {p.name for p in (FACTORY / "references").glob("*.md")} + self.assertTrue(dev_reference_names.issubset(factory_reference_names)) + self.assertIn("factory-run.md", factory_reference_names) + + factory_script_names = {p.name for p in (FACTORY / "scripts").glob("*.py")} + self.assertEqual(factory_script_names, {"skill-metrics.py", "architecture-check.py"}) + + def test_backticked_reference_and_script_paths_resolve_under_factory(self) -> None: + problems = [] + for md_path in sorted(FACTORY.rglob("*.md")): + if "templates" in md_path.parts: + continue + text = md_path.read_text(encoding="utf-8") + for raw in BACKTICK.findall(text): + if not (SKILL_ROOT_PATH.match(raw) or RELATIVE_PATH.match(raw)): + continue + resolved = resolve(md_path.parent, raw) + if not resolved.is_file(): + problems.append(f"{md_path}: {raw!r} -> {resolved} does not exist") + self.assertEqual(problems, [], "\n".join(problems)) + + +if __name__ == "__main__": + unittest.main() diff --git a/factory/references/architecture.md b/factory/references/architecture.md new file mode 100644 index 0000000..e1f6fb6 --- /dev/null +++ b/factory/references/architecture.md @@ -0,0 +1,92 @@ +# Architecture Overview + +`docs/architecture.md` in the consuming project: one repo-tracked file describing the system at high level - what it is for, which components make it up, how they talk, and where its boundaries with the outside world lie. +It is the page a new engineer reads on day one and an agent reads before touching anything, and it exists because every run otherwise rebuilds that picture from the file tree, gets it subtly wrong, and never writes it down. + +It is not a registry and not a ledger: the decisions, contracts, and dependency rules live in their own files with their own notation and consumers. +This file is plain prose and tables at the altitude of the whole system, and nothing in it is machine-enforced beyond staying current with the code it describes. + +## Notation + +``` +# Architecture + +Purpose: one paragraph - what the system does, for whom, and the shape of a request through it. +Captured: 2026-09-15 (full, ship) - Updated: 2026-09-18 (commit 4f2a1c9) + +## Components + +| Component | Responsibility | Lives at | Talks to | +| --------- | -------------- | -------- | -------- | +| api | HTTP surface, auth, request validation | `src/api/` | app | +| app | use cases, orchestration, transactions | `src/app/` | domain, infra | +| domain | pure business rules and entities | `src/domain/` | nothing | +| infra | Postgres, payment-svc client, mail | `src/infra/` | domain (types only) | +| worker | queue consumers for async jobs | `src/worker/` | app | + +## Flows + +### Checkout +1. `POST /checkout` (api) validates the cart and calls `placeOrder` (app). +2. app reserves stock in domain, writes the order via infra, enqueues `charge-order`. +3. worker charges through payment-svc and emails the receipt. + +## Boundaries + +| Boundary | Kind | Owned by | Notes | +| -------- | ---- | -------- | ----- | +| Postgres | store | infra | single database, migrations in `db/migrations/` | +| payment-svc | external HTTP | infra | webhooks arrive at `POST /webhooks/payment` (api) | +| Redis queue | queue | worker | at-least-once delivery, consumers are idempotent | + +## Cross-cutting + +Auth: JWT checked in api middleware, `requireUser`; config: `src/config.ts` from env only; observability: structured logs, OpenTelemetry traces from api and worker. + +## Entry points + +`src/api/main.ts` (HTTP server), `src/worker/main.ts` (queue consumer), `scripts/migrate.ts`. +``` + +Five sections in this order - Components, Flows, Boundaries, Cross-cutting, Entry points - under a Purpose paragraph and a dated `Captured` / `Updated` line. +Every component names the path it lives at, in backticks - that is what lets the checker tell a stale overview from a current one. +Flows are the three to six journeys that explain why the components exist, as numbered steps naming the component at each hop; not every endpoint, not every function. +Boundaries are everything the system does not own: stores, queues, external services, files, clocks worth naming. + +## What belongs here + +The level a new engineer needs on day one and a reviewer needs before reading a diff: components and their responsibilities, the main flows, the boundaries, the cross-cutting mechanisms, the entry points. +Not here: function-level detail, per-endpoint lists, configuration values, decisions and their alternatives, boundary guarantees, allowed dependency edges - those have their own files. +Keep it under roughly 150 lines; an overview that needs more is describing more than one system, or describing it too closely. + +## The checker + +`architecture-check.py` (in this toolkit's shared `scripts/`) only checks that the overview is still true of the code, nothing more: + +```bash +python3 {skill-root}/../../scripts/architecture-check.py docs/architecture.md [--touched file ...] +``` + +- Exit 2: the file is missing - the caller runs an initial capture, never a fix agent. +- Exit 1: violations, one per line - a required section missing, a backticked path that no longer exists, or a `--touched` file no component's path covers (`uncharted`). +- Exit 0: the overview is current for what was checked. + +## Initial capture + +When a skill needs the overview and the file is absent, capture the whole system in one pass before continuing - a partial overview that only covers the current change misleads the next reader more than none. +Dispatch read-only explorers in a single message, one per top-level source area (the directories under the source root, or the packages of a monorepo), each reporting for its area: components with paths and responsibilities, what they call and what calls them, external boundaries they touch, entry points, and the cross-cutting mechanisms they participate in. +Merge their results into the notation above, resolving each flow end to end across areas, then loop the checker until it exits 0. +Write the `Captured` line with the date and the skill, tell the user the overview is new and where it is, and commit it as its own `docs(architecture): capture the system overview` commit. +Never fabricate a component or flow to make the overview look complete; an area the explorers could not characterize is listed with `unverified` in its responsibility cell, so the next reader knows to look. + +## Continuous updates + +The overview changes when the system changes, in the same commit or batch: + +- **build** updates it as part of the change set that adds, removes, renames, or moves a component, changes a flow, or adds a boundary, and runs the checker before that change set's commit. +- **commit** runs the checker over every batch (its sync-checks step 4d): a touched file that is uncharted, a stale path, or a flow or boundary the diff visibly changed is a targeted edit committed as `docs(architecture): ...`, never a rewrite. +- **scope** starts its prior-art explorers from the overview, and names the components and flows a change will add or reshape in the spec's scope section so build knows what to update. +- **ship**'s architecture lens reads it for orientation - where things live and how they talk - and treats a diff that reshapes a component or flow without updating the overview as a finding. + +Edits are minimal and factual: change the row, step, or cell that is now wrong, refresh the `Updated` line, leave the rest alone. +A section rewritten in a run that touched one component is a sign the run drifted into documentation work it was not asked for. diff --git a/factory/references/ci-parity.md b/factory/references/ci-parity.md new file mode 100644 index 0000000..382ac06 --- /dev/null +++ b/factory/references/ci-parity.md @@ -0,0 +1,52 @@ +# CI Parity and PR Follow-through + +The local quality loop is not complete while a required pull-request check is +known red. Spec validation and feature E2E prove the promised change; CI parity +proves the repository's merge gate against the exact checkout being proposed. + +## Discover the merge gate + +Read the repository's pull-request workflows under `.github/workflows/` and +any scripts they call. Build a list of project-owned commands from required +jobs: tests, lint/type/build, generated-artifact checks, screenshot/report +suites, packaging, and repository validation. + +Do not attempt to reproduce GitHub-owned setup actions locally. Reproduce the +project command after performing its documented local setup. Prefer a +repository-provided aggregate target when it covers the same jobs. + +Record the commands and outcomes in the plan's `implementation-notes.md`. + +## Run before proposing a PR + +Run every reproducible project-owned required-check command against the final +checkout, after feature E2E and after any hardening edits. + +- A failure is work to fix, including a test described as flaky, unrelated, or + pre-existing. Diagnose and remove its nondeterminism; do not add retries, + sleeps, or looser assertions. +- "Pre-existing" is not a waiver. Prove it by running the same command on the + merge base in an isolated worktree. If the base also fails and fixing it is + materially outside scope, present the evidence as a human call. Do not call + the branch PR-ready while the required check remains red. +- A command that cannot run locally because it needs GitHub-only credentials or + infrastructure is marked `remote-only`, with the reason. It is verified by + PR follow-through rather than silently skipped. + +Only offer or create the PR once all reproducible required checks are green and +all remote-only checks are identified. + +## Follow the PR to green + +After the user authorizes a push/PR and the PR exists: + +1. Watch required checks to a terminal state. +2. For each failure, fetch the failing job log, reproduce its project command + locally when possible, fix the root cause, run the narrow regression and the + full CI-parity command, commit, and push. +3. Repeat until every required check is green or a genuine human call is + reached. + +Do not stop at "CI restarted" when the user asked to finish or ship the change. +Do not rerun a failed job unchanged unless the log proves an external service +failure; deterministic failures require a code or test fix. diff --git a/factory/references/contracts.md b/factory/references/contracts.md new file mode 100644 index 0000000..8e5c3c2 --- /dev/null +++ b/factory/references/contracts.md @@ -0,0 +1,67 @@ +# Contract Registry + +`docs/contracts.md` in the consuming project: one repo-tracked file recording facts about behavior at boundaries - what one side guarantees, what the other side relies on, what can never happen. +It exists because reviewers and fixers invent boundary premises when none are recorded ("someone retries upstream", "this endpoint might return null"), and a wrong premise produces a wrong finding. + +Contracts are not decisions, and the difference is when consumers may read them. +A decision is a *justification* - it pre-forgives findings, so `docs/decisions.md` enters a run only after judgment (see the recommender contract in [decision-ledger.md](decision-ledger.md)). +A contract is a *premise* - a fact about what the code does at a boundary, the same class of context as a code map's intent and structure - so contracts enter *before* the walk. +That timing difference is why the two live in separate files: a walker retrieving contracts must not pull in recorded conclusions. + +## Notation + +``` +# Contracts + +## Payments + +C-webhook-retry: payment-svc delivers each webhook once; it never retries - callers own retry + guaranteed by: deliverWebhook (single attempt, no loop) + relied on by: billing worker, reconciliation job + (2026-08-21, webhook-delivery/spec.md) + +C-order-status: GET /orders never returns a status outside {pending, paid, failed} + guaranteed by: OrderStatus enum at serializeOrder ? verify: migration path for legacy rows + relied on by: mobile order screen + (2026-08-21, commit a1b2c3d) +``` + +Each entry: a `C-` kebab slug, one sentence stating the guarantee or invariant, a `guaranteed by:` line naming the enforcing code, a `relied on by:` line naming the sites that assume it, and a date + source. +When a recorded decision created the guarantee, the decision's slug is the source - `(2026-08-21, D-retry-ownership)` - which makes breaking the contract traceable to the choice behind it; a contract with no parent decision cites its spec, audit, or commit as usual. +Evidence marks and their maintenance follow [decision-ledger.md](decision-ledger.md) Notation exactly. + +Entries are filed under the area that owns the *guarantee* side - that's where the enforcing code lives, so that's where a change breaks it. +The `relied on by:` line makes the entry findable from the other side; a consumer searching by either module's name retrieves it. +One global file, one entry per contract: splitting per module would force either an arbitrary owner or a duplicate that drifts. + +## What qualifies + +A fact qualifies when it crosses a boundary and someone could plausibly assume it differently: retry and idempotency ownership, value domains and nullability of responses, delivery and ordering semantics, which side validates, which errors can actually surface. +Facts the type system already states at the boundary don't qualify - the compiler is their registry. +Module-internal behavior doesn't qualify - it has no second side to mislead. + +## How consumers use contracts + +- **Premise, pre-walk.** + Reviewers, verifiers, and fixers working near a recorded boundary read the matching entries before forming claims about the other side. + A claim that contradicts a cited contract needs to explain why the contract is wrong, not assume it away. +- **Refutation strength follows the evidence mark.** + A contract whose guarantee carries a citation can refute a finding outright. + A contract carrying `? verify:` can only weaken one - it records that somebody asserted the guarantee, not that anybody checked it. + Never refute on a `?`. +- **Violations are findings.** + A change to a guarantee site that breaks the stated guarantee, while reliance sites still assume it, is a first-class finding - the contract names exactly who gets hurt. + +## Who writes entries + +Contract writes are candidates until the user approves them, like every other durable write in this toolkit. + +- **Commit capture and maintenance.** + The commit skill captures new guarantees a commit establishes and re-checks recorded contracts whose guarantee sites the diff touches, with typed verdicts per the ledger-capture step of its sync-checks reference. + This check is what keeps citations trustworthy enough to refute anything. +- **Review refutations.** + When verification refutes a finding by discovering a boundary fact recorded nowhere - the guard, the value domain, the retry owner - that fact becomes a `C-` write candidate, so the next run inherits the premise instead of re-assuming. +- **Spec invariants.** + The spec's scope section (written by `scope`) records inputs, outputs, and invariants per change; the invariants that cross a boundary promote here at spec promotion time. +- **Audit recovery.** + An audit that observes an undocumented reliance ("billing assumes exactly-once but nothing guarantees it") records the gap as a `C-` candidate with the `?` on whichever side is unverified. diff --git a/factory/references/decision-ledger.md b/factory/references/decision-ledger.md new file mode 100644 index 0000000..2386611 --- /dev/null +++ b/factory/references/decision-ledger.md @@ -0,0 +1,120 @@ +# Decision Ledger + +`docs/decisions.md` in the consuming project: one repo-tracked file that accumulates design decisions across specs and audits. +Spec files are per-change and live in `.dev/{plan-name}/`; the ledger holds what outlives a change - decisions a future spec or audit could collide with. +It is what stops a settled question from being re-litigated every run. + +## Notation + +The marks, used identically here and in a spec's research section: `✓` chosen, `✗` rejected, `?` open, `⚠` accepted downside, `⊘` not doing (with the condition that would reopen it), each line with a short "because" clause. +This file is the canonical definition; consumers cite it rather than restating it. +Every entry keeps its `D-` slug and adds a date and a source - the spec file (`{plan-name}/spec.md`) or audit that produced it. + +### Evidence marks + +A because clause often leans on a claim about the world - "no abuse observed", "matches user expectations". +When the claim is an observed fact that would flip the decision if false, it carries one of two marks: + +- **A citation, in parentheses** - the claim was checked when written. + Code is cited by function name, never line number (line numbers rot on every edit; a function name survives until a rename). + Data is cited by its source - the log, query, metric, or dashboard. + A user statement is cited as `user ()`. + The entry's date scopes the citation: it means *verified then*, and consumers judge staleness at read time like everything else here. +- **`? verify: `** - nobody checked, even at write time. + Inline `?` after a claim is distinct from a line-leading `?` (an open alternative) by position. + +Judgment clauses - "operational burden", "adds a dependency for one call site" - are arguments, not evidence, and take no mark. +The mark is only signal while it's rare; marking every clause drowns it. + +Two maintenance rules keep the marks honest. +A flow that re-reads an entry and can check its `?` right now resolves it - the mark becomes the citation (a ledger write, subject to the consumer's candidate flow). +A citation that no longer resolves - the function renamed away, the dashboard gone - downgrades to `? verify:`; it is never silently removed. + +One `⊘` recorded about this notation itself: typed gap kinds (code-checkable vs. needs-runtime-data vs. needs-user) are not doing - no consumer dispatches on the distinction, so `verify:` stays free text; reopen when three flows would route differently on it. + +## Layout + +Principles at the top, then one section per codebase area. + +``` +# Decisions + +## Principles + +P-no-new-infra: no new infrastructure for an unproven need + promoted 2026-08-08 - recurred in D-response-caching, D-job-queue, D-metrics-store + +## Auth + +risk: high - domains: security, data (updated 2026-08-02, review oauth2-providers) + +D-session-length: How long do sessions last? (2026-07-14, login-sessions/spec.md) + ✓ 30 days sliding - matches user expectations ? verify: no support-ticket data checked ⚠ stolen-token window is long + ✗ 24h fixed - support burden from daily re-login + +D-rate-limit-login: Rate limit the login endpoint? (2026-08-02, security audit) + ⊘ not doing - no abuse observed (checkLoginAttempts audit log, none flagged); reopen if failed-login volume exceeds 100/day +``` + +## Area headers + +Two dated lines open each area section. +Both are priors, and a prior biases whoever holds it - so consumers read them only at the edges of a run: before dispatch (routing, gating how much autonomy a fix gets) or after findings exist (presentation, escalation). +Never during the walk itself, where they would shape what gets found - a walker who knows "high risk" starts seeing danger everywhere, and one who knows "the user knows this area cold" starts deferring. +Headers change how findings are *treated*, never whether things get *found*. + +`risk: - domains: (updated , )`. +Maintained mechanically from commit `Severity:`/`Risk:` trailers by the commit skill and by review runs, and its two halves age differently. +**Domains accumulate** - they're sticky facts about the area (auth touches tokens forever), so new `Risk:` values union in and are never removed mechanically; only a human prunes one, when the area genuinely sheds it. +**The level overwrites** - it's an observation about the most recent trailer-carrying changes (max of their `Severity:` values), not an all-time high-water mark. +Never ratchet: an area whose risky code was removed must be able to come back down, and the date tells consumers how stale the observation is. +Commits without severity trailers leave the level untouched. +A section without a risk line carries no signal - treat it as unassessed, not as safe. + +## Promoting principles + +A `P-` entry names a rationale that has recurred. +When the same because-clause logic picks or rejects alternatives in a third decision, promote it: kebab slug, one-line statement, the decisions it recurred in. +Never author a principle ahead of recurrence - three citations is the bar. +Once named, cite it by ID (`✗ Redis - P-no-new-infra`); "violates P-no-new-infra" replaces re-deriving the argument in reviews and audits. + +## What gets promoted + +From a finished spec or audit, copy the entries that pass this test: **would a future spec or audit in this area need to know this was decided?** +Approach choices, `⊘` lines with reopen conditions, and accepted-`⚠` tradeoffs usually pass; change-local trivia (a variable name, an internal function split) doesn't. +Copy entries verbatim with date + source; don't rewrite them. + +## Recommender contract + +For any flow that emits recommendations. +The order is the point: + +1. **The contract starts after judgment.** + A consumer skill brings the ledger into scope only at its reporting step - findings formed, not yet presented. + Analysis that reads recorded conclusions before forming its own inherits them instead of testing them; a fresh pass that collides with a recorded decision is signal, not waste. + When wiring a new consumer, sequence it so the ledger's first mention is its first use - earlier phases never name it, and never carry a "don't read it yet" guard, which only advertises the file where it must stay out of scope. +2. **Reconcile.** + Grep the ledger sections for the touched areas and classify every finding that collides with an entry: + - `still-holds` - a `⊘` or accepted `⚠` covers the finding and its reopen condition isn't met. + Suppress the finding but report the check ("D-rate-limit-login still holds - failed logins ~20/day"). + A silent suppression is indistinguishable from never having checked. + If the reopen condition can't be verified this run, the report carries the gap instead of implying a check - "still holds `? verify:` current failed-login volume" - a suppression resting on unverified evidence is still a suppression, but says so. + - `reopened` - the entry's reopen condition now holds. + Raise it citing the condition and the evidence that tripped it, not as a fresh recommendation. + - `diverged` - the fresh analysis reached a different conclusion than a recorded `✓` or `⊘`, and no reopen condition explains it. + Surface the disagreement as its own item: either the old decision missed something or the new analysis lacks its context. + Never silently suppress, never silently override. + + Reconcile also maintains evidence marks on the entries it touched, per Notation; those are ledger writes and wait for step 4 like any other. + + Findings with no collision pass through unchanged. + The reconcile report also states whether the pass was `clean` or `contaminated` - contaminated meaning recorded conclusions entered context before findings were formed (an unlucky grep, a file read that pulled them in). + A suppression from a contaminated pass proves nothing; say which findings were exposed. +3. **Recover.** + The walk saw decisions nobody recorded - timeouts, retry counts, validation gaps, structural choices, defaults of any kind. + For each one worth remembering (promotion test), ask what problem it was solving and record a ledger entry with the answer - or `no known problem - unexamined default`, which marks a decision nobody made and invites ratification. + Recovery rides along with reporting; it is not a separate pass over the code. +4. **Write back.** + After the user responds to findings: a declined recommendation becomes a `⊘` line with a concrete reopen condition, dated, sourced to this audit - declined always gets written; that's what makes the next run stateful. + An accepted one becomes a `✓` entry if it passes the promotion test, or warrants a recommended scope run if it carries enough decisions to need one. + A recommendation that's real but needs a call the run can't make becomes an `[open]` entry carrying the alternatives - escalations get a durable home instead of dying in a run directory. diff --git a/factory/references/dependency-rules.md b/factory/references/dependency-rules.md new file mode 100644 index 0000000..0e0ca84 --- /dev/null +++ b/factory/references/dependency-rules.md @@ -0,0 +1,47 @@ +# Dependency Rules + +`docs/dependencies.md` in the consuming project: one repo-tracked file declaring which modules may depend on which, in a shape a checker can parse and agents cannot argue with. +It exists because prose architecture rules soften into guidelines inside a long context; a checker's exit code does not. +`scope` drafts and updates the rules when a change touches module boundaries; `ship` enforces them in its gauntlet phase with a generated checker. + +## Notation + +A human-readable header, then one fenced `rules` block the checker parses: + +```` +# Dependencies + +Domain logic stays pure; the UI reaches it only through the app layer. + +```rules +[modules] +domain = src/domain/** +app = src/app/** +ui = src/ui/** +infra = src/infra/** + +[allowed] +app -> domain +ui -> app +infra -> domain +``` +```` + +Semantics, kept deliberately small: + +- `[modules]` names each module and binds it to one or more path globs (comma-separated). Every source file the checker scans must match exactly one module; a file matching none or several is itself a violation, so the map stays honest. +- `[allowed]` lists the permitted dependency edges as `from -> to, to`. **Anything not listed is forbidden** - an allowlist, not a denylist, so a new dependency is a deliberate edit to this file, never a drive-by import. +- Dependencies within a module are always allowed. Edges are not transitive: `ui -> app` and `app -> domain` do not grant `ui -> domain`; write the edge if it is wanted. +- A dependency is any static reference the language makes checkable: imports, includes, requires. Runtime indirection (dependency injection, events) is invisible to the checker by design - that is what makes inverting a dependency the standard fix. + +## Who does what + +- **scope** drafts the rules. A change that adds a module, adds an edge, or is blocked by an existing edge lands in the spec as a decision (`D-` entry with the alternatives: add the edge, invert the dependency, insert an interface, split the module). The chosen resolution edits this file as part of a change set - the edit is visible in review, never implicit. +- **ship** enforces them in its gauntlet phase. Its checker parses the `rules` block, maps changed files to modules, extracts their static dependencies, and fails on any edge not in `[allowed]`. Fix agents resolve violations by changing the code (invert, interface, split), never by editing this file - a rule change is a decision that belongs to a spec, not to a fix loop. +- **ship**'s architecture lens treats this file as settled context: a diff that conforms needs no boundary debate; a diff that edits the rules is reviewed as the decision it is. + +## What belongs here + +Module-level edges only. +Function-level or file-level rules drown the signal and rot fast; the compiler and `docs/contracts.md` cover finer grain. +A project without meaningful module boundaries yet does not need this file - `ship` skips the check when the file is absent, and says so rather than inventing rules. diff --git a/factory/references/factory-run.md b/factory/references/factory-run.md new file mode 100644 index 0000000..db3ecb6 --- /dev/null +++ b/factory/references/factory-run.md @@ -0,0 +1,59 @@ +# Factory Run Protocol + +Owned by the `run` skill. Every phase copy (`scope`, `scope-review`, `build`, `ship`) reads this file for the result envelope and the unattended policy; `run/SKILL.md` also reads it for the run state schema and the launch prompt shape. + +## Run state + +`.dev/{plan}/factory-run.json`, written only by the `run` skill, via `run-state.py`. It exists from `init` on, before any branch. Fields: + +- `plan`, `request`, `base` (the default branch at `init`). +- `branch`, `spec_sha256`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `dirty_files` - all written at `handoff`. +- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. +- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. +- `repairs`: an ordered list of `{phase, attempt, description, files: [], evidence: []}`. + +## Result envelope: `factory.result/1` + +Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its last action, one file per attempt. The orchestrator reads it leniently: a missing or unparseable file is a failed attempt with that reason, and an unrecognized extra key is ignored rather than rejected. + +```json +{ + "schema": "factory.result/1", + "phase": "build", + "skill_path": "/abs/path/to/factory/skills/build/SKILL.md", + "status": "done", + "reason": "", + "artifacts": ["path/or/description"], + "counts": {}, + "auto_decided": ["escalation text -> option chosen"], + "stop": { "kind": "secret.found", "action": "what a person must do" } +} +``` + +- `schema` - always the literal `factory.result/1`. +- `phase` - the phase name. +- `skill_path` - the exact path the launch prompt passed for `{phase}-skill-root`'s `SKILL.md`, echoed back so a subagent having opened the file it was given is visible in a file the orchestrator already reads (D-skill-locating). +- `status` - `done`, `failed`, or `stopped`. +- `reason` - required when not `done`; the failure or stop reason in plain text. +- `artifacts` - paths or short descriptions of what the phase produced. +- `counts` - phase-specific numbers (scenario counts, decisions made), never judgment evidence on their own. +- `auto_decided` - escalations the phase answered itself under the unattended policy. +- `stop` - present only when `status` is `stopped`: `kind` is `secret.found` or `action.destructive`, `action` is the exact step a person must take. + +## The unattended policy + +After the go, no phase skill asks a person anything. An escalation that dev's version would ask a human about is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. + +`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. Every other phase copy, and `run/SKILL.md`'s own person-routing lines, live inside `` / `` blocks or outside the scan root entirely. + +## Launch prompt shape + +Each phase subagent's prompt carries, in order: + +1. The phase's skill path: the absolute path to `factory/skills/{phase}/SKILL.md`, with the instruction to read `{phase}-skill-root` as that path's parent directory (a subagent reading a file cannot resolve `{phase}-skill-root}`-style placeholders on its own). +2. The plan name and the plan directory's absolute path. +3. The attempt number. +4. On a relaunch: the previous attempt's reason and explicit guidance naming what it did and what is required instead - never "try again". +5. The scratch root for this attempt: `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/`. +6. The result path this attempt must write: `.dev/{plan}/results/{phase}-{attempt}.json`. +7. The instruction that this phase never launches another phase and, after the go, never asks a person anything - decide under the unattended policy above and record it. diff --git a/factory/references/jira.md b/factory/references/jira.md new file mode 100644 index 0000000..9ccda17 --- /dev/null +++ b/factory/references/jira.md @@ -0,0 +1,115 @@ +# Jira mirror reference + +This reference is used only when the consuming repository opts into Jira in +`.dev/config.json`. The `.dev/` files remain the source of truth; Jira mirrors +the records identified by the keys persisted in those files. + +## Configuration and activation + +Read `.dev/config.json` before doing any Jira work. The supported shape is: + +```json +{ + "jira": { + "enabled": true, + "site": "acme.atlassian.net", + "project": "PROJ" + } +} +``` + +`jira.project` is required when enabled. `jira.site` is optional and should be +passed to `acli` when present. If the file is absent, malformed, Jira is +missing, or `jira.enabled` is false, the workflow is pure-local. In the +disabled case do not mention Jira, ask a Jira question, or invoke `acli`. + +When enabled, check that `acli` is installed and authenticated before the first +operation. A useful check is: + +```text +acli jira auth status [--site SITE] +``` + +If the check fails, stop and ask the user to install or authenticate `acli`, or +disable Jira sync, per the failure protocol below. + +## Command shapes + +The exact flags may vary slightly with the installed `acli` release. Preserve +the intent and verify every created issue with `workitem view`. Treat a +non-zero exit, invalid JSON, missing key, or ambiguous result as a failure. + +List open Initiatives for the configured project: + +```text +acli jira workitem search --project PROJ --type Initiative --status open --json [--site SITE] +``` + +Ask the user to choose one returned Initiative. If none is returned, stop with +a clear message that an existing open Initiative is required; never create one. + +Create the spec Epic linked to the chosen Initiative, then verify it: + +```text +acli jira workitem create --project PROJ --type Epic --summary SUMMARY --parent INITIATIVE-1 --json [--site SITE] +acli jira workitem view EPIC-1 --json [--site SITE] +``` + +Create a task issue for a change set under the Epic, then verify it: + +```text +acli jira workitem create --project PROJ --type Task --summary SUMMARY --parent EPIC-1 --json [--site SITE] +acli jira workitem view TASK-1 --json [--site SITE] +``` + +Discover the project's available transitions once per issue type per run, from +a representative issue, and reuse the names; re-discover only after a failed or +ambiguous transition. Match the displayed names exactly and require one +unambiguous transition for the requested state: + +```text +acli jira workitem transition-list --issue EPIC-1 --json [--site SITE] +acli jira workitem transition --issue EPIC-1 --transition "In Progress" --json [--site SITE] +``` + +Use the same discovery and transition sequence for task issues. + +When a change set is superseded by a spec revision, transition the old issue +to its available closed state and add the required comment: + +```text +acli jira workitem transition-list --issue TASK-1 --json [--site SITE] +acli jira workitem transition --issue TASK-1 --transition "Done" --json [--site SITE] +acli jira workitem comment --issue TASK-1 --body "superseded by change set 3" --json [--site SITE] +``` + +## Persistence and timing + +After the Epic is created, write `Jira: PROJ-1` directly under the spec title +in `spec.md`. After each task issue is created, write its key directly under +the matching change set in the spec's change plan. + +`scope` lists Initiatives during its interview and creates the Epic only +after the change plan is final. It then creates one Task per change set; a +re-run creates Tasks only for change sets that don't carry a key yet. On +supersede, close and comment the old issue before creating and persisting the +replacement issue. + +build transitions the persisted Epic to In Progress at the start. The +orchestrator transitions each persisted change-set issue to In Progress +immediately when its wave is dispatched, and to Done only after the change set +is verified and its commit lands. + +## Failure protocol + +Jira is first-class when enabled. Any failed auth check, command, JSON parse, +verification, missing key, missing Initiative, unknown transition, or +ambiguous transition stops the skill immediately. Explain the failed +operation and ask the user how to proceed. Never silently skip Jira, continue +with divergent local-only state, invent an issue key, or mark a transition +successful without verification. + +The Epic is not polled or closed by the skills. When build creates the work +branch and when ship opens the PR, include the Epic key in the branch name and +at the start of the PR title. A documented Jira automation rule closes the Epic +after the linked PR is merged. diff --git a/factory/references/plan-layout.md b/factory/references/plan-layout.md new file mode 100644 index 0000000..f1bb69d --- /dev/null +++ b/factory/references/plan-layout.md @@ -0,0 +1,34 @@ +# Plan Directory Layout + +`.dev/{plan-name}/` is the durable home of one change, named by a kebab-case slug for the outcome. +This reference owns the layout, the locating convention, and the diff scope; skills state only their own role. + +## Files and owners + +| File | Written by | Read by | +| ---- | ---------- | ------- | +| `request.md` | the `run` skill, before `scope` starts | `scope` | +| `spec.md` | `scope`; `scope-review` (verified refinements only) | everyone downstream | +| `spec-review_N.md` | `scope-review` (next free index) | `scope` (remediation), re-reviews | +| `implementation-notes.md` | `build` (append-only) | `ship`, `to-pitch`, `to-quiz` | +| `review_N.md` | `ship` (next free index) | `scope` (remediation), re-reviews | +| `pr.md` | `ship` (phase 3, overwritten per run) | the PR tool via `--body-file`; re-runs | +| `factory-run.json` | the `run` skill only | the `run` skill's judgment loop | +| `results/{phase}-{attempt}.json` | the phase skill, as its last action | the `run` skill's judgment loop | +| `.dev/config.json` | the user | any skill with Jira behavior | + +Each producing skill also renders an HTML companion under `/tmp/{project-slug}/reports/` per [reporting.md](reporting.md), named by that skill. +One writer per file; every other skill only reads. + +## Locating the plan directory + +Match the current branch name to a `.dev/{plan-name}/` slug; fall back to commit messages, then to the only directory in a plausible state for the skill (e.g. the only spec whose change sets aren't done). +Ask which one only if more than one is a plausible match; otherwise proceed without waiting. + +## Diff scope + +Skills that operate on "the change" (`ship`) scope to the local branch against the default branch (`git merge-base HEAD origin/{default}`), diffed with local git only (never `gh pr diff`), excluding lockfiles, build output, minified files, binaries, fonts, and snapshots: + +``` +':!*lock*' ':!go.sum' ':!dist/' ':!build/' ':!*.min.*' ':!*.map' ':!*.png' ':!*.jpg' ':!*.gif' ':!*.webp' ':!*.woff*' ':!*.ttf' ':!**/__snapshots__/' +``` diff --git a/factory/references/reporting.md b/factory/references/reporting.md new file mode 100644 index 0000000..5c14c18 --- /dev/null +++ b/factory/references/reporting.md @@ -0,0 +1,23 @@ +# Report Artifacts + +Every producing skill renders its output as self-contained HTML under `/tmp/{project-slug}/reports/`. +This reference owns the shared etiquette; each skill states only its own output filename and data shape. + +## Rendering + +- Copy the skill's template to the output path, replacing **only the data block** between its `*_DATA_START` / `*_DATA_END` markers. + The rendering engine below the markers is generic and reads only that shape - never touch it on a data refresh. +- Before first authoring or restyling a template, load an installed artifact or frontend design skill; a plain data refresh on an existing template doesn't need it again. +- Open the rendered file with the host's browser integration when available; otherwise give the user a clickable local path. + Do not fail solely because GUI launch is unavailable. + +## Publishing + +The local file is the deliverable. +Publish with an artifact-publishing tool only when the user asks for a shareable link, using a stable per-skill favicon and a title and description naming the artifact's subject. +Never publish unprompted, and if the host has no publisher, the local HTML remains the deliverable - say so instead of apologizing. + +## Reading another skill's report + +Consumers of a rendered report read its data block, not the whole file, and extract only the fields they need. +In particular, `E2E_DATA` embeds one base64 `dataUri` per screenshot step: never ingest those payloads - scenario ids, titles, statuses, and the `summary` are the useful content. diff --git a/factory/scripts/architecture-check.py b/factory/scripts/architecture-check.py new file mode 100755 index 0000000..3f2417a --- /dev/null +++ b/factory/scripts/architecture-check.py @@ -0,0 +1,135 @@ +#!/usr/bin/env python3 +"""Check that docs/architecture.md is present, well-formed, and current. + +Usage: + python3 architecture-check.py docs/architecture.md [--root DIR] [--touched FILE ...] + +Exit 2 when the file is missing (run an initial capture), 1 with one violation +per line when the overview is stale or malformed, 0 when it is current for what was +checked. Violations: a required section missing, a backticked path that does +not exist under --root, and a --touched file that no component's "Lives at" +path covers ("uncharted"). +""" + +from __future__ import annotations + +import argparse +import glob +import re +import sys +from pathlib import Path + +SECTIONS = ["Components", "Flows", "Boundaries", "Cross-cutting", "Entry points"] +HEADER = re.compile(r"^##\s+(.+?)\s*$", re.M) +BACKTICK = re.compile(r"`([^`\n]+)`") +PATH_LIKE = re.compile(r"^[^\s]+(/[^\s]*|\.[A-Za-z0-9]{1,6})$") +TABLE_ROW = re.compile(r"^\|(.+)\|\s*$") + + +def sections(text: str) -> dict[str, str]: + found: dict[str, str] = {} + matches = list(HEADER.finditer(text)) + for index, match in enumerate(matches): + end = matches[index + 1].start() if index + 1 < len(matches) else len(text) + found[match.group(1)] = text[match.end():end] + return found + + +def path_exists(root: Path, raw: str) -> bool: + candidate = raw.strip().rstrip("/") + if any(ch in candidate for ch in "*?["): + return bool(glob.glob(str(root / candidate), recursive=True)) + return (root / candidate).exists() + + +def table_rows(section: str) -> list[list[str]]: + rows = [] + for line in section.splitlines(): + match = TABLE_ROW.match(line) + if not match: + continue + cells = [cell.strip() for cell in match.group(1).split("|")] + if all(set(cell) <= set("-: ") for cell in cells): + continue + rows.append(cells) + return rows[1:] if rows else [] + + +def component_paths(section: str) -> dict[str, list[str]]: + paths: dict[str, list[str]] = {} + for cells in table_rows(section): + if not cells: + continue + paths[cells[0]] = [p.strip().rstrip("/") for p in BACKTICK.findall(" ".join(cells[1:])) if PATH_LIKE.match(p.strip())] + return paths + + +def covered(touched: str, patterns: list[str]) -> bool: + for pattern in patterns: + if any(ch in pattern for ch in "*?["): + base = pattern.split("*", 1)[0].rstrip("/") + if base and touched.startswith(base + "/"): + return True + elif touched == pattern or touched.startswith(pattern + "/"): + return True + return False + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("map") + parser.add_argument("--root", default=".") + parser.add_argument("--touched", nargs="*", default=[]) + args = parser.parse_args() + + root = Path(args.root).resolve() + map_path = Path(args.map) + if not map_path.exists(): + print(f"missing: {map_path} - run an initial capture") + return 2 + text = map_path.read_text(encoding="utf-8", errors="replace") + violations: list[str] = [] + + if not re.search(r"^Purpose:", text, re.M): + violations.append("section: no 'Purpose:' line") + if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.M): + violations.append("section: no dated 'Captured:' line") + found = sections(text) + order = [name for name in found if name in SECTIONS] + for name in SECTIONS: + if name not in found: + violations.append(f"section: missing '## {name}'") + if order != [name for name in SECTIONS if name in found]: + violations.append(f"section: out of order, expected {', '.join(SECTIONS)}") + + for raw in BACKTICK.findall(text): + candidate = raw.strip() + if PATH_LIKE.match(candidate) and not path_exists(root, candidate): + violations.append(f"stale: `{candidate}` does not exist under {root}") + + components = component_paths(found.get("Components", "")) + if "Components" in found and not components: + violations.append("section: Components table has no rows") + for name, paths in components.items(): + if not paths: + violations.append(f"component: '{name}' names no backticked path in its row") + + all_paths = [p for paths in components.values() for p in paths] + for touched in args.touched: + relative = touched.strip() + try: + relative = str(Path(touched).resolve().relative_to(root)) + except ValueError: + pass + if not covered(relative, all_paths): + violations.append(f"uncharted: {relative} is covered by no component's path") + + if violations: + print("\n".join(violations)) + return 1 + print(f"architecture-check: ok ({len(components)} components, {len(args.touched)} touched files covered)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/factory/scripts/skill-metrics.py b/factory/scripts/skill-metrics.py new file mode 100755 index 0000000..d9d2b7b --- /dev/null +++ b/factory/scripts/skill-metrics.py @@ -0,0 +1,333 @@ +#!/usr/bin/env python3 +"""Measure one skill run: time, tokens, agents, tool calls, and git delta. + +Usage: + python3 skill-metrics.py start {skill} + python3 skill-metrics.py end {skill} [--count key=value ...] + +`start` snapshots the git state and locates the session transcript under +$CLAUDE_CONFIG_DIR (default ~/.claude), anchoring on the transcript line that +invoked the skill. `end` sums everything from that anchor across the main +transcript and every subagent transcript the run spawned, diffs git against +the snapshot, prints a markdown table, and appends a row to .dev/metrics.jsonl +so later runs can be compared against earlier ones. Every value is measured; +nothing here is narrated from memory. Missing transcript or git degrade to +"n/a" rather than failing the run. +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import statistics +import subprocess +import sys +import time +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +TEST_FILE = re.compile(r"(^|/)(tests?|specs?|__tests__)(/|$)|[._-](test|spec)s?\.[A-Za-z]+$|^test_.*\.py$") +TOKEN_KEYS = ("input_tokens", "cache_read_input_tokens", "cache_creation_input_tokens", "output_tokens") + + +def now_iso() -> str: + return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.%fZ") + + +def parse_iso(value: str) -> float: + return datetime.strptime(value[:19], "%Y-%m-%dT%H:%M:%S").replace(tzinfo=timezone.utc).timestamp() + + +def run_git(*args: str) -> str | None: + try: + return subprocess.run(["git", *args], capture_output=True, text=True, check=True).stdout + except (subprocess.CalledProcessError, FileNotFoundError): + return None + + +def repo_root() -> Path: + top = run_git("rev-parse", "--show-toplevel") + return Path(top.strip()) if top else Path.cwd() + + +def state_dir() -> Path: + path = Path("/tmp") / repo_root().name / "metrics" + path.mkdir(parents=True, exist_ok=True) + return path + + +# --- transcript ------------------------------------------------------------- + +def find_transcript() -> Path | None: + config = Path(os.environ.get("CLAUDE_CONFIG_DIR", "~/.claude")).expanduser() + slug = re.sub(r"[^A-Za-z0-9]", "-", str(Path.cwd())) + project = config / "projects" / slug + if not project.is_dir(): + return None + files = sorted(project.glob("*.jsonl"), key=lambda p: p.stat().st_mtime, reverse=True) + return files[0] if files else None + + +def iter_lines(path: Path, start: int = 0): + with path.open(encoding="utf-8") as handle: + for index, raw in enumerate(handle): + if index < start: + continue + try: + yield index, json.loads(raw) + except json.JSONDecodeError: + continue + + +def find_anchor(path: Path, skill: str) -> tuple[int, str | None]: + """Line index and timestamp of the last invocation of this skill.""" + marker = re.compile(rf"/(?:[\w-]+:)?{re.escape(skill)}") + anchor, stamp, count = 0, None, 0 + for index, obj in iter_lines(path): + count = index + 1 + if obj.get("type") != "user": + continue + content = (obj.get("message") or {}).get("content") + if isinstance(content, str) and marker.search(content): + anchor, stamp = index, obj.get("timestamp") + if stamp is None: + anchor = count + return anchor, stamp + + +def aggregate(path: Path, start: int, since: float | None) -> dict: + """Token, turn and tool totals for one transcript from a line index.""" + usage: dict[str, dict] = {} + tools: Counter = Counter() + models: Counter = Counter() + first_stamp = None + for _, obj in iter_lines(path, start): + stamp = obj.get("timestamp") + if first_stamp is None and stamp: + first_stamp = stamp + if obj.get("type") != "assistant": + continue + message = obj.get("message") or {} + message_id = message.get("id") or obj.get("uuid") + if message.get("usage"): + usage[message_id] = message["usage"] + if message.get("model"): + models[message["model"]] = 1 + for block in message.get("content") or []: + if isinstance(block, dict) and block.get("type") == "tool_use": + tools[block.get("name", "?")] += 1 + if since is not None and first_stamp and parse_iso(first_stamp) < since: + return {} + tokens = {key: sum(int(u.get(key) or 0) for u in usage.values()) for key in TOKEN_KEYS} + return {"turns": len(usage), "tokens": tokens, "tools": dict(tools), "models": sorted(models)} + + +def subagent_files(transcript: Path) -> list[Path]: + folder = transcript.with_suffix("") / "subagents" + return sorted(folder.glob("*.jsonl")) if folder.is_dir() else [] + + +# --- git --------------------------------------------------------------------- + +def numstat(ref: str | None) -> tuple[int, int, set[str]]: + args = ["diff", "--numstat"] + ([ref] if ref else []) + output = run_git(*args) or "" + added = removed = 0 + files: set[str] = set() + for line in output.splitlines(): + parts = line.split("\t") + if len(parts) != 3: + continue + added += int(parts[0]) if parts[0].isdigit() else 0 + removed += int(parts[1]) if parts[1].isdigit() else 0 + files.add(parts[2]) + untracked = (run_git("ls-files", "--others", "--exclude-standard") or "").split() + for path in untracked: + files.add(path) + try: + added += sum(1 for _ in Path(path).open(encoding="utf-8", errors="ignore")) + except OSError: + pass + return added, removed, files + + +def git_snapshot() -> dict: + head = run_git("rev-parse", "HEAD") + added, removed, files = numstat("HEAD" if head else None) + return {"head": head.strip() if head else None, "added": added, "removed": removed, "files": sorted(files)} + + +def git_delta(start: dict) -> dict | None: + head = start.get("head") + if not head: + return None + added, removed, files = numstat(head) + commits = run_git("rev-list", "--count", f"{head}..HEAD") + return { + "commits": int(commits.strip()) if commits and commits.strip().isdigit() else 0, + "files": len(files), + "test_files": sum(1 for f in files if TEST_FILE.search(f)), + "added": max(0, added - start["added"]), + "removed": max(0, removed - start["removed"]), + "baseline_dirty_files": len(start["files"]), + } + + +# --- commands ---------------------------------------------------------------- + +def cmd_start(skill: str) -> int: + transcript = find_transcript() + anchor, stamp = find_anchor(transcript, skill) if transcript else (0, None) + state = { + "skill": skill, + "started_at": stamp or now_iso(), + "anchored": stamp is not None, + "transcript": str(transcript) if transcript else None, + "anchor_line": anchor, + "git": git_snapshot(), + } + (state_dir() / f"{skill}.json").write_text(json.dumps(state, indent=2), encoding="utf-8") + where = f"transcript line {anchor}" if stamp else "now (no invocation marker found)" + print(f"metrics: {skill} run started, anchored at {where}") + return 0 + + +def fmt_tokens(tokens: dict) -> str: + return "in {} / cache read {} / cache write {} / out {}".format( + *(human(tokens[key]) for key in TOKEN_KEYS) + ) + + +def human(value: float) -> str: + for unit, size in (("M", 1_000_000), ("k", 1_000)): + if value >= size: + return f"{value / size:.1f}{unit}" + return str(int(value)) + + +def duration(seconds: float) -> str: + minutes, secs = divmod(int(seconds), 60) + hours, minutes = divmod(minutes, 60) + return f"{hours}h {minutes:02d}m" if hours else f"{minutes}m {secs:02d}s" + + +def total(tokens: dict) -> int: + return sum(tokens.get(key, 0) for key in TOKEN_KEYS) + + +def cmd_end(skill: str, counts: dict[str, str]) -> int: + state_file = state_dir() / f"{skill}.json" + if not state_file.exists(): + print(f"metrics: no start snapshot for {skill}; run `start {skill}` at invocation", file=sys.stderr) + return 1 + state = json.loads(state_file.read_text(encoding="utf-8")) + since = parse_iso(state["started_at"]) + elapsed = time.time() - since + + main: dict = {} + agents: list[dict] = [] + transcript = Path(state["transcript"]) if state.get("transcript") else None + if transcript and transcript.exists(): + main = aggregate(transcript, state["anchor_line"], None) + agents = [a for a in (aggregate(p, 0, since) for p in subagent_files(transcript)) if a] + agent_tokens = {key: sum(a["tokens"][key] for a in agents) for key in TOKEN_KEYS} + all_tokens = {key: main.get("tokens", {}).get(key, 0) + agent_tokens[key] for key in TOKEN_KEYS} + tools = Counter(main.get("tools", {})) + for agent in agents: + tools.update(agent["tools"]) + delta = git_delta(state["git"]) + + record = { + "skill": skill, + "started_at": state["started_at"], + "ended_at": now_iso(), + "seconds": int(elapsed), + "anchored": state["anchored"], + "turns": main.get("turns", 0), + "agents": len(agents), + "tools": dict(tools), + "tokens_main": main.get("tokens", {}), + "tokens_agents": agent_tokens, + "tokens_total": total(all_tokens), + "git": delta, + "counts": counts, + } + ledger = repo_root() / ".dev" / "metrics.jsonl" + previous = [] + if ledger.exists(): + for line in ledger.read_text(encoding="utf-8").splitlines(): + try: + row = json.loads(line) + except json.JSONDecodeError: + continue + if row.get("skill") == skill: + previous.append(row) + ledger.parent.mkdir(parents=True, exist_ok=True) + with ledger.open("a", encoding="utf-8") as handle: + handle.write(json.dumps(record) + "\n") + + top_tools = ", ".join(f"{name} {n}" for name, n in tools.most_common(4)) or "none" + rows = [ + ("duration", duration(elapsed) + ("" if state["anchored"] else " (from start call, invocation marker not found)")), + ("orchestrator turns / tool calls", f"{main.get('turns', 0)} / {sum(tools.values())} ({top_tools})"), + ("agents dispatched", str(len(agents))), + ("tokens orchestrator", fmt_tokens(main["tokens"]) if main else "n/a (transcript not found)"), + ("tokens agents", fmt_tokens(agent_tokens) if agents else "0"), + ("tokens total", human(total(all_tokens)) if main else "n/a"), + ] + if delta: + rows.append(( + "git since start", + f"{delta['commits']} commits, {delta['files']} files (+{delta['added']}/-{delta['removed']}), " + f"{delta['test_files']} test files" + + (f"; {delta['baseline_dirty_files']} files were already dirty" if delta["baseline_dirty_files"] else ""), + )) + else: + rows.append(("git since start", "n/a (not a git repo)")) + for key, value in counts.items(): + rows.append((key.replace("_", " "), value)) + if previous and main: + med_tokens = statistics.median(r["tokens_total"] for r in previous if r.get("tokens_total")) + med_secs = statistics.median(r["seconds"] for r in previous) + change = (total(all_tokens) - med_tokens) / med_tokens * 100 if med_tokens else 0 + rows.append(( + f"vs previous {skill} runs", + f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " + f"this run {change:+.0f}% tokens", + )) + + width = max(len(name) for name, _ in rows) + print(f"## {skill} run metrics\n") + print(f"| {'metric'.ljust(width)} | value |") + print(f"| {'-' * width} | ----- |") + for name, value in rows: + print(f"| {name.ljust(width)} | {value} |") + print(f"\nledger: {ledger}") + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + sub = parser.add_subparsers(dest="command", required=True) + sub.add_parser("start").add_argument("skill") + end = sub.add_parser("end") + end.add_argument("skill") + end.add_argument("--count", action="append", default=[], metavar="KEY=VALUE", + help="skill-specific measured counter to include, repeatable") + args = parser.parse_args() + if args.command == "start": + return cmd_start(args.skill) + counts = {} + for item in args.count: + if "=" not in item: + parser.error(f"--count expects KEY=VALUE, got {item!r}") + key, value = item.split("=", 1) + counts[key.strip()] = value.strip() + return cmd_end(args.skill, counts) + + +if __name__ == "__main__": + sys.exit(main()) From 1115104b712e86856af91ce3ce66d7b348f41f18 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 11:15:20 +0200 Subject: [PATCH 03/30] feat(factory): phase skill copies What: copy dev's scope, scope-review, build, and ship skills into factory/skills/, rewrite every unattended-check hit with a factory-policy recorded decision instead of a question to a person, add each skill's Factory context section, and repoint scope-review's cross-references inside factory/. Fix the one-line Codex spawn_agent naming gap in both the factory copy and dev/skills/ship/references/orchestration.md's Native transport prose, so a copied ship reaches Codex's real transport rung. Why: these are the four phases the run skill orchestrates unattended; the dev fix keeps the source skill's own transport documentation accurate too. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- dev/skills/ship/references/orchestration.md | 8 +- factory/skills/build/SKILL.md | 105 ++++++ factory/skills/build/references/e2e-report.md | 50 +++ factory/skills/build/references/layers.md | 50 +++ factory/skills/build/references/mocking.md | 52 +++ factory/skills/build/references/parallel.md | 63 ++++ factory/skills/build/references/tests.md | 91 +++++ factory/skills/build/scripts/check-tests.py | 144 ++++++++ .../skills/build/templates/e2e-report.html | 248 ++++++++++++++ factory/skills/scope-review/SKILL.md | 132 ++++++++ .../skills/scope-review/references/lenses.md | 55 ++++ factory/skills/scope/SKILL.md | 163 +++++++++ factory/skills/scope/references/bootstrap.md | 72 ++++ .../skills/scope/references/data-schema.md | 64 ++++ .../skills/scope/references/reverse-mode.md | 20 ++ factory/skills/scope/scripts/lint-spec.py | 208 ++++++++++++ factory/skills/scope/templates/spec.html | 151 +++++++++ factory/skills/ship/SKILL.md | 143 ++++++++ factory/skills/ship/references/data-schema.md | 46 +++ factory/skills/ship/references/gauntlet.md | 56 ++++ factory/skills/ship/references/lenses.md | 102 ++++++ .../ship/references/orchestration-heavy.md | 108 ++++++ .../skills/ship/references/orchestration.md | 150 +++++++++ .../skills/ship/references/pull-request.md | 75 +++++ factory/skills/ship/references/remediation.md | 42 +++ .../skills/ship/references/report-format.md | 50 +++ factory/skills/ship/references/tools.md | 79 +++++ .../skills/ship/scripts/aggregate-findings.py | 185 +++++++++++ factory/skills/ship/scripts/pr-evidence.py | 311 ++++++++++++++++++ factory/skills/ship/scripts/run-pi-agents.sh | 69 ++++ factory/skills/ship/templates/review.html | 214 ++++++++++++ .../skills/ship/references/orchestration.md | 8 +- 32 files changed, 3306 insertions(+), 8 deletions(-) create mode 100644 factory/skills/build/SKILL.md create mode 100644 factory/skills/build/references/e2e-report.md create mode 100644 factory/skills/build/references/layers.md create mode 100644 factory/skills/build/references/mocking.md create mode 100644 factory/skills/build/references/parallel.md create mode 100644 factory/skills/build/references/tests.md create mode 100755 factory/skills/build/scripts/check-tests.py create mode 100644 factory/skills/build/templates/e2e-report.html create mode 100644 factory/skills/scope-review/SKILL.md create mode 100644 factory/skills/scope-review/references/lenses.md create mode 100644 factory/skills/scope/SKILL.md create mode 100644 factory/skills/scope/references/bootstrap.md create mode 100644 factory/skills/scope/references/data-schema.md create mode 100644 factory/skills/scope/references/reverse-mode.md create mode 100755 factory/skills/scope/scripts/lint-spec.py create mode 100644 factory/skills/scope/templates/spec.html create mode 100644 factory/skills/ship/SKILL.md create mode 100644 factory/skills/ship/references/data-schema.md create mode 100644 factory/skills/ship/references/gauntlet.md create mode 100644 factory/skills/ship/references/lenses.md create mode 100644 factory/skills/ship/references/orchestration-heavy.md create mode 100644 factory/skills/ship/references/orchestration.md create mode 100644 factory/skills/ship/references/pull-request.md create mode 100644 factory/skills/ship/references/remediation.md create mode 100644 factory/skills/ship/references/report-format.md create mode 100644 factory/skills/ship/references/tools.md create mode 100755 factory/skills/ship/scripts/aggregate-findings.py create mode 100755 factory/skills/ship/scripts/pr-evidence.py create mode 100755 factory/skills/ship/scripts/run-pi-agents.sh create mode 100644 factory/skills/ship/templates/review.html diff --git a/dev/skills/ship/references/orchestration.md b/dev/skills/ship/references/orchestration.md index c6aece4..57cc2ec 100644 --- a/dev/skills/ship/references/orchestration.md +++ b/dev/skills/ship/references/orchestration.md @@ -30,10 +30,10 @@ explicitly requested heavyweight runs only lives in ## Native transport Use the host's plain subagent tool exactly as provided (the Agent tool on -Claude Code, the `task` tool on opencode). Launch all agents of a batch in a -single parallel call: on opencode that means issuing every `task` call of the -batch in one message with the default general subagent. Native subagents -receive the prompt contracts below directly. +Claude Code, `spawn_agent` on Codex, the `task` tool on opencode). Launch all +agents of a batch in a single parallel call: on opencode that means issuing +every `task` call of the batch in one message with the default general +subagent. Native subagents receive the prompt contracts below directly. Result delivery is file-based, because on some hosts subagents run in the background and their final message never reaches the orchestrator: diff --git a/factory/skills/build/SKILL.md b/factory/skills/build/SKILL.md new file mode 100644 index 0000000..84685ae --- /dev/null +++ b/factory/skills/build/SKILL.md @@ -0,0 +1,105 @@ +--- +name: build +description: Execute the change sets of a spec at .dev/{plan-name}/spec.md, proving every scenario with a test at its tagged layer across unit/integration/e2e, running independent change sets in parallel and looping until every change set is done and the e2e suite passes. Use when the user asks to build or implement a spec or its change sets. +disable-model-invocation: true +--- + +# Build + +Execute every change set of a spec, proving each `tests:` scenario with a real test at its tagged layer, until all change sets are done and the e2e run is green. + +Input: `.dev/{plan-name}/spec.md` (from `scope`), located per [../../references/plan-layout.md](../../references/plan-layout.md). +If there is no `spec.md` or its change plan is empty, report `failed` with reason "no spec"; without one there are no layer-tagged scenarios to implement against. + +Read [references/layers.md](references/layers.md), [references/tests.md](references/tests.md), and [references/mocking.md](references/mocking.md) before implementing anything yourself; [references/parallel.md](references/parallel.md) points subagents at them. + +## Workflow + +1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. +2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. +3. For each wave, run its change sets in parallel per the same reference, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). +4. Move straight to the next wave. Never stop after one change set or wave to ask about review. +5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. +6. Then run the full e2e pass per "The e2e layer" below over the whole spec, and loop on failures until it is green. +7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. +8. After the e2e report and CI-parity gate, close per "Closing message". Build never pushes or opens a PR; that is `ship`'s phase 3. + +## Jira sync + +Read `.dev/config.json`; when `jira.enabled` is true, follow +[../../references/jira.md](../../references/jira.md) from before the first `acli` call - it owns the +command shapes, the transition timing, and the failure protocol. The +orchestrator alone invokes `acli`; subagent prompts and the `parallel.md` +contract do not change. + +With an absent or disabled config, perform no Jira behavior or mention. +If still on the default branch, create the work branch before the first commit, +named per jira.md when Jira is enabled; never push it and never open a PR - +`ship` pushes and opens the PR with evidence after its gauntlet and review. + +## Closing message + +Every build run first prints the measured run metrics, pasting the table verbatim: + +```bash +python3 {build-skill-root}/../../scripts/skill-metrics.py end build --count change_sets=N --count scenarios=N --count e2e_passed=N --count e2e_failed=N +``` + +Then it ends with the same two lines, in this order - a green run, a blocked gate, and a run with open deviations all get both: + +1. `Next step: run ship over this work, pointed at .dev/{plan-name}/implementation-notes.md and the {plan-name}-e2e-report.html.` Recommend it; never launch it yourself. +2. Then any question left for the user - a blocked gate, an unresolved deviation. + +A blocked gate never replaces line 1. +Neither does a failed e2e loop: say what is blocked, then still point at `ship`. + +## The change-set loop + +Each change set, whether you run it yourself or a subagent runs it, follows the same loop: + +- Test at the seams the spec's scope section declares, per [references/tests.md](references/tests.md); if the declared boundary is wrong or missing, follow its fallback and log the change under Deviations - do not stall on it. +- Implement in **vertical slices**: one scenario's behavior at a time, its test written before or right after the code - the enforced outcome is what matters, not the ritual order. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. +- Run the change set's own tests and typecheck continuously; once the change set is green, run the spec's Validation block verbatim - it is the wider suite plus typecheck/lint - and only report done when it passes clean. + +## Rules of the loop + +- **Every test must have been seen red.** A test that has never failed proves nothing: earn its green by writing it before the code, or by briefly breaking the behavior once after. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. +- **One slice at a time.** One seam, one behavior, one test, one minimal implementation per cycle. +- **Refactoring is not part of the loop.** It belongs to `ship`'s review phase. +- **Keep going.** A red test, a failing e2e scenario, or an edge case that contradicts the spec is work to do, not a reason to hand back. Fix it, log the deviation, continue. Stop early only when a blocking question makes further work unsafe or wasted. + +## The e2e layer + +E2E scenarios are proven by running the actual application against the **fully mocked environment** defined in [references/mocking.md](references/mocking.md#the-e2e-environment). Run this once per spec, after all change sets are committed, covering every `[e2e]` scenario across change sets. + +1. **Launch the app.** Invoke an installed `run` skill with the mocked environment configured when the host supports direct skill invocation. Otherwise inspect the repository's documented commands and start the app directly. When no safe launch command can be determined, record the decision (mock the launch, skip the affected `[e2e]` scenarios, or report `failed` naming what a person must supply) and continue. +2. **Drive it and capture evidence**, per scenario: + - `kind: "frontend"` - use available browser automation (the host browser integration or Playwright) to exercise the scenario, one screenshot per meaningful step, embedded as a base64 data URI. + - `kind: "non-frontend"` - capture the entity's real before/after state from the run's own output or fixtures. +3. **Never fabricate a screenshot or a data-model-state entry.** Both come from this actual run. +4. **Loop until green.** A failed scenario is a bug: diagnose it, fix the code (a new red-green cycle at the right layer), re-run and re-capture that scenario. Never flip a status to pass without a fresh capture. If a scenario fails three times on the same root cause, write what you found into Deviations and report the phase `failed` with the root cause as the reason. +5. Map the results onto `E2E_DATA` per [references/e2e-report.md](references/e2e-report.md) and render `templates/e2e-report.html` to `/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`, opening and publishing per [../../references/reporting.md](../../references/reporting.md). + +## Implementation notes + +Maintain `.dev/{plan-name}/implementation-notes.md`, appended after each change set completes, never written once at the end. +It is the shared state across waves - parallel change-set agents can't see each other's conversation, only this file and the code - and the evidence `ship`, `to-pitch`, and `to-quiz` read later. + +```markdown +## Change set {n}: {title} +- What was done: ... +- Seams tested: ... +- Tests added: {path::test name}, ... # or "none - {reason}"; the checker reads this line +- Deviations from spec: {edge case found} -> conservative choice made: {what/why} # only when a deviation occurred +``` + +This file is a short running log, not a rendered report. + +## Anti-patterns + +- **Horizontal slicing** - all tests first, then all implementation. Tests then verify an imagined shape and go insensitive to change. +- The other tells - implementation-coupled tests, tautological assertions, top-heavy testing - are defined in [references/tests.md](references/tests.md) and [references/layers.md](references/layers.md); flag and fix them on sight. + +## Factory context + +Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. diff --git a/factory/skills/build/references/e2e-report.md b/factory/skills/build/references/e2e-report.md new file mode 100644 index 0000000..b8f00e1 --- /dev/null +++ b/factory/skills/build/references/e2e-report.md @@ -0,0 +1,50 @@ +# E2E_DATA Schema + +The shape to populate in `templates/e2e-report.html` between the `E2E_DATA_START` / `E2E_DATA_END` markers. +Replace the whole object per the shared etiquette in [../../../references/reporting.md](../../../references/reporting.md). + +```js +const E2E_DATA = { + title: "string - the spec's title", + planName: "string - the .dev/{plan-name} slug", + generatedAt: "string - ISO date of the run this report captures", + + // Decides how each scenario renders: a screenshot gallery, or a data-model-state table. + kind: "frontend" | "non-frontend", + + scenarios: [ + { + id: "string", + title: "string - one line, e.g. 'guest checks out with an expired coupon'", + given: "string", + when: "string", + then: "string", + status: "pass" | "fail", + + // kind === "frontend": one entry per meaningful step, screenshot embedded as a data URI. + screenshots: [ + { step: "string", caption: "string", dataUri: "data:image/png;base64,..." } + ], + + // kind === "non-frontend": one entry per meaningful step. Also a valid supplement for a + // frontend scenario when a step's real effect is a data change a screenshot can't show. + dataModelState: [ + { step: "string", caption: "string", entity: "string", before: "object or string", after: "object or string" } + ], + + logsOrOutput: "string - optional, captured stdout or response body", + durationMs: 0 + } + ], + + summary: { total: 0, passed: 0, failed: 0 } +}; +``` + +## Filling it in honestly + +- **Never fabricate a screenshot or a data-model-state entry.** Both must come from an actual run of the actual application - a screenshot invented to look plausible, or a before/after pair guessed instead of captured, defeats the entire point of this report being the enforceable proof behind an e2e criterion. +- **kind is chosen once per spec**, based on whether the system under test has a UI a screenshot could meaningfully show. A CLI, a backend API, a batch job: `"non-frontend"`. Anything a user clicks through: `"frontend"`. +- **dataModelState is a valid supplement even for a frontend scenario** when a step's real effect is invisible on screen (a queued job, a row written to a table the UI doesn't reflect yet). +- **status must reflect what actually happened.** A scenario that failed and was then fixed gets re-run and re-captured, not silently flipped to pass. +- **summary must match the scenarios array** - recompute it from the real counts, don't hand-write it separately. diff --git a/factory/skills/build/references/layers.md b/factory/skills/build/references/layers.md new file mode 100644 index 0000000..d7ecc5d --- /dev/null +++ b/factory/skills/build/references/layers.md @@ -0,0 +1,50 @@ +# Test Layers + +Every scenario on a change set's `tests:` line in `spec.md` is tagged `[unit]`, `[integration]`, or `[e2e]`. +A scenario is not met until a concrete test exists at its tagged layer - a prose input -> outcome with no test behind it is not done, however confident the implementation looks. +The tag is what keeps tests acting as the spec: it says where the proof has to live, so it can't quietly stay in prose. + +## The pyramid + +Cost, speed, and breadth of failure differ by orders of magnitude between the layers, so their populations should too: many unit tests, fewer integration tests, few e2e scenarios. + +| Layer | Population | Runtime | What it proves | +| ----- | ---------- | ------- | -------------- | +| Unit | most of the suite | milliseconds | one business rule, exactly | +| Integration | a middle band | sub-second to seconds | two owned components really agree | +| E2E | a handful per spec | seconds to minutes | a whole user journey actually works | + +Two rules follow, and they matter more than the ratios: + +- **Push every check down to the cheapest layer that can still fail for the right reason.** If a rule can be wrong in a pure function, test it in a pure function. Proving a rounding rule through a browser click is a slow test that also localizes badly: when it fails, it doesn't tell you where. +- **Push up only what lower layers structurally cannot see.** Wiring, configuration, serialization across a boundary, and the shape of a real user journey are invisible to a unit test no matter how many you write. That is what the upper layers are for, and why they exist at all. + +Two shapes to avoid: the **ice-cream cone**, where the suite is mostly e2e and every change costs a long red-green cycle, and the **hourglass**, where unit and e2e are both fat but nothing tests real collaboration, so integration bugs surface only in production. + +## Unit + +Pure business logic, no I/O. +Fast enough to run on every save. +Mock only at the true architectural boundaries in [mocking.md](mocking.md) - everything else inside the application stays real. +A unit scenario describes a computation or a business rule: "expired coupon is rejected," not "checkout endpoint returns 400." + +## Integration + +Validates real collaboration between two or more components this codebase owns, at the seam between them: a service against a real test database, a module against a real module it calls. +Never reaches the outside world. +Prefer real wiring or a fake over a mock at the boundary, per [mocking.md](mocking.md). +An integration scenario describes a cross-component contract: "the order API persists the order and a repository read returns it back." + +## E2E + +Drives the actual running application - the built artifact, not a test-harness shortcut - through a realistic end-user scenario, against the **mocked environment** of [mocking.md](mocking.md#the-e2e-environment). +It never runs against production or a shared staging environment; the point is a repeatable journey, not a live probe. +This is the only layer allowed to drive a browser. +It produces the HTML scenario report in [e2e-report.md](e2e-report.md): screenshots for frontend systems, before/after data-model state for everything else. +An e2e scenario describes user-observable, end-to-end behavior: "a guest can complete checkout with an expired coupon and sees the correct error." +Keep the count small and reserve it for critical journeys - each scenario is also a piece of evidence someone reads in the report. + +## Where tests.md and mocking.md apply + +[tests.md](tests.md)'s seams, behavior-over-implementation, tautology, and naming rules govern how you write a test at any layer. +[mocking.md](mocking.md) governs what may be replaced: architectural boundaries only at unit and integration, and the environment-only mocking that e2e is allowed. diff --git a/factory/skills/build/references/mocking.md b/factory/skills/build/references/mocking.md new file mode 100644 index 0000000..26af070 --- /dev/null +++ b/factory/skills/build/references/mocking.md @@ -0,0 +1,52 @@ +# Mocking Guidelines + +Mocks exist to make tests deterministic and fast, not to isolate every class from every other class. +Every mock is a place where the test stops verifying reality, so use as few as possible. + +## Mock only at architectural boundaries + +Replace a dependency only when it crosses a boundary you don't control in the test: + +- Network calls to third-party services (payment providers, external APIs) +- The system clock and randomness +- Anything genuinely slow or flaky in a test environment (email delivery, message queues) + +Everything inside the application - services, repositories wired to a test database, domain objects - should be real. +If the internal wiring is wrong, a test full of internal mocks will still pass, which makes it worse than no test. + +## Never mock internal collaborators + +If you feel the need to mock a class your own codebase owns, that is a design signal, not a testing problem. +Either test at a higher seam where the collaborator can just run, or the collaborator itself is a seam worth agreeing on with the user. + +The tell that a mock is wrong: the test asserts *that* a method was called (`toHaveBeenCalledWith`) instead of *what the outcome was*. +Interaction assertions couple the test to the implementation; state and output assertions couple it to behavior. + +## Prefer fakes over mocks at the boundary + +When you do replace a boundary dependency, prefer a small working fake over a per-test stub: + +- An in-memory repository instead of stubbing each query +- A fake clock you can advance instead of freezing time per test +- A recording fake for outbound calls (a fake mail sender with a `sent` list) instead of call-count assertions + +Fakes keep behavior consistent across tests and survive refactors; ad-hoc stubs encode one test's assumptions. + +## The e2e environment + +E2E has no mocks *inside* the application and no real world *outside* it. +The application under test is the real built artifact with its real internal wiring; everything it depends on beyond its own boundary is mocked, seeded, and deterministic: + +- Third-party HTTP replaced by a local stub server or recorded fixtures, with no credentials that could reach a real provider. +- A dedicated database or store, seeded to a known state before the run and torn down after. +- A fixed clock and seeded randomness, so the same scenario produces the same report twice. +- Outbound side effects (mail, payments, webhooks, queues) captured by a recording fake rather than delivered. + +Never point an e2e run at production or a shared staging environment. +A live run makes the scenario unrepeatable, its before/after state unreliable as evidence, and its side effects real. +If the mocked environment for a dependency doesn't exist yet, building it is part of the work, not a reason to fall back to the real one. + +## Determinism + +Inject the clock and randomness at the seam rather than patching globals. +A test that patches `Date.now` globally is fragile and can leak into other tests; a component that accepts a clock is testable by construction. diff --git a/factory/skills/build/references/parallel.md b/factory/skills/build/references/parallel.md new file mode 100644 index 0000000..d3b6a05 --- /dev/null +++ b/factory/skills/build/references/parallel.md @@ -0,0 +1,63 @@ +# Parallel Execution + +`scope` ordered the change plan so each change set builds only on the ones before it, and listed each change set's files. +This file is how to cash that in: you are the orchestrator, subagents are the implementers. + +## Waves + +Group the change sets into waves by consecutive-disjoint batching: + +- Walk change sets in numeric order; the spec author's ordering is the dependency order. +- Grow the current wave with the next change set only when its file list is disjoint from every change set already in the wave AND it consumes nothing a change set in the wave introduces (a module, function, endpoint, or decision outcome - your judgment while reading the spec). +- Any overlap or doubt excludes that change set from the wave; it anchors the next wave. +- A change set past an excluded one may still join the current wave, under a doubled condition: disjoint from, and consuming nothing introduced by, every change set in the wave AND every earlier change set not yet committed. Doubt excludes it - the skip-ahead is the same dependency proxy applied against everything still unfinished, so it cannot produce an order the sequential walk would forbid. + +Sequential-by-default means parallelism is a pure optimization that can never produce a wrong order. +Never start work belonging to the next wave while the current wave is in flight. + +## Launching a wave + +Launch one `Agent` per change set, **all in a single message** so they run concurrently. +A wave of one change set needs no subagent: implement it yourself in the main thread. + +Each agent prompt contains: + +1. The role: `You implement exactly one change set of a spec. Other agents implement sibling change sets concurrently; stay inside your change set's file list.` +2. The absolute path to `spec.md`, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The spec is a self-contained handoff by design; the agent reads all of it, then implements only its own change set. +3. The absolute path to this skill's `SKILL.md`, with the instruction to follow its "The change-set loop" and "Rules of the loop" sections - read like the other references, not pasted into the prompt. +4. Hard constraints: + - Implement only this change set's `[unit]` and `[integration]` scenarios. `[e2e]` scenarios are run once per spec by the orchestrator afterwards - do not launch the app. + - Do not commit, stage, or touch git state. The orchestrator commits. + - Do not edit files outside your change set's file list. If the change set genuinely needs a file another change set owns, stop and report it as a conflict instead of editing it. + - Do not edit `implementation-notes.md`. Report your entry; the orchestrator appends it. + - Run the spec's Validation block and report its real result. A red result is a fact to report, not something to hide or paper over. +5. The output contract below. + +Output contract (the agent's final message must be exactly one fenced JSON block): + +```json +{ + "changeSet": 3, + "status": "done | blocked", + "whatWasDone": "2-4 lines", + "seamsTested": ["..."], + "testsAdded": ["path::test name"], + "deviations": ["edge case found -> conservative choice made: what/why"], + "selfValidation": { "commands": ["..."], "result": "pass | fail", "output": "the tail that matters" }, + "conflicts": ["file another change set owns that this change set needed"] +} +``` + +## After a wave + +1. Verify rather than trust the reports: run the spec's Validation block yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. +2. Commit each change set's work on the current branch, in number order, one commit per change set. A skipped-ahead change set's commit waits until every earlier change set is committed, so history keeps the spec's order. +3. Append each change set's entry to `implementation-notes.md` from `whatWasDone`, `seamsTested`, `testsAdded`, and `deviations`. `testsAdded` becomes the entry's `Tests added:` line, which the scenario checker reads. + +A `status: blocked` change set, a failing validation, or a reported conflict is yours to finish in the main thread before the next wave starts - do not carry a red change set forward and do not relaunch the same agent on the same failure more than once. +If two change sets in a wave edited the same file anyway, reconcile it yourself and log it under Deviations. + +## When not to parallelize + +- The spec has one change set, or every change set's files overlap with the one before it: run them sequentially yourself. +- Two change sets batched into a wave visibly collide anyway: treat that as a `scope` sizing miss, run them sequentially, and note it in `implementation-notes.md`. diff --git a/factory/skills/build/references/tests.md b/factory/skills/build/references/tests.md new file mode 100644 index 0000000..78c7939 --- /dev/null +++ b/factory/skills/build/references/tests.md @@ -0,0 +1,91 @@ +# Test Examples + +Worked examples of the principles in SKILL.md: behavior over implementation, tests as specification, independent expected values. +See [layers.md](layers.md) for how a scenario's tag decides which layer its test lives at. + +## Seams + +A **seam** is the public boundary a test exercises, never internals. +Naming the seam before writing the test is what keeps testing effort on real interfaces instead of spreading over every private helper: "what is the public interface here, and which boundary would a caller actually cross?" + +The spec's scope section declares the public inputs and outputs of the change, so the seam for a scenario - the boundary its input crosses - is normally a decision you inherit, not one to reopen. +Reopen it only when the declared boundary is wrong or missing: pick the nearest real public boundary, use it, and log the change under Deviations in `implementation-notes.md`. +The signal that a seam is wrong is usually a mocking urge - see [mocking.md](mocking.md)'s "never mock internal collaborators." + +## Behavior test vs implementation-coupled test + +The seam here is the public `Cart` interface. + +Good - exercises the seam, asserts observable behavior: + +```ts +test("user can checkout with a valid cart", async () => { + const cart = new Cart(); + cart.add(product("book", 1200), 2); + + const receipt = await cart.checkout(validPayment()); + + expect(receipt.totalCents).toBe(2400); + expect(receipt.status).toBe("paid"); +}); +``` + +Bad - reaches inside and verifies internal wiring: + +```ts +test("checkout calls the tax calculator", async () => { + const taxCalc = jest.spyOn(cart["taxCalculator"], "compute"); + await cart.checkout(validPayment()); + expect(taxCalc).toHaveBeenCalledWith(2400); +}); +``` + +The bad test breaks if tax computation moves, is renamed, or gets inlined, even though checkout behavior is identical. +The good test survives all of those refactors. + +## Side-channel verification + +Bad - asserts through the database instead of the interface: + +```ts +await api.createUser({ name: "Ada" }); +const row = await db.query("SELECT * FROM users WHERE name = 'Ada'"); +expect(row.status).toBe("active"); +``` + +Good - observes the result the way a caller would: + +```ts +await api.createUser({ name: "Ada" }); +const user = await api.getUser("Ada"); +expect(user.status).toBe("active"); +``` + +If the schema changes but the API contract holds, only the bad test breaks. + +## Independent expected values + +Bad - tautological, recomputes the expectation the way the code does: + +```ts +expect(priceWithVat(100)).toBe(100 * 1.21); +``` + +Good - the expected value comes from a worked example or the spec: + +```ts +// Spec section 4.2: 100.00 EUR at 21% VAT is 121.00 EUR +expect(priceWithVat(100)).toBe(121); +``` + +If someone changes the VAT logic incorrectly, the tautological test still passes; the literal catches it. + +## Test names as specification + +Name tests after the capability, not the method under test. + +- Good: `"expired coupon is rejected at checkout"` +- Bad: `"test applyCoupon returns false"` + +A reader should be able to reconstruct what the system does from the test names alone. +Use the vocabulary the code and existing tests already use, so names match how the team talks about the feature. diff --git a/factory/skills/build/scripts/check-tests.py b/factory/skills/build/scripts/check-tests.py new file mode 100755 index 0000000..7d90d25 --- /dev/null +++ b/factory/skills/build/scripts/check-tests.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""Check that a spec's tests: scenarios really landed as tests. + +Usage: python3 check-tests.py .dev/{plan-name} [--repo-root .] +Exit 0 when every change set is accounted for, 1 with one problem per line +otherwise. Every test named in implementation-notes.md must exist on disk with +that name in it, so a claimed test that was never written cannot pass as done. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +NOTE_TESTS = re.compile(r"^\s*-\s*Tests added:\s*(.*)$", re.IGNORECASE) + + +def spec_scenarios(spec: Path, problem) -> dict[int, int]: + """Scenario count per change set, from the spec's change plan.""" + counts: dict[int, int] = {} + in_plan = False + current = None + for number, line in enumerate(spec.read_text(encoding="utf-8").splitlines(), start=1): + heading = HEADING.match(line) + if heading: + in_plan = heading.group(1).lower() == "change plan" + continue + if not in_plan: + continue + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + counts.setdefault(current, 0) + continue + tests = TESTS.match(line) + if not tests or current is None: + continue + value = tests.group(1).strip() + if value.lower().startswith("none"): + continue + counts[current] = len([s for s in value.split(";") if s.strip()]) + if not counts: + problem(f"{spec}: no change sets found under '## Change plan'") + return counts + + +def note_entries(notes: Path) -> dict[int, list[str]]: + """Tests named per change set, from implementation-notes.md.""" + entries: dict[int, list[str]] = {} + current = None + for line in notes.read_text(encoding="utf-8").splitlines(): + entry = NOTE_ENTRY.match(line) + if entry: + current = int(entry.group(1)) + entries.setdefault(current, []) + continue + named = NOTE_TESTS.match(line) + if not named or current is None: + continue + value = named.group(1).strip() + if value.lower().startswith("none"): + continue + entries[current].extend(t.strip() for t in value.split(",") if t.strip()) + return entries + + +def check_named_test(reference: str, repo_root: Path, change_set: int, problem) -> None: + """A named test must be path::name, and that name must be in that file.""" + if "::" not in reference: + problem(f"change set {change_set}: '{reference}' is not in path::test name form") + return + path, _, name = reference.partition("::") + target = repo_root / path.strip() + if not target.is_file(): + problem(f"change set {change_set}: {path.strip()} does not exist") + return + if name.strip() not in target.read_text(encoding="utf-8", errors="replace"): + problem( + f"change set {change_set}: {path.strip()} contains no test " + f"named '{name.strip()}'" + ) + + +def main(argv: list[str]) -> int: + args: list[str] = [] + repo_root = Path(".") + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--repo-root" and rest: + repo_root = Path(rest.pop(0)) + else: + args.append(value) + if len(args) != 1: + print("usage: check-tests.py [--repo-root .]", file=sys.stderr) + return 2 + + plan = Path(args[0]) + spec, notes = plan / "spec.md", plan / "implementation-notes.md" + for path in (spec, notes): + if not path.is_file(): + print(f"no {path.name} at {path}", file=sys.stderr) + return 2 + + problems: list[str] = [] + + def problem(message: str) -> None: + problems.append(message) + + counts = spec_scenarios(spec, problem) + entries = note_entries(notes) + + for change_set, scenarios in sorted(counts.items()): + if change_set not in entries: + problem(f"change set {change_set} has no entry in {notes}") + continue + named = entries[change_set] + if scenarios and len(named) < scenarios: + problem( + f"change set {change_set}: {scenarios} scenario(s) specced, " + f"{len(named)} test(s) named" + ) + for reference in named: + check_named_test(reference, repo_root, change_set, problem) + + for change_set in sorted(set(entries) - set(counts)): + problem(f"change set {change_set} is in {notes} but not in the spec's change plan") + + for message in problems: + print(message) + if problems: + print(f"\n{len(problems)} problem(s); the spec is not fully implemented.") + return 1 + print(f"{plan}: clean - {len(counts)} change sets, every named test found.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/factory/skills/build/templates/e2e-report.html b/factory/skills/build/templates/e2e-report.html new file mode 100644 index 0000000..5550d44 --- /dev/null +++ b/factory/skills/build/templates/e2e-report.html @@ -0,0 +1,248 @@ + + + + + +E2E Report + + + + +
+
+

E2E Report

+
+
+ +
+ +
+
+
+
+ + + + diff --git a/factory/skills/scope-review/SKILL.md b/factory/skills/scope-review/SKILL.md new file mode 100644 index 0000000..728dba1 --- /dev/null +++ b/factory/skills/scope-review/SKILL.md @@ -0,0 +1,132 @@ +--- +name: scope-review +description: Review and auto-refine a settled spec before build starts - a fresh-context agent panel checks the plan against the actual repo for infeasible change sets, missing failure paths, semantic contradictions, and untestable scenarios, then verified findings are applied to spec.md (and, where they touch settled decisions or cross-boundary invariants, to docs/decisions.md and docs/contracts.md) by refine agents and the panel re-runs; findings only the user can decide are asked as questions at the end and their answers applied and promoted the same way, so a finished run hands build a spec - and ledger - ready to implement, with no separate scope pass needed to promote the changes. Use after scope settles a spec and before build implements it. +disable-model-invocation: true +--- + +# Scope Review + +Review the spec with agents that did not write it, then refine it in place - before any implementation exists, and without a human in the loop. +A defect caught here costs a spec edit; the same defect after build costs a re-implementation, so this loop runs to completion on its own and ends with a spec build can start on - not a findings list to triage, and not a handoff back to `scope`. +The few findings only the user can decide are asked as questions at the end of the run, and the answers are applied before it finishes. +The panel judges the plan against the actual repo, not against the conversation that produced it. +You are the orchestrator: run tools, dispatch agents, apply the loop, report - your own reading of the spec is not a lens, and findings reach the spec only through verification. +This skill edits `spec.md`, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. + +`scope`'s own phase-5 reviewer hunts while the spec is still being drafted, from the spec file alone. +This skill is the standalone deeper pass: fresh agents with repo access, adversarial verification, and automatic refinement - worth running when the change is large or risky, or when build will run in a different session. + +Invoking this skill is the task - locate the spec yourself and start immediately; do not ask what to review. +First run `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py start scope-review` so the wrap-up can measure this run. + +## 1. Locate and gate + +- Locate the plan directory per [../../references/plan-layout.md](../../references/plan-layout.md) and read its `spec.md`; no spec: say so and stop - there is nothing to review. +- Run `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` once. If it reports anything, stop and recommend finishing the `scope` run: refinement here presumes a mechanically settled spec, and repairing an unfinished draft is `scope`'s job, not this loop's. +- Read the target repo's `docs/decisions.md`, `docs/contracts.md`, and `docs/dependencies.md` where they exist; their entries are premises the lenses cite. +- `spec-review_N.md` files present -> unresolved escalations from the highest-numbered one become verification items for round 1, and new reports continue the numbering. + +## 2. The review-refine loop + +Run up to two rounds; each round is panel -> verify -> refine. +A third panel means the refinements are churning, not converging - stop and escalate what remains. + +1. **Panel.** Run batch 1 (all four lenses) and batch 2 (verifiers) per the mechanics below. + From round 2 on, brief the lenses on what round 1 refined: their job is regressions in the refined material and their own unresolved findings, not a fresh full-spectrum hunt - a round that raises no BLOCK is convergence, and its CONCERNs go verified into the report rather than through another refine cycle. +2. **Aggregate.** Apply verdicts with the script below; REFUTED findings drop, a BLOCK survives only when CONFIRMED. +3. **Split.** Sort surviving findings into refinable and escalations per the authority rules below. +4. **Refine.** No refinable findings: exit the loop. Otherwise dispatch fresh-context refine agents - one per independent group of findings, launched in a single message - each given only its findings, the spec path, and the refinement rules below. They edit `spec.md` only. +5. **Re-gate.** Loop `lint-spec.py` until clean, fixing mechanical fallout with the same refine agents; then start the next round so fresh eyes judge the refined spec. + +Exit the loop when a panel raises nothing refinable; two rounds of the same finding surviving refinement is itself an escalation. +Escalations collected across the rounds go to the interview below, not to a handoff. + +## 3. Authority and refinement rules + +Refine agents resolve conflicts by this order - each level beats everything below it: + +1. The user's recorded intent: the spec's stated problem, scope, and interview outcomes. +2. The repo's reality: what the code actually contains beats what the spec claims about it. +3. A settled `✓` decision: a change set contradicting its linked decision is rewritten to match the decision, not the other way around. +4. The spec's prose. + +Refinable: false premises about the repo (rewrite the entry against the real code, including the extra work that reveals), change sets contradicting their linked decisions, missing test scenarios for stated invariants and failure paths, untestable scenarios (replace with one provable at that layer), and gaps whose resolution is forced once the repo is consulted. +Escalations - never auto-applied, queued for the interview instead: anything that would flip a `✓` decision to a rejected alternative, change the user-visible scope or behavior, add or drop a dependency, or contradict the user's recorded intent. +A refine agent that cannot fix its finding without crossing that line marks it escalated and leaves the spec alone. +Refinements follow `scope`'s notation: decision entries keep their slugs and marks, change sets keep their numbering, new scenarios carry layer tags. +A refinement that adds or rewrites a decision entry, or a cross-boundary invariant, is promoted to the ledger immediately per step 5 - it does not wait for the interview. + +## 4. Resolve escalations under factory policy + +After the loop, answer each escalation yourself under the unattended policy in [../../references/factory-run.md](../../references/factory-run.md): pick the recommended option when one is defensible, else the safest alternative, and record it in `auto_decided`. +Apply each answer immediately with a refine pass: update the decision entry's marks and because clauses, rewrite the affected change sets and `tests:` lines, keep `scope`'s notation, and loop `lint-spec.py` until clean. +An answer that resolves cleanly in place ends that escalation; verify the applied refinement yourself against the repo rather than re-running a panel for it. +When the answer settles or flips a decision, or changes a cross-boundary invariant, promote it to the ledger per step 5 as soon as the refine pass lands - do not wait for the run to finish. + +One outcome cannot resolve under policy: an escalation that invalidates the change's premise or opens a genuinely new effort - a new sub-effort with its own decision tree - exceeds what factory policy may decide. +Report the phase `failed` with reason `rescope`, naming what a future `scope` run must revisit; never guess at a new effort's shape. + +Verdict: `Verdict: APPROVED` always - a run that cannot settle every escalation under policy is reported `failed` with `rescope` instead of a partial verdict. + +## 5. Promote to the ledger + +Every settled decision entry and cross-boundary invariant that this run added or changed in `spec.md` - from refinement or from an answered escalation - gets promoted the same run, so build never has to wait on a separate `scope` pass for it. +Apply the promotion test in [../../references/decision-ledger.md](../../references/decision-ledger.md): copy qualifying decisions into `docs/decisions.md` verbatim, dated, sourced to this spec, evidence marks included, and promote recurring rationales to `P-` principles under the same bar. +Promote cross-boundary invariants to `docs/contracts.md` under the same test, phrased for the relying side - same as `scope` phase 8. +A decision or invariant that fails the promotion test, or a `? verify:` mark still open, stays in `spec.md` only and is not forced into the ledger. + +## 6. Panel mechanics + +Lenses live in [references/lenses.md](references/lenses.md): `feasibility`, `completeness`, `consistency`, `testability` - all four, every round. +Run the batches on the transport selected by [../ship/references/orchestration.md](../ship/references/orchestration.md), which owns transport choice, result-file delivery, and the batch mechanics, with these substitutions: + +- The artifact under review is `spec.md`, not a diff: hand each agent the spec path and the repo root instead of a diff file, and drop the diff-location line from the prompt contract. +- Lens definitions and shared rules come from this skill's [references/lenses.md](references/lenses.md). +- Verifiers refute findings against the spec file and the actual repo; `file`/`line` in a finding points into `spec.md` unless a repo path is named. +- Scratch root: `/tmp/scope-review-{session-id}/round-{R}/`. + +If no transport is available, stop and explain that this panel-based skill cannot preserve its verification contract. +Between the batches, number the findings the verifiers get, and after batch 2 aggregate: + +```bash +python3 {ship-skill-root}/scripts/aggregate-findings.py plan {batch-1 results} +python3 {ship-skill-root}/scripts/aggregate-findings.py aggregate {batch-1 results} {batch-2 results} --expected {lenses} +``` + +## 7. Write the report + +Write `.dev/{plan-name}/spec-review_N.md` at the next free index, one per run, covering all rounds: + +```markdown +# Spec review N - {plan-name} - {date} + +Verdict: APPROVED | APPROVED WITH DEFERRALS +Rounds: {R} - {finding counts per round} + +## Refinements applied +### R1 - {lens} - {one-line title} +{spec.md:line} - {what was wrong} -> {what the spec says now} + +## Escalations resolved +### E1 - {lens} - {one-line title} +Asked: {the question} - Answered: {the user's decision} -> {what the spec says now} + +## Promoted to the ledger +{decision slug or contract entry} -> `docs/decisions.md` | `docs/contracts.md` + +## Deferred +### D1 - {lens} - {one-line title} +{the question still open, and why it exceeded this run: premise invalidated, new effort, or unanswered} + +## Strengths +{the good notes worth keeping, deduplicated} +``` + +## Wrap up + +Open the chat summary with the table from `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py end scope-review --count findings_verified=N --count findings_refuted=N --count refinements_applied=N --count escalated=N`, pasted verbatim. +Then summarize in the same message: the verdict, what was refined and what factory policy decided (so the loop's edits stay auditable after the fact), what was promoted to `docs/decisions.md` and `docs/contracts.md`, and a link to the report. + +## Factory context + +Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. diff --git a/factory/skills/scope-review/references/lenses.md b/factory/skills/scope-review/references/lenses.md new file mode 100644 index 0000000..0e5fa5f --- /dev/null +++ b/factory/skills/scope-review/references/lenses.md @@ -0,0 +1,55 @@ +# Spec Review Lenses + +Each lens is one agent in the panel. +The artifact is `spec.md`; the repo is the evidence. +Every lens reads the whole spec but stays in its lane; prompt assembly and delivery follow ship's [orchestration.md](../../ship/references/orchestration.md) with the substitutions in this skill's SKILL.md. + +## feasibility + +Does the change plan survive contact with the repo? + +- Every file a change set names exists, or the set says it is new; the described edit is possible at that site - the function, hook, or config it assumes is really there. +- Prior art and idioms the spec cites exist where it says they do. +- Premises about current behavior are checked against the code, never trusted: an "X already handles Y" claim that is false is a BLOCK naming the site. +- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on. + +## completeness + +What will the implementer hit that the spec never mentions? + +- Failure paths and edge cases with no decision, no scope entry, and no test scenario. +- Second-order work the change implies but no change set carries: migrations, config, deploy steps, doc updates. +- The medium-decision trap: choices that look small but swing how much work is done, silently defaulted instead of argued. +- Call sites and consumers affected by the change but outside every change set's file list; Grep for them. + +## consistency + +The spec against itself and the project's ledgers, semantically - `lint-spec.py` already enforced the mechanics. + +- A change set that contradicts its linked `✓` decision, or quietly implements a `✗` rejected alternative. +- A scope invariant or error-handling entry that no change set enforces. +- A decision colliding with `docs/decisions.md` without acknowledging it, a plan that would break a `docs/contracts.md` guarantee while reliance sites still assume it, or a new dependency edge `docs/dependencies.md` forbids. + +## testability + +Will the `tests:` lines produce real proof? + +- Each scenario is provable at its tagged layer, and concrete enough that whoever writes it invents nothing - a vague scenario is a finding even when the feature is right. +- Scenarios cover the scope's invariants and failure paths, not just happy paths. +- A `tests: none - {reason}` whose reason does not hold is a finding. +- `[e2e]` scenarios are runnable in the mocked environment build uses; a scenario needing a live third party will fail there. + +## Shared rules (include in every lens prompt) + +Severity ladder: + +- **BLOCK**: implementation following this spec would fail or build the wrong thing; name the concrete site or scenario. +- **CONCERN**: a gap worth a conversation before build; include what you checked. +- **NIT**: minor polish. + +`file`/`line` point into `spec.md` unless a repo path is named. +Stay in your lane; other lenses cover the rest. +A finding based on a guess about unseen code is not a finding - read the repo first. +Every finding needs a location, a one-line title, and a 2-4 line detail naming the evidence. +Also report up to 5 genuinely good things your lens noticed. +Output contract: the same JSON as ship's lens agents, defined in [orchestration.md](../../ship/references/orchestration.md). diff --git a/factory/skills/scope/SKILL.md b/factory/skills/scope/SKILL.md new file mode 100644 index 0000000..c14c8c0 --- /dev/null +++ b/factory/skills/scope/SKILL.md @@ -0,0 +1,163 @@ +--- +name: scope +description: Spec a change by interviewing for the real problem, arguing every design decision against alternatives, and writing a self-contained spec with a change plan at .dev/{plan-name}/spec.md. Use to plan work before implementation, to turn accepted review findings into new change sets, or to audit the decisions already embedded in existing code or commit history. +disable-model-invocation: true +--- + +# Scope + +The spec is the deliverable: do not implement anything. +It is a fresh-context handoff - planning happens here, implementation happens in another session (often another model) that reads only the spec. + +The numbered sections are an addressing scheme, not a march. Four orderings are load-bearing: catalog decisions before arguing them, argue them before reading the project's ledger, settle the spec before promoting anything out of it, and never let the change plan outrun an open decision. The rest is judgment - work it in whatever order the change calls for. +Before the interview, run `python3 {scope-skill-root}/../../scripts/skill-metrics.py start scope` so the retrospective can measure the run. + +## 1. Interview + +Interview the user until you understand what they want, one consequential question at a time, using the host's structured user-input tool when available. Then write down what you understood. +Whenever you recommend an option - here, in the decision talk-through, or in follow-up questions - put a confidence score next to it (`Confidence: 70%`) with the one fact that would most change it. +Score how likely it is the right solution given the evidence, not how strongly you prefer it: a taste call, an unverified premise, or an option chosen without having read the relevant code scores low. +A recommendation at `Confidence: 75%` or above is not a question: apply it and keep going, and write `auto-applied at Confidence: NN%` into the chosen line's because clause so the ledger shows who decided. +Ask only below 75%, and, whatever the score, when the recommendation changes what was asked for: a different problem than the request names, a requirement dropped, or a new non-goal. +At the end of the interview, list every auto-applied recommendation once, in one block, so a single reply can overturn any of them before the spec is written down. + +The request often arrives one level too low: a solution ("add rate limiting with Redis") hides the problem it solves, a symptom hides the cause that picks the fix. +Climb up before interviewing about the change itself - what led to this, what they observed, what would look different if it worked - then pressure-test the premise: what data shows the problem is real, and does it point where they think? Committing to a fix before the cause means speccing the wrong change well. +Once the problem holds up, brainstorm alternatives at the problem level, including other layers entirely (rate limit at the ALB instead of in the app); the user's original ask becomes one alternative among them, and the whole question lands in the research section as the first decision. +If the premise doesn't survive, that resolves as a `⊘` not-doing line, not as a failed interview. + +The interview is latency-bound by the user, so hide machine work inside it: once the change's rough area is clear, launch background read-only subagents on phase 2's prior-art hunt - existing idioms, helpers, earlier attempts, the areas the change touches - so their results are waiting when the interview closes. +They start from `docs/architecture.md` ([../../references/architecture.md](../../references/architecture.md)), the high-level overview of the system; when the project has none, the same explorers run its initial capture first, so the spec is argued against an overview instead of a file tree. +These explorers never read `docs/decisions.md`: the ledger must stay out of context until the phase 5 reconcile, and phase 9 treats an early leak as contamination. + +Then size the change: **small** - a handful of decisions, an area the user knows, limited blast radius - or **full**, everything else. +Tell the user which you picked and why; they can override. Small changes run the lighter variants where marked below - the spec file, its argued decisions, scope, and change plan always happen. + +**Iterate small.** Bias toward the smallest spec that delivers observable behavior - a slice implemented and looked at teaches more than a bigger plan. +A big request becomes a first slice specced now, with later slices as `⊘` lines naming what reopens them; reopen the spec per "Starting from review findings" once build and ship land. The ledger carries the decisions across slices, so going small loses nothing argued. + +## 2. Catalog the decisions + +Look at the code, starting from the phase 1 explorers' results - prior art first: existing idioms, helpers, and earlier attempts - so decisions get argued against what the codebase already does, not from scratch. Identify all the design decisions involved: + +- **Big** - overall approach. +- **Medium** - things that seem simple but can cause big shifts in how much work is done (store a file on S3 or locally, retry strategy). The trickiest tier - make sure you catch all of them. +- **Small** - timeout lengths, shapes of data structures. + +Pay special attention to edge cases and error handling. How the change gets verified is a decision too: what level to test at, what needs a real dependency versus a fake, what can't be tested and why. + +**Blind spot pass** (full-size only): don't grade your own catalog - you'll reread it the way you wrote it. Get a second one from something that hasn't seen your reasoning, reliably a subagent handed only the user's original request and the relevant code - not the conversation, not your catalog. +Both of its inputs are settled when the interview ends, so launch it before you start cataloging; it hunts while you do, and the fold-in is the sync point. +Fold the diff in: what it found and you didn't are blind spots, what you found and it didn't deserves a second look. +Then tell the user what their framing didn't account for: constraints already in the code, behavior the change would break, second-order work, and what a mature solution handles in this domain that they wouldn't know to ask about. + +Between the catalog and the talk, gather each big and medium decision's evidence - call sites, existing constraints, what the code already does - in one parallel subagent batch, one agent per decision area, so each discussion starts armed instead of pausing to grep. +Talk through the big and medium decisions with the user, highest-impact first; when a decision's criteria are taste-driven, sketch the alternatives concretely instead of describing them in prose. + +## 3. Research section + +Pick a kebab-case `{plan-name}` naming the outcome and write `spec.md` in the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)), dated under its title, in three sections: `## Research`, `## Scope`, `## Change plan`. +Research comes first, one entry per decision, ordered by impact. +Every decision gets a stable ID: `D-` plus a short kebab slug (`D-file-storage`); keep a slug once assigned. + +Each entry: the decision phrased as a question, then one line per alternative in the mark and because-clause notation of [../../references/decision-ledger.md](../../references/decision-ledger.md), which owns it. +Evidence marks apply from phase 1 onward, so the premise pressure-test data lands here cited or marked. +A decision still being argued is marked `[open]`; one that resolves to not doing it gets `✗` on every alternative and a closing `⊘`. +`⚑` marks a line waiting on the user. It may stay open in research and scope, but the change plan never links a decision that is `[open]` or carries one. + +``` +D-input-validation: How should input be validated? + ✓ custom validation - rules fit in ~20 lines, no new dependency ⚠ we maintain edge cases ourselves + ✗ validation library - adds a dependency for one call site + +D-file-storage: Where do uploaded files live? [open] + ? S3 - survives redeploys, needs bucket + IAM work + ⚑ ask: expected file sizes and retention? + +D-response-caching: Should responses be cached? + ✗ Redis - operational burden for an unproven need + ⊘ not doing - no measured latency problem ? verify: p95 from prod metrics; reopen if p95 exceeds 500ms +``` + +When a later section references a decision, echo the resolution in parentheses - `D-file-storage (✓ S3)`, `(open)`, `(⊘ not doing)` - so the reader only jumps back for the why. + +## 4. Scope section + +Record the scope: inputs, outputs, invariants, and error handling. +An invariant that crosses a boundary - another module will rely on it without seeing the enforcing code - gets drafted in contract notation ([../../references/contracts.md](../../references/contracts.md)) so phase 8 can promote it. +A change that adds a module or dependency edge, or collides with `docs/dependencies.md`, is a decision with the alternatives named in [../../references/dependency-rules.md](../../references/dependency-rules.md); the chosen resolution edits that file inside a change set, never implicitly. +Name the components and flows the change adds, removes, or reshapes, in the overview's terms, so build knows what in `docs/architecture.md` its change sets must update. +Efforts have second-order effects - capture them as nested sub-efforts, each carrying its own decisions back into the research section (rate limiting in scope means Redis setup, which carries config and deploy decisions). +Record considered non-goals as `⊘` lines with a because clause - things someone weighed and cut, not mere omissions. +End the scope with a `### Validation` block listing the repo's real typecheck/test/lint/build commands, discovered from `package.json`, a `Makefile`, CI config, or equivalent - never guess `npm test` into a `pytest` repo; ask if you cannot determine them. +Writing style for the spec: ELI12, no similes or metaphors. + +## 5. Review and research + +1. Spawn a subagent reviewer with the spec file only - not the conversation, not your reasoning - before starting any of the work below, so it hunts while you research. Give it a hunting job, not a checklist: find the decisions this spec makes without realizing it, the alternatives rejected without a stated reason, and the chosen options whose downsides the spec doesn't admit. It succeeds by finding problems; "looks complete" is a failed review. For small changes, run this hunt yourself against the spec file. +2. While it runs, research how to implement everything: exact call sites, APIs, the idioms from the phase 2 prior-art hunt. Ask the user when you hit weird stuff. +3. Also while it runs, reconcile the drafted decisions against the project's `docs/decisions.md`, if it keeps one, per the recommender contract in [../../references/decision-ledger.md](../../references/decision-ledger.md). The decisions were argued fresh, so this diff means something; a `diverged` classification is raised with the user and marked `⚑` until resolved. +4. When the reviewer returns, ask remaining questions and update the doc. Anything still unanswerable stays a `⚑` line. + +## 6. Change plan + +Short fragmented sentences. Link decisions by ID wherever one applies, echoing the choice. +Each change set ends with one `;`-separated `tests:` line - concrete scenarios as input -> expected outcome, each tagged `[unit]`, `[integration]`, or `[e2e]`, covering happy path, edge cases, and failure paths; a set with nothing to test says `tests: none - {reason}`. +Specific enough that whoever writes the tests invents nothing; the author tags layers here because a fresh implementation session can't recover that intent. +Order change sets so each builds only on the ones before it; keep file lists disjoint where possible - `build` parallelizes consecutive change sets whose files don't overlap. + +``` +1. Change set 1 + a. file 1 - describe changes - decisions: D-input-validation (✓ custom) + b. file 2 - describe changes + tests: [unit] payload missing name -> 400 naming the field; [integration] valid payload -> 200 and row written; [e2e] 11MB upload -> rejected before the transfer starts + +2. Change set 2 + a. file 3 - describe changes + tests: none - deploy config only, verified by the deploy itself +``` + +Then loop `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` until it exits clean. +It owns the mechanics above; the spec is not final while it reports anything. + +## 7. Visualize + +Full-size changes only. Map the settled spec onto `SPEC_DATA` per [references/data-schema.md](references/data-schema.md) and render [templates/spec.html](templates/spec.html) to `/tmp/{project-slug}/reports/{plan-name}-spec.html`, opening and publishing per [../../references/reporting.md](../../references/reporting.md). +Once the spec is settled, this phase, the phase 8 promotion, and the Jira sync have no ordering between them - overlap them rather than running a march. + +## 8. Promote to the ledger + +Copy into `docs/decisions.md` every decision that passes the promotion test in [../../references/decision-ledger.md](../../references/decision-ledger.md). +Entries go in verbatim, dated, sourced to this spec, evidence marks included; a `? verify:` never gets silently dropped, and promotion is the cheapest moment to check what's checkable now. +Promote recurring rationales to `P-` principles per the ledger's bar. +Cross-boundary invariants from phase 4 promote to `docs/contracts.md` under the same test, phrased for the relying side. + +## 9. Run retrospective + +Audit the run itself. Three checks, each reported as a typed line - silence reads as "never ran": + +- **Contamination** - `clean | contaminated`. Did the ledger enter context before the phase 5 reconcile? If so, name the decisions drafted after exposure: their reconcile outcomes are inherited, and a `still-holds` on them proves nothing. +- **Sizing** - `held | mis-sized`. Did the small/full call survive? Name the evidence when it didn't. +- **Catalog gaps** - `none | gap`. What did the reviewer or blind-spot pass find that phase 2 should have caught? A missed *category* is a proposed edit to the phase 2 list - propose it to the user, never apply it silently. + +Then print the measured run metrics with `python3 {scope-skill-root}/../../scripts/skill-metrics.py end scope --count decisions=N --count change_sets=N --count scenarios=N`, pasting its table verbatim. + +## Factory context + +Read `request.md` in the plan directory as the request; the `run` skill already created the directory and named the plan, so use it rather than naming one. This phase keeps its interview - it is the one interactive phase and is exempt from `check_factory_unattended`'s scan - and still ends with one explicit go question. Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. + +## Jira sync + +Read `.dev/config.json`; when `jira.enabled` is true, follow [../../references/jira.md](../../references/jira.md) from before the first `acli` call - it owns the command shapes, the Initiative/Epic/Task timing, and the failure protocol. +With an absent or disabled config, no Jira behavior or mention. + +## Starting from review findings + +When the plan directory ([../../references/plan-layout.md](../../references/plan-layout.md)) holds `review_N.md` or `spec-review_N.md` files, the highest-numbered review's accepted Blockers and Concerns are the interview's opening agenda: each becomes an open decision, and rejected or deferred findings land as `⊘` lines so they are visibly not dropped. +Append new change sets to the existing `spec.md` with continued numbering - never renumber - and apply the review's Decision reconciliation section to `docs/decisions.md`. +A defect change set includes the review's triggering scenario as its reproduction. + +## Reverse mode + +When the user points at existing code instead of a planned change ("audit the decisions in the sync layer"), follow [references/reverse-mode.md](references/reverse-mode.md): inventory the decisions already embedded in the code, reconcile them against the ledger, and write ratified entries and unexamined defaults out - no spec file is produced. +To seed a ledger from commit history instead of code, follow [references/bootstrap.md](references/bootstrap.md). diff --git a/factory/skills/scope/references/bootstrap.md b/factory/skills/scope/references/bootstrap.md new file mode 100644 index 0000000..4a1b3d8 --- /dev/null +++ b/factory/skills/scope/references/bootstrap.md @@ -0,0 +1,72 @@ +# Ledger Bootstrap + +Build `docs/decisions.md` in one pass from commit history. +The commit skill's message format records `Why:` / `Considered:` / `Constraint:` / `Directive:` bodies and `Severity:` / `Risk:` trailers - in a repo that has used it for a while, the decisions are already written down, scattered across hundreds of commits. +Bootstrap lifts the durable ones into the ledger; after that, commit-time capture (the ledger-capture step of the commit skill's sync-checks reference) keeps it current and this never runs again. + +Run for the whole repo or scoped to an area the user names. + +## 1. Harvest + +Inventory candidate commits - catalog only, no qualifying yet: + +```bash +# Commits that argued an alternative - the strongest decision signal +git log --format='%H %s' --grep='Considered:' --all-match + +# Commits whose Why: defends a tradeoff (harvested in step 2's read) +git log --format='%H %s' --grep='Why:' + +# Trailer-carrying commits, for risk lines +git log --format='%H %s' --grep='Severity:\|Risk:' +``` + +Group the candidates by area using the commit scope slug (`feat(auth): ...` -> area `Auth`). +Scope slugs are the area vocabulary - don't invent a second taxonomy. + +## 2. Extract + +For each area, read the candidate commits' full bodies (`git show --format=fuller --no-patch `) and draft `D-` entries in ledger notation: + +- The decision as a question, from `What:` + `Why:` context. +- `✓` the chosen approach, because clause distilled from `Why:`. +- `✗` each rejected alternative from `Considered:`, with its stated reason. +- `⚠` any downside the body admits. +- Dated with the commit date, sourced to the hash. + +Qualification bar is the ledger-capture step's: a choice qualifies when someone could plausibly argue it differently later - `Considered:` exists, or the `Why:` defends a tradeoff. +Routine implementation narration doesn't qualify no matter how detailed. + +Evidence marks ([../../../references/decision-ledger.md](../../../references/decision-ledger.md) Notation): a claim the body makes from observation - "we saw timeouts in prod" - is a dated claim sourced to the hash; that's a citation, and its staleness is judged at read time. +Reserve `? verify:` for evidence the body asserts but nothing ever demonstrated. +Don't blanket-mark extracted entries - a ledger seeded from history where every clause carries a `?` has buried the mark. + +The same read harvests contracts ([../../../references/contracts.md](../../../references/contracts.md)): a body stating a boundary guarantee other code relies on - most often in `Directive:` or `Constraint:` fields ("callers must not retry", "never returns partial results") - drafts a `C-` entry for `docs/contracts.md`. +Before recording, confirm the guarantee against the current code and cite the enforcing function; a guarantee only history asserts gets `? verify:`. +History is a thin source for contracts - the boundaries themselves are the real one, and an audit walk seeds the registry better than commits do; harvest opportunistically here, don't sweep for them. + +For a large history, spawn one subagent per area in parallel, each returning drafted entries for its area. +Subagents get the commit bodies as their source - not the ledger, not each other's drafts. + +## 3. Reconcile + +Each drafted entry gets a typed disposition - report the counts per area: + +- `recorded` - enters the ledger. +- `superseded` - a later commit revisited the same decision; the latest resolution is recorded, this entry is folded into it (earlier `✗` alternatives are kept - they're the argument history). +- `stale` - the code the decision governed no longer exists (verify: the touched files or the chosen mechanism are gone). Not recorded; a decision about deleted code is trivia. +- `unqualified` - didn't pass the bar on a closer read. + +Dedupe by slug across areas before writing. + +## 4. Risk lines + +Mechanical, from the trailer harvest, per the area-header semantics in [../../../references/decision-ledger.md](../../../references/decision-ledger.md): domains are the all-time union of `Risk:` values per area (sticky facts); the level is the max `Severity:` across the **recent window only** - last 90 days of trailer commits - because levels observe recent change, they don't ratchet from history. +Date each line `(updated , bootstrap)`. + +## 5. Write and present + +Assemble `docs/decisions.md` from the ledger layout: principles (only if a rationale already recurs across 3+ extracted decisions - the bar doesn't lower for bootstrap), then area sections with risk lines and entries ordered by date. +Present the per-area summary - entries recorded, superseded, stale, unqualified - and let the user prune before committing the file. +Commit as `docs(decisions): bootstrap ledger from commit history`. +Harvested contracts assemble into `docs/contracts.md` the same way - presented for pruning with the rest, committed separately as `docs(contracts): bootstrap from commit history`. diff --git a/factory/skills/scope/references/data-schema.md b/factory/skills/scope/references/data-schema.md new file mode 100644 index 0000000..ee5aaf9 --- /dev/null +++ b/factory/skills/scope/references/data-schema.md @@ -0,0 +1,64 @@ +# SPEC_DATA Schema + +The shape to populate in `templates/spec.html` between the `SPEC_DATA_START` / `SPEC_DATA_END` markers. +Replace the whole object per the shared etiquette in [../../../references/reporting.md](../../../references/reporting.md). + +```js +const SPEC_DATA = { + title: "string - the spec's title", + planName: "string - the .dev/{plan-name} slug", + date: "string - YYYY-MM-DD, the spec header date", + summary: "string - one or two sentences: the problem and the chosen direction", + + // One entry per decision, in the research section's impact order. + decisions: [ + { + id: "string - the D- slug, e.g. D-file-storage", + question: "string - the decision phrased as a question", + status: "decided | open | not-doing", + alternatives: [ + { + mark: "chosen | rejected | open", // renders as ✓ / ✗ / ? + text: "string - the alternative", + because: "string - the because clause, evidence marks included", + downside: "string - the ⚠ clause; omit when none" + } + ], + notDoing: "string - the ⊘ line with its reopen condition; only when status is not-doing", + flags: ["string - each ⚑ line still waiting on the user"] // omit or [] when none + } + ], + + scope: { + overview: "string - the one-sentence overall task", + efforts: [ + { + title: "string - effort name", + description: "string - one sentence", + decisions: ["D-slug (✓ choice)", "..."], // echoes, omit when none + children: [ /* same shape, nested sub-efforts */ ] + } + ], + nonGoals: ["string - each ⊘ line with its because clause"], + invariants: ["string - inputs/outputs/invariants/error handling worth surfacing"], + validation: ["string - each real repo command from the Validation block"] + }, + + changeSets: [ + { + n: 1, // matches the change plan numbering + title: "string - one line", + items: [ { file: "string - path", change: "string - what happens there", decisions: ["D-slug (✓ choice)"] } ], + tests: [ { layer: "unit | integration | e2e | none", scenario: "string - input -> expected outcome, or the none-reason" } ] + } + ] +}; +``` + +## Filling it in honestly + +- **This object mirrors `spec.md`, section for section** - decisions, scope, and change sets come straight from the file you just wrote. A section left empty here reads as a gap in the spec. +- **Marks are the content** - the because clauses, ⚠ downsides, and ⊘ reopen conditions are why the artifact is worth opening; never flatten them into bare labels. +- **flags belong to open decisions only** - a decision the change plan links never carries one. Surface any that remain loudly rather than hiding them. +- **Don't invent a decision that wasn't argued** - a decision with one alternative and no because clause is a statement, not a decision; leave it out or argue it first. +- **tests entries are the acceptance criteria** - keep layer tags accurate; `build` and `ship` treat them as the enforceable spec. diff --git a/factory/skills/scope/references/reverse-mode.md b/factory/skills/scope/references/reverse-mode.md new file mode 100644 index 0000000..22bce4e --- /dev/null +++ b/factory/skills/scope/references/reverse-mode.md @@ -0,0 +1,20 @@ +# Reverse Mode: Recover Implicit Decisions + +When the user points at existing code instead of a planned change - "audit the decisions in the sync layer", "what did the AI decide here?" - the change-speccing phases invert into an inventory of decisions already made but never argued. + +There's no change to understand, so there is no interview. + +1. **Catalog in two passes.** + First pass: inventory every embedded decision without judging any - timeouts, retry counts, limits, page and buffer sizes, validation rules and their gaps, storage and serialization choices, concurrency and ordering assumptions, error-handling policies, defaults of any kind. + Include values that look fine; judging while cataloging skips them. + Second pass: for each entry, ask what problem the value was solving. + Record the answer - or `no known problem - unexamined default`. + That line is the discriminator: those decisions were never made by anyone and are up for grabs. +2. **Reconcile** against the project's decision ledger, if it keeps one - `docs/decisions.md`, format in [../../../references/decision-ledger.md](../../../references/decision-ledger.md). + Catalog entries already recorded -> verify the code still matches the recorded choice (a mismatch is `diverged` - raise it). + The rest are new. +3. **Talk through the new ones worth deciding**, ordered by blast radius if the value is wrong. + A decision the user ratifies gets `✓` with the real because clause. + One worth changing gets a recommended normal scope run; never launch it yourself. +4. **Write out.** + Reverse mode produces no spec file - everything lands in the ledger: ratified decisions and unexamined defaults not worth deciding, recorded with their `no known problem` line if they pass the promotion test so the next audit doesn't re-litigate them. diff --git a/factory/skills/scope/scripts/lint-spec.py b/factory/skills/scope/scripts/lint-spec.py new file mode 100755 index 0000000..2056f78 --- /dev/null +++ b/factory/skills/scope/scripts/lint-spec.py @@ -0,0 +1,208 @@ +#!/usr/bin/env python3 +"""Check a spec.md against the mechanical rules of the scope skill. + +Usage: python3 lint-spec.py .dev/{plan-name}/spec.md +Exit 0 when the spec is clean, 1 with one problem per line otherwise. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +SLUG = re.compile(r"^(D-[a-z0-9]+(?:-[a-z0-9]+)*):\s*(.+)$") +BAD_SLUG = re.compile(r"^(D-\S+):") +ALTERNATIVE = re.compile(r"^\s+([✓✗?⊘])\s+(.*)$") +FLAG = re.compile(r"^\s+⚑") +ECHO = re.compile(r"(D-[a-z0-9-]+)\s*\(([^)]*)\)") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) +LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") +HEADING = re.compile(r"^##\s+(.*?)\s*$") + +CHOSEN, REJECTED, OPEN, NOT_DOING = "✓", "✗", "?", "⊘" + + +def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: + """Split numbered lines into research / scope / change plan buckets.""" + found: dict[str, list[tuple[int, str]]] = {} + current = None + for number, line in enumerate(lines, start=1): + heading = HEADING.match(line) + if heading: + name = heading.group(1).lower() + current = name if name in ("research", "scope", "change plan") else None + if current: + found.setdefault(current, []) + continue + if current: + found[current].append((number, line)) + return found + + +def read_decisions(research: list[tuple[int, str]], problem) -> dict[str, dict]: + decisions: dict[str, dict] = {} + current = None + for number, line in research: + header = SLUG.match(line) + if not header and BAD_SLUG.match(line): + problem(number, f"{BAD_SLUG.match(line).group(1)} is not a kebab-case D- slug") + current = None + continue + if header: + slug, question = header.group(1), header.group(2) + if slug in decisions: + problem(number, f"{slug} is defined twice; slugs are unique and stable") + current = decisions.setdefault( + slug, {"line": number, "open": "[open]" in question, "marks": [], "flagged": False} + ) + continue + if current is None: + continue + if FLAG.match(line): + current["flagged"] = True + continue + alternative = ALTERNATIVE.match(line) + if not alternative: + continue + mark, text = alternative.group(1), alternative.group(2) + current["marks"].append(mark) + if " - " not in text: + problem(number, "alternative has no ' - because' clause") + if mark == NOT_DOING: + current["not_doing"] = True + if "reopen" not in text.lower(): + problem(number, "not-doing line states no condition that would reopen it") + if mark == CHOSEN: + current["choice"] = text.split(" - ")[0].strip().lower() + return decisions + + +def check_decisions(decisions: dict[str, dict], problem) -> None: + for slug, decision in decisions.items(): + chosen = decision["marks"].count(CHOSEN) + # An [open] decision is still being argued and may not have its alternatives yet. + if not decision["open"] and len(decision["marks"]) < 2: + problem(decision["line"], f"{slug} argues fewer than two alternatives") + if decision.get("not_doing"): + if chosen: + problem(decision["line"], f"{slug} is not-doing yet marks an alternative chosen") + elif decision["open"]: + if chosen: + problem(decision["line"], f"{slug} is [open] yet marks an alternative chosen") + if OPEN not in decision["marks"]: + problem(decision["line"], f"{slug} is [open] with no ? alternative") + elif chosen != 1: + problem(decision["line"], f"{slug} has {chosen} chosen alternatives; expected exactly one") + + +def check_echoes( + body: list[tuple[int, str]], decisions: dict[str, dict], in_change_plan: bool, problem +) -> None: + for number, line in body: + for slug, echo in ECHO.findall(line): + decision = decisions.get(slug) + if decision is None: + problem(number, f"{slug} is echoed but never argued in the research section") + continue + resolution = echo.strip().lower() + if decision.get("not_doing"): + expected = resolution.startswith(NOT_DOING) + elif decision["open"]: + expected = resolution == "open" + elif resolution.startswith(CHOSEN): + shorthand = resolution[1:].strip() + expected = not shorthand or shorthand in decision.get("choice", "") + else: + expected = False + if not expected: + problem(number, f"{slug} echo '({echo})' does not match its resolution") + if in_change_plan and (decision["open"] or decision["flagged"]): + problem(number, f"the change plan links {slug}, which is still open or flagged") + + +def check_change_plan(body: list[tuple[int, str]], problem) -> None: + counts: dict[int, int] = {} + current = None + for number, line in body: + if "⚑" in line: + problem(number, "the change plan carries a ⚑; resolve it before the plan is final") + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + counts.setdefault(current, 0) + continue + tests = TESTS.match(line) + if not tests: + continue + if current is None: + problem(number, "tests: line outside any change set") + continue + counts[current] += 1 + value = tests.group(1).strip() + if value.lower().startswith("none"): + if " - " not in value: + problem(number, "'tests: none' states no reason") + continue + if not value: + problem(number, "tests: line is empty") + continue + for scenario in value.split(";"): + scenario = scenario.strip() + if scenario and not LAYER.match(scenario): + problem(number, f"scenario '{scenario[:40]}' carries no [unit]/[integration]/[e2e] tag") + for change_set, seen in sorted(counts.items()): + if seen != 1: + problem(None, f"change set {change_set} has {seen} tests: lines; expected exactly one") + + +def check_validation(lines: list[str], problem) -> None: + for index, line in enumerate(lines): + if line.strip().lower() == "### validation": + if any(rest.strip() and not rest.startswith("#") for rest in lines[index + 1 : index + 12]): + return + problem(index + 1, "the Validation block lists no commands") + return + problem(None, "no '### Validation' block in the scope section") + + +def main(argv: list[str]) -> int: + if len(argv) != 2: + print("usage: lint-spec.py ", file=sys.stderr) + return 2 + path = Path(argv[1]) + if not path.is_file(): + print(f"no spec at {path}", file=sys.stderr) + return 2 + lines = path.read_text(encoding="utf-8").splitlines() + + problems: set[tuple[int, str]] = set() + + def problem(number: int | None, message: str) -> None: + where = f"{path}:{number}" if number else str(path) + problems.add((number or 0, f"{where}: {message}")) + + found = sections(lines) + for name in ("research", "scope", "change plan"): + if name not in found: + problem(None, f"no '## {name.title()}' section") + + decisions = read_decisions(found.get("research", []), problem) + check_decisions(decisions, problem) + check_echoes(found.get("scope", []), decisions, False, problem) + check_echoes(found.get("change plan", []), decisions, True, problem) + check_change_plan(found.get("change plan", []), problem) + check_validation(lines, problem) + + for _, message in sorted(problems): + print(message) + if problems: + print(f"\n{len(problems)} problem(s); the spec is not final.") + return 1 + print(f"{path}: clean - {len(decisions)} decisions argued.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/factory/skills/scope/templates/spec.html b/factory/skills/scope/templates/spec.html new file mode 100644 index 0000000..75329a0 --- /dev/null +++ b/factory/skills/scope/templates/spec.html @@ -0,0 +1,151 @@ + + + + + +Decision Spec + + + +
+ + + diff --git a/factory/skills/ship/SKILL.md b/factory/skills/ship/SKILL.md new file mode 100644 index 0000000..afe8cf2 --- /dev/null +++ b/factory/skills/ship/SKILL.md @@ -0,0 +1,143 @@ +--- +name: ship +description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every check passes, phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. +disable-model-invocation: true +--- + +# Ship + +Ship a change in three phases: a deterministic gauntlet that fixes mechanics, a review panel that judges meaning, then a pull request that carries proof the change works. +Prompted quality rules soften into guidelines as a context grows; a checker's exit code does not, and judgment belongs to a verified panel, not to one context. +You are the orchestrator for both phases: run tools, dispatch agents, aggregate, report. +Never weaken a check to make it pass, and never stand in for the panel - your own reading of the code is not a lens. + +Invoking this skill is the task - detect the diff yourself and start immediately; do not ask what to ship. +First run `python3 {ship-skill-root}/../../scripts/skill-metrics.py start ship` so the wrap-up can measure this run instead of recalling it. + +## Phase selection + +Default is phase 1 (gauntlet), phase 2 (review), then phase 3 (pull request): the panel reviews the diff as it stands after the gauntlet's fixes, so its findings are about meaning, not mechanics already settled, and the PR carries both verdicts. +On request, run a single phase: "gauntlet only" (or "harden only") runs phase 1 alone; "review only" runs phase 2 alone - the right mode when mutating fixes are unwanted, such as on someone else's PR. "No PR" or "local only" runs the default flow without phase 3. +The phases differ in contract: phase 1 mutates the repo (fix agents edit code, tools get committed, accepted thresholds land in `docs/decisions.md`); phase 2 is strictly read-only and never edits code or the ledger, but a BLOCK verdict hands the diff to the remediation loop below, which mutates like phase 1; phase 3 commits, pushes, and opens or updates the PR, never force-pushing. + +## Scope + +Both phases share one scope. + +- PR number given -> target that PR; else the host's PR tooling, when it has any, for an open PR; else the local branch against the default branch. +- Gather the diff per the diff-scope rules in [../../references/plan-layout.md](../../references/plan-layout.md): local git only, standard exclusions. +- Locate the plan directory by the same reference's convention and read its `spec.md`; degrade gracefully without one. +- Full-repo runs only when the user asks - they are expensive, and the loops are the same. +- No reviewable files: report `failed` with that reason and stop. Diff is tiny (1-2 files) or huge (>25k lines): proceed and note it in `auto_decided` rather than spending extra agents confirming. + +## Phase 1: the gauntlet + +Eight checks, cheapest first - the five read-only analyzers scan as one parallel batch, the three test-contending ones run sequentially, per the loop reference; [references/tools.md](references/tools.md) owns their definitions and the acquisition ladder. +Never weaken or skip a check because acquiring its tool is work. + +1. **Project static analysis** - every linter, type checker, and format checker the repo already configures, run over the in-scope files. +2. **Security scan** - secrets, vulnerable dependencies, and static security rules; a found secret escalates immediately, it is never quiet fix-agent work. +3. **Dead code** - symbols the diff added that nothing references, and symbols it orphaned by removing their last caller. +4. **Duplication** - token-level clones the diff introduced against the rest of the repo. +5. **Dependency rules** against `docs/dependencies.md` ([../../references/dependency-rules.md](../../references/dependency-rules.md) owns the checker semantics; absent file: skip with a clear note, never invent rules). +6. **Coverage-weighted complexity** per in-scope function. +7. **Flakiness** - the tests the diff added or touched, repeated and shuffled until trusted. +8. **Mutation testing** over the in-scope source. + +Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). +When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. +Before phase 2, always run the required pull-request commands per +[../../references/ci-parity.md](../../references/ci-parity.md), even when no +fix agent fired. Feature E2E is not a substitute for a repository screenshot, +packaging, or report job that CI requires. + +## Phase 2: the review panel + +Review the diff with a panel of concern-focused agents, verify every finding against the repo, and write the report as a plan artifact. + +### 1. Load intent + +The review checks the diff against what was specced, not just general quality. + +- Write the diff to a scratch file; agents read it from there, not inline. +- Read the target repo's `docs/contracts.md` first, if it exists ([../../references/contracts.md](../../references/contracts.md)): its boundary guarantees are premises. Excerpt the entries whose guarantee or reliance sites the diff touches into the brief handed to every lens; a diff that breaks a cited guarantee while reliance sites still assume it is finding material. +- From the spec read in Scope: the scope section for invariants, boundaries, and error handling, the change plan for per-set files and layer-tagged `tests:` scenarios, and the research section for the decision rationale used to judge deviations. +- Also read `.dev/{plan-name}/implementation-notes.md` if present (deviations logged during implementation) and the latest `/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`'s data block, per the reading rules in [../../references/reporting.md](../../references/reporting.md). Both are additional evidence for the `spec-conformance` and `tests` lenses, not just the diff itself. +- No spec: derive `{plan-name}` from the branch name and fall back to the PR body and commits for intent. +- Build a short brief: what the change does, what's on the critical path, what was specced. + +### 2. Check for previous reviews + +`review_N.md` files present -> this is a re-review: extract unresolved findings from the highest-numbered one as verification items (fixed or not?), and the new report gets the next index. + +### 3. Select the panel + +Read [references/lenses.md](references/lenses.md); select only lenses with surface in this diff (a docs-only change skips performance). +Include `spec-conformance` whenever a spec was found, and always include `simplify` - every diff has simplification surface. +At most one diff-specific custom lens (migrations, concurrency, i18n) when clearly warranted, defined in the same shape as the built-ins. +Tell the user which lenses you selected and why before launching. + +### 4. Run the review panel + +Run the two batches - all lens agents first, then all verifiers - on the transport selected by [references/orchestration.md](references/orchestration.md), which owns transport choice, result-file delivery, and the prompt contracts. +If no transport is available, stop and explain that this panel-based phase cannot preserve its verification contract. +Between the batches, number the findings the verifiers get: + +```bash +python3 {ship-skill-root}/scripts/aggregate-findings.py plan {batch-1 results} +``` + +Every BLOCK and CONCERN it lists is adversarially verified - the verifier's only job is to refute it against the actual repo. + +### 5. Aggregate + +The same script applies the verdicts, so these rules live in one place instead of softening across a long context: + +```bash +python3 {ship-skill-root}/scripts/aggregate-findings.py aggregate {batch-1 results} {batch-2 results} --expected {lenses you selected} +``` + +It drops REFUTED findings, keeps a BLOCK only when CONFIRMED and carries every other surviving finding as a CONCERN, merges duplicates by file:line while crediting each lens that found them, names the lenses that failed or returned nothing parseable, and sets the verdict. +A failed lens doesn't abort the review - the report names it, so the verdict is honestly partial. + +Then, with findings verified, read the target repo's `docs/decisions.md` if it keeps one, and classify each colliding finding per the recommender contract in [../../references/decision-ledger.md](../../references/decision-ledger.md). Read-only - this phase never edits the ledger; classifications land in the report's Decision reconciliation section, and ledger writes belong to the follow-up `scope` run. + +### 6. Write the report + +Write `.dev/{plan-name}/review_N.md` in the shape of [references/report-format.md](references/report-format.md), at the next free index. + +### 7. Publish the report + +Map the report onto `REVIEW_DATA` per [references/data-schema.md](references/data-schema.md) and render [templates/review.html](templates/review.html) to `/tmp/{project-slug}/reports/review_N.html`, opening and publishing per [../../references/reporting.md](../../references/reporting.md) (stable review favicon; title names the plan and review number). + +## Blocker remediation + +A BLOCK verdict is work before it is a question: run up to two rounds per [references/remediation.md](references/remediation.md), each a fresh-context fix pass over the confirmed blockers, a re-harden of the touched files, and a re-review at the next index. +PASS or CONCERNS ends the loop; a blocker still standing after round 2, or one an agent escalated as needing a spec or decision change, is recorded as a decision and stays in the PR body per [references/pull-request.md](references/pull-request.md), never a question. +Skipped in review-only mode. + +## Phase 3: the pull request + +Open the PR automatically, with evidence a reviewer can see before reading the diff, per [references/pull-request.md](references/pull-request.md). + +1. Commit what the gauntlet left uncommitted, branch off the default branch if still on it, and push. +2. Build the Evidence section from the e2e report with `python3 {ship-skill-root}/scripts/pr-evidence.py extract`, publishing frontend screenshots to the `pr-evidence` branch with its `publish` command; without an e2e report, capture the evidence now per the reference - screenshots for a UI, a labeled before/after pair otherwise, and a red-on-base, green-on-branch reproducing test for every bug fix. +3. Write `.dev/{plan-name}/pr.md` in the reference's body shape and loop `pr-evidence.py check` on it until it passes; the check, not your judgment, decides whether the proof is real enough. +4. Create the PR (draft when a real blocker survived remediation) or update the one that already exists, then follow its required checks to green per [../../references/ci-parity.md](../../references/ci-parity.md). + +## Wrap up + +Summarize whichever phases ran in one chat message, opening with the measured run metrics: + +```bash +python3 {ship-skill-root}/../../scripts/skill-metrics.py end ship --count violations_found=N --count violations_fixed=N --count violations_surviving=N --count findings_verified=N --count findings_refuted=N --count remediation_rounds=N --count blockers_cleared=N --count evidence_items=N +``` + +Pass only counters you tallied from tool output and the aggregate script; the table it prints (time, tokens, agents, tool calls, git delta, trend against earlier runs) is pasted verbatim, never retyped. +For the gauntlet, per tool: violations found, fixed, and surviving (with the recorded decision each survivor is waiting on); name the tools acquired or built this run and where they live; state the scope honestly - "hardened the diff" is not "hardened the repo". +For the review: the final verdict and top findings, linking every `review_N.md` this run wrote, the local HTML report, and the published URL when one was requested and created; per remediation round, which blockers cleared and which survived. +For the pull request: its URL, draft or ready, what the Evidence section shows and where it came from, and the state of its required checks. + +## Factory context + +A found secret is a `stopped` result with kind `secret.found`, naming the commit to purge; a destructive action outside the run branch is `stopped` with kind `action.destructive`. Neither is retried or overridden. `gh pr create` refusing with "none of the git remotes configured for this repository point to a known GitHub host" is reported `failed` with that text as the reason and the pushed branch plus `pr.md` in `artifacts[]` - never `stopped`, never retried as an auth fault. `pr_url` and `draft` go in the result's fields when a pull request opens. Read `factory-run.json` and the result envelope in [../../references/factory-run.md](../../references/factory-run.md), then write `.dev/{plan}/results/{phase}-{attempt}.json` as the last action. Never name or launch the next phase. diff --git a/factory/skills/ship/references/data-schema.md b/factory/skills/ship/references/data-schema.md new file mode 100644 index 0000000..e1f7bac --- /dev/null +++ b/factory/skills/ship/references/data-schema.md @@ -0,0 +1,46 @@ +# REVIEW_DATA Schema + +The shape to populate in `templates/review.html` between the `REVIEW_DATA_START` / `REVIEW_DATA_END` markers, per the shared etiquette in [../../../references/reporting.md](../../../references/reporting.md). +This mirrors `review_N.md`'s structure exactly - it's the same content, published as the artifact a PR reviewer actually opens. + +```js +const REVIEW_DATA = { + title: "string - Review {N}: {title}", + planName: "string - the .dev/{plan-name} slug, or empty if no spec was found", + reviewIndex: 1, + verdict: "PASS" | "CONCERNS" | "BLOCK", + panel: ["lens", "..."], // lenses run + failedLenses: ["lens", "..."], // omit or [] if none failed + base: "string - branch or PR reference", + date: "string - ISO date", + summary: "string - 1-2 sentence summary, markdown-lite", + + specConformance: "string - the spec-conformance summary, markdown-lite; omit the section entirely (set to '') if no spec was found", + + previousFindings: [ + { finding: "string", status: "fixed" | "still open" } + ], // [] unless this is a re-review + + blockers: [ + { lenses: ["lens", "..."], file: "string", line: 0, title: "string", detail: "string - markdown-lite, the triggering scenario as confirmed by verification" } + ], + concerns: [ + { lenses: ["lens", "..."], file: "string", line: 0, title: "string", detail: "string" } + ], + nits: [ + { file: "string", lens: "string", issue: "string" } + ], + + whatsGood: [ + { lens: "string", note: "string" } + ], + + nextStep: "string - what to fix first and why, markdown-lite" +}; +``` + +## Filling it in honestly + +- This is a direct transcription of `review_N.md` into data, not a separate editorial pass - the two must agree. +- `blockers`/`concerns` only include findings that survived adversarial verification (CONFIRMED, or PLAUSIBLE demoted to CONCERN) - never a REFUTED finding. +- `previousFindings` is only non-empty on a re-review; omit the section in the template when empty rather than showing "none." diff --git a/factory/skills/ship/references/gauntlet.md b/factory/skills/ship/references/gauntlet.md new file mode 100644 index 0000000..1e7270b --- /dev/null +++ b/factory/skills/ship/references/gauntlet.md @@ -0,0 +1,56 @@ +# The Gauntlet Loop + +How phase 1 of `ship` runs its eight checks to completion. +The check definitions and their acquisition ladder live in [tools.md](tools.md); this file owns the loop, the fix agents, and the thresholds. + +## The loop + +The five read-only analyzers - static analysis, security, dead code, duplication, dependency rules - scan as one parallel batch: their initial runs mutate nothing, so collect all five violation lists concurrently, dispatch fix agents grouped by independent area across the combined lists, and re-run all five together until clean. +The expensive three - complexity x coverage, flakiness, mutation - stay sequential, cheapest first: they contend for the test runner, and each one's input shifts with every fix the previous one landed. + +For each tool (the batched five count as one), in order: + +1. Run it; collect the violations. +2. Dispatch fixes: one fresh-context agent per independent area, launched in a single message, each given only the violation list for its area, the relevant file paths, and the fix vocabulary below. +3. Re-run the tool until clean, then run the spec's Validation block (or the repo's test suite) to prove the fixes broke nothing; skip that run when the tool dispatched no fixes. +4. A violation that resists two fix rounds on the same root cause, or that the change seems to legitimately require, is recorded as a decision and left in place: it is either a real defect, a threshold worth changing, or a rule the spec should have amended, and the choice made is noted in `auto_decided` with the reason. + +## Exit: refresh the e2e evidence + +Fix agents edit code, so once any tool dispatched fixes, the e2e report build wrote no longer describes the code phase 2 will judge. +Before leaving phase 1, re-run the spec's `[e2e]` scenarios exactly as build does - same mocked environment, data mapping, and template, all owned by [../../build/SKILL.md](../../build/SKILL.md) and its references - overwriting `/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`. +A scenario that fails here is a violation like any other: dispatch fixes and loop under the same two-round rule. +Never write new e2e scenarios in this step; generation belongs to build, this step only re-executes. +Skip it when no tool dispatched fixes (the report is still fresh), when there is no spec or no e2e suite to run (note that in the wrap-up), or in review-only mode, which never reaches phase 1. + +## Exit: prove CI parity + +After the e2e refresh decision, run the required pull-request commands using +[../../../references/ci-parity.md](../../../references/ci-parity.md). This gate +always runs in the default two-phase flow and gauntlet-only mode, even when no +fix agent changed code. + +A red required check is a violation. "Pre-existing" requires merge-base proof, +recorded as a decision with that proof; it is never a note that permits a +PR-ready verdict. When a fix changes behavior or test orchestration, re-run +the affected command and continue until the full CI-parity set is green. + +## Fix vocabulary + +- Resolve a static analysis finding by fixing the code it points at, never by suppressing it inline or loosening the tool's config; a finding worth suppressing is recorded as a decision instead. +- Resolve a vulnerable dependency by upgrading it; an upgrade that breaks the build is recorded as a decision. A found secret is always an immediate `stopped` result with kind `secret.found`, never a quiet fix. +- Delete dead code outright; never comment it out or exclude it from the detector. +- Collapse a clone by extracting one shared helper or calling the one that already exists. +- Cut a complexity-coverage score by splitting the function or covering its paths. +- Resolve a dependency violation by inverting the dependency, inserting an interface, or splitting the module. +- De-flake a test by removing its nondeterminism (time, ordering, shared state, network), never by adding retries, sleeps, or looser assertions. +- Kill a mutant by adding the test that catches it. + +Fix agents never edit thresholds, rules files, or the tools themselves, and never delete a test to make a mutant moot. + +## Thresholds are decisions + +Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. +Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. +When the user accepts a different threshold, record it in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. +Never adjust a threshold silently to make a run pass. diff --git a/factory/skills/ship/references/lenses.md b/factory/skills/ship/references/lenses.md new file mode 100644 index 0000000..bb74f3e --- /dev/null +++ b/factory/skills/ship/references/lenses.md @@ -0,0 +1,102 @@ +# Review Lenses + +Each lens is one agent in the panel. +Select only lenses with surface in the diff; every lens sees the whole diff but stays in its lane. +Prompt assembly and delivery live in [orchestration.md](orchestration.md); this file owns the lens definitions and the shared rules. + +## spec-conformance + +Include whenever a spec directory was found. +Checks the diff against `spec.md` - its scope, change plan, and decisions - not against general quality: + +- Does each in-scope change set's `tests:` scenario exist as a real test, **at the layer it was tagged with** (`[unit|integration|e2e]`)? A scenario met by a test at the wrong layer (a unit test standing in for an e2e scenario) is a finding, not a pass. +- Do the scope section's invariants and error-handling entries hold in the diff? A broken invariant is a BLOCK. +- Does the diff implement each change set's `✓` decisions, without smuggling in a `✗` rejected alternative? A `⚠` accepted downside that manifests is expected, not a finding - the spec already admits it. +- Is a `docs/contracts.md` guarantee cited in the brief broken by the diff while its reliance sites still assume it? That is a BLOCK naming who gets hurt. +- Is every `[e2e]`-tagged scenario backed by a passing entry in `{plan-name}-e2e-report.html`? New user-facing behavior with no matching e2e scenario, or a scenario whose `status` is `fail`, is a BLOCK. +- Check `implementation-notes.md` for logged Deviations: does the conservative choice still satisfy the spec's scope and decisions, or does it need a call-out to the reviewer? An unresolved deviation that changes user-facing behavior from what the spec promised is a CONCERN at minimum. +- If a change set named a README or user-facing doc file, is that change present in the diff? A promised doc update that never happened is a CONCERN. +- Does the diff do significant work no change set describes? Scope creep is a CONCERN, not a crime; name it so the reviewer can decide. + +## correctness + +Bugs the compiler and tests would miss. +Walk the hardest paths first: concurrency and ordering hazards, boundary conditions (empty, one, max, off-by-one), null/absent paths, mutation while iterating, broken invariants, state-machine holes, unsafe retries, error paths, time and timezone handling, resource lifecycle, type-system escapes. +For each suspect path: state the invariant, name the input that breaks it, trace it through the code. +A BLOCK must name the triggering input; "this could race" without a scenario is a CONCERN. + +## security + +Authn/authz gaps, injection, secrets in code or logs, unsafe deserialization, crypto misuse, path traversal, SSRF, supply-chain risk in new dependencies, overly broad permissions. +Rate exploitability in this codebase, not theoretical severity. + +## architecture + +Module boundaries, layering, dependency direction, abstraction quality, coupling introduced by the diff. +Judge coherence against the patterns already established in the surrounding code; deviation from an established convention is a finding, personal style preference is not. +When the repo keeps `docs/dependencies.md`, treat its edges as settled: a conforming diff needs no boundary debate, and a diff that edits the rules file is reviewed as the decision it is, not as incidental churn. +Read `docs/architecture.md` first, when the repo keeps one, for orientation on where things live and how they talk ([../../../references/architecture.md](../../../references/architecture.md)): a diff that reshapes a component, flow, or boundary the overview describes without updating it is a finding. + +## tests + +Test quality per this plugin's philosophy, not raw coverage: + +- Tests must live at public seams; tests that mock internal collaborators, test private methods, or verify through side channels are implementation-coupled findings. +- Tautological tests (assertion recomputes the expected value the way the code does) are findings; expected values need an independent source of truth. +- New behavior in the diff without a test at its seam is a finding. (Missing e2e backing for user-facing behavior belongs to spec-conformance, not here.) +- Brittle patterns: global time/random patching, order-dependent tests, interaction assertions where a state assertion would do. + +## simplify + +Always include this lens; every diff has simplification surface. +The gauntlet's duplication, dead code, and complexity checkers are its deterministic front line, so spend this lens on what they cannot see: semantic duplication below the token threshold, generality nothing asked for, indirection that earns nothing. + +Unnecessary complexity that a simpler version would avoid, with behavior held constant: + +- Reinvented helpers: logic the codebase or standard library already provides; Grep for the existing helper before flagging and name it in the finding. +- Duplication introduced by the diff: the same logic now living in two places that will silently diverge. +- Speculative generality: abstractions, parameters, or config with exactly one caller and no planned second one. +- Needless indirection: layers that only forward calls, wrappers that wrap nothing. +- Wrong altitude: low-level mechanics inlined into high-level flow (or vice versa) that a small extraction would clarify. + +Every finding must sketch the simpler alternative concretely enough that the verifier can check it preserves behavior; "this feels complex" is not a finding. +Mostly CONCERN and NIT; reserve BLOCK for duplicating an existing tested helper. +This lens is about making the new code smaller and clearer, not about bugs (correctness), boundaries (architecture), or deleting unused code (dead-code). + +## performance + +Algorithmic complexity on real data sizes, N+1 queries, hot-path allocations, blocking I/O on async paths, missing pagination or streaming for unbounded data, cache invalidation. +Only flag what is on a path that plausibly matters; micro-optimizations are NITs at best. + +## api-design + +Naming, type signatures, error returns, consistency with the codebase's existing API idioms, backwards compatibility of anything already public. + +## errors-observability + +Swallowed errors, catch-and-continue without logging, missing context in log lines at decision points, errors that will be undebuggable at 3 AM, retry loops without backoff or limits. + +## docs + +Comments and docs touched or needed by the diff: stale or now-misleading comments, missing WHY comments on non-obvious constraints, public API docstrings. +Doc updates promised by the spec belong to spec-conformance, not here. + +## dead-code + +The gauntlet's detector already caught provably unreferenced symbols; this lens hunts what needs reading, not reference counting. +Unreachable branches, commented-out code, leftover debug scaffolding, vestigial parameters, dead flags and config keys, code only its own tests still call. +Verify with Grep across the whole repo before flagging; check tests, dynamic string-keyed lookups, and external API surface. +If you cannot prove it dead, report at most a CONCERN and say what you could not rule out. + +## Shared rules (include in every lens prompt) + +Severity ladder: + +- **BLOCK**: a concrete problem you can name with the scenario that triggers it; would misbehave in production or violates the spec. +- **CONCERN**: a path you cannot convince yourself is safe, or a gap worth a conversation; include what you checked. +- **NIT**: minor polish; skip NITs entirely on diffs over 15 files. + +Stay in your lane; other lenses cover the rest. +A finding based on a guess about unseen code is not a finding. +Every finding needs file, line, a one-line title, and a 2-4 line detail naming the evidence. +Also report up to 5 genuinely good things your lens noticed. diff --git a/factory/skills/ship/references/orchestration-heavy.md b/factory/skills/ship/references/orchestration-heavy.md new file mode 100644 index 0000000..3fa9a76 --- /dev/null +++ b/factory/skills/ship/references/orchestration-heavy.md @@ -0,0 +1,108 @@ +# Heavy Mode: Workflow Tool (explicit opt-in only) + +Use this only when the user has explicitly asked for a workflow or an exhaustive run; it triggers the dynamic-workflow confirmation dialog. +It buys schema-validated outputs, pipelining (findings verify while other lenses still review), and `/workflows` progress. +Pass `diffFile`, `brief`, `planContext`, `lenses` (as `{key, prompt}` with the full assembled lens prompt), and `priorFindings` via `args`. + +```js +export const meta = { + name: 'ship', + description: 'Concern-based diff review with adversarial verification of findings', + phases: [ + { title: 'Review', detail: 'one agent per selected lens' }, + { title: 'Verify', detail: 'refute-or-confirm each finding' }, + { title: 'Recheck', detail: 'prior findings fixed?' }, + ], +} + +const FINDINGS = { + type: 'object', + required: ['verdict', 'findings', 'good'], + properties: { + verdict: { enum: ['PASS', 'CONCERNS', 'BLOCK'] }, + findings: { + type: 'array', + items: { + type: 'object', + required: ['severity', 'file', 'line', 'title', 'detail'], + properties: { + severity: { enum: ['BLOCK', 'CONCERN', 'NIT'] }, + file: { type: 'string' }, + line: { type: 'integer' }, + title: { type: 'string' }, + detail: { type: 'string' }, + scenario: { type: 'string' }, + }, + }, + }, + good: { type: 'array', items: { type: 'string' }, maxItems: 5 }, + }, +} + +const VERDICT = { + type: 'object', + required: ['status', 'reason'], + properties: { + status: { enum: ['CONFIRMED', 'PLAUSIBLE', 'REFUTED'] }, + reason: { type: 'string' }, + }, +} + +const RECHECK = { + type: 'object', + required: ['fixed', 'evidence'], + properties: { + fixed: { type: 'boolean' }, + evidence: { type: 'string' }, + }, +} + +const { diffFile, brief, planContext, lenses, priorFindings } = args + +const lensPrompt = (l) => [ + `You are the "${l.key}" voice on a code-review panel; other lenses cover other concerns.`, + l.prompt, + `Brief: ${brief}`, + planContext ? `Plan context:\n${planContext}` : '', + `Read the full diff from ${diffFile}. Read surrounding source files when the diff is not enough.`, + `Return findings for your lens only.`, +].filter(Boolean).join('\n\n') + +const verifyPrompt = (f, l) => [ + `Adversarially verify this code-review finding from the "${l.key}" lens. Your job is to REFUTE it.`, + JSON.stringify(f), + `From the diff at ${diffFile}, read the hunks for the finding's file (search by path - do not ingest the whole diff), plus the actual files in the repo. Check whether the claimed problem is real:`, + `does the scenario actually trigger, is the code path reachable, does something else already handle it?`, + `CONFIRMED = you reproduced the reasoning against real code and it holds.`, + `PLAUSIBLE = you could not refute it, but could not fully confirm the scenario either.`, + `REFUTED = the finding is wrong, unreachable, or already handled; say exactly why.`, +].join('\n') + +const rechecks = priorFindings.length + ? parallel(priorFindings.map((f) => () => + agent( + `A previous review reported:\n${JSON.stringify(f)}\nRead the current code and the diff at ${diffFile}. Is it fixed now? Cite the fixing change or what is still missing.`, + { label: `recheck:${f.file}`, phase: 'Recheck', schema: RECHECK }, + ).then((r) => r && { ...f, recheck: r }))) + : Promise.resolve([]) + +const results = await pipeline( + lenses, + (l) => agent(lensPrompt(l), { label: `review:${l.key}`, phase: 'Review', schema: FINDINGS }), + (review, l) => { + if (!review) return null + const toVerify = review.findings.filter((f) => f.severity !== 'NIT') + return parallel(toVerify.map((f) => () => + agent(verifyPrompt(f, l), { label: `verify:${f.file}:${f.line}`, phase: 'Verify', schema: VERDICT }) + .then((v) => v && { ...f, lens: l.key, verification: v }))) + .then((verified) => ({ + lens: l.key, + good: review.good, + nits: review.findings.filter((f) => f.severity === 'NIT').map((f) => ({ ...f, lens: l.key })), + findings: verified.filter(Boolean), + })) + }, +) + +return { lenses: results.filter(Boolean), rechecks: (await rechecks).filter(Boolean) } +``` diff --git a/factory/skills/ship/references/orchestration.md b/factory/skills/ship/references/orchestration.md new file mode 100644 index 0000000..57cc2ec --- /dev/null +++ b/factory/skills/ship/references/orchestration.md @@ -0,0 +1,150 @@ +# Orchestration + +The review always uses independent agents in two batches. Select the transport +without changing the panel: + +Both transports use the same scratch root, such as `/tmp/ship-{session-id}/`, +with a `results/` directory per batch. Result files in that directory are the +only channel for agent output. A batch in flight means the run is not over: +never end your turn while launched agents are still outstanding - wait for the +host's completion signal, read the results, and continue. Never retrieve results by messaging an agent, +waiting for relayed replies, or re-spawning an agent that already finished; +a fresh spawn has none of the original context and its output must not be used. + +1. **Native transport (preferred):** when the host provides a multi-agent + facility, launch all agents in a batch in one native parallel call. This is + the existing Claude Code, Codex, and opencode behavior; do not route those + hosts through subprocesses. +2. **Pi transport:** when `PI_CODING_AGENT=true` and `pi` is available, write + one prompt file per agent and run + [scripts/run-pi-agents.sh](../scripts/run-pi-agents.sh). The runner launches + one isolated `pi --print` process per prompt concurrently and inherits the + current provider, model, and reasoning level. +3. **Unavailable:** if neither transport exists, stop. The orchestrator must + not impersonate the panel or review the code itself. + +No dynamic workflows are involved by default. A Workflow-tool variant for +explicitly requested heavyweight runs only lives in +[orchestration-heavy.md](orchestration-heavy.md); do not read it otherwise. + +## Native transport + +Use the host's plain subagent tool exactly as provided (the Agent tool on +Claude Code, `spawn_agent` on Codex, the `task` tool on opencode). Launch all +agents of a batch in a single parallel call: on opencode that means issuing +every `task` call of the batch in one message with the default general +subagent. Native subagents receive the prompt contracts below directly. + +Result delivery is file-based, because on some hosts subagents run in the +background and their final message never reaches the orchestrator: + +1. Before launching a batch, create `{scratch-root}/{batch}/results/`. +2. Each agent's prompt instructs it to write its JSON report to + `{scratch-root}/{batch}/results/{safe-agent-name}.json` as its last action, + then end with a one-line confirmation naming that path. The result file is + the deliverable; the final message is not. +3. When the host signals that the batch's agents have finished, read the + result files. +4. A missing or unparseable file after one re-read marks only that agent as + failed, same as a crashed agent. + +Agents stay read-only in the repository; their only write is their own result +file under the scratch root. + +## Pi transport + +Use the shared scratch root. For each batch: + +1. Create `prompts/` and `results/` directories dedicated to that batch. +2. Write one `{safe-agent-name}.md` prompt per agent. Include the complete + prompt contract from the relevant batch below. Tell the process to inspect + the repository but never modify it. +3. From the repository root, run: + + ```bash + bash {ship-skill-root}/scripts/run-pi-agents.sh \ + {batch-root}/prompts \ + {batch-root}/results + ``` + +4. Read each `{safe-agent-name}.out` as that agent's final response. A matching + `.err` or missing/unparseable `.out` marks only that agent as failed. +5. Keep the scratch files until aggregation is complete so malformed output is + auditable, then remove them. + +The runner disables child skills, extensions, prompt templates, and sessions +to prevent recursion and state leakage. It retains project context files and +read-oriented tools (`read`, `bash`, `grep`, `find`, `ls`) so each process can +inspect the diff, surrounding code, and tests independently. Do not add +`write` or `edit`. + +## Batch 1: lens agents + +Launch one agent per selected lens, all in a single message so they run concurrently. +Each lens prompt is assembled from: + +1. A role line: `You are the "{key}" voice on a code-review panel; other lenses cover other concerns.` +2. The lens definition plus the shared rules from `lenses.md`. +3. The brief, and the spec context when a spec was found. +4. The diff location: `Read the full diff from {diffFile}; the repo working tree holds the post-diff state. Read repo files whenever the diff alone is not enough to judge.` +5. The output contract below. + +Also state: `Do not modify repository files. This is a read-only review.` +On the native transport, add the result-file instruction from the transport +section; on Pi, the runner captures the final message instead, which must be +exactly one fenced JSON block. + +Output contract (the JSON every lens agent must produce): + +```json +{ + "verdict": "PASS | CONCERNS | BLOCK", + "findings": [ + { + "severity": "BLOCK | CONCERN | NIT", + "file": "path/relative/to/repo", + "line": 1, + "title": "one line", + "detail": "2-4 lines naming the evidence", + "scenario": "what triggers it (required for BLOCK)" + } + ], + "good": ["up to 5 bullets"] +} +``` + +If an agent errors or returns something unparseable, record it as a failed lens and continue. +Never retry a failed lens more than once. + +## Batch 2: verifiers + +`scripts/aggregate-findings.py plan` collects every BLOCK and CONCERN from the +lens results and numbers them (`f1`, `f2`, ...) so verdicts map back. NITs skip +verification. Duplicates are not merged yet: two lenses that flagged the same +line get refuted independently, and the merge happens after their verdicts. +Launch one verifier per finding, again all in a single message. +If that would exceed 10 verifiers, group the findings by file into at most 10 +shards and give each verifier its shard; each finding is still verified +independently, one verdict object per finding. +Verifier prompt: + +``` +Adversarially verify each of these code-review findings, reported by the +named lens. Your job is to REFUTE them. +{findings as JSON, each with its id} +From the diff at {diffFile}, read the hunks for your findings' files (search +by path - do not ingest the whole diff), plus the actual files in the repo. +Check whether each claimed problem is real: does the scenario actually +trigger, is the claim accurate against the files, does something else +already handle it? +Produce exactly one JSON array with one object per finding: +[{"id": "f1", "status": "CONFIRMED | PLAUSIBLE | REFUTED", "reason": "..."}] +CONFIRMED = you reproduced the reasoning against real files and it holds. +PLAUSIBLE = you could not refute it, but could not fully confirm it either. +REFUTED = wrong, unreachable, or already handled; say exactly why. +``` + +Delivery follows the transport: on native, the verifier writes the array to +its result file; on Pi, it replies with the array as one fenced JSON block. + +Aggregating the verified results is `aggregate-findings.py aggregate` per the skill's Aggregate step - never another agent. diff --git a/factory/skills/ship/references/pull-request.md b/factory/skills/ship/references/pull-request.md new file mode 100644 index 0000000..9c0b6ea --- /dev/null +++ b/factory/skills/ship/references/pull-request.md @@ -0,0 +1,75 @@ +# The Pull Request + +How phase 3 of `ship` turns a hardened, reviewed branch into an open pull request that carries its own proof. +A PR without evidence asks the reviewer to trust the description; this phase makes the change visibly work before anyone reads the diff. + +## When it runs + +Phase 3 runs in the default flow, after the review report is written and the remediation loop in [remediation.md](remediation.md) has run its course: PASS and CONCERNS open a PR ready for review, a real blocker - one that survived both remediation rounds - opens a draft PR with the blockers listed first, so the work is preserved and CI runs while the recorded decision stands. +It is skipped in "gauntlet only" and "review only" runs and when the user says "no PR" or "local only"; a review-only run on someone else's PR never pushes anything. +Never force-push, never rebase, and never touch a branch other than the work branch and the evidence branch. + +## Commit and push + +1. Fixes the gauntlet left uncommitted are committed now, one commit per tool, in `commit`'s `type(scope): subject` plus What/Why shape. + Name files explicitly; never `git add -A`, never a `Co-Authored-By` line. +2. On the default branch, create the work branch first: `{plan-name}`, or `{EPIC-KEY}-{plan-name}` when [../../../references/jira.md](../../../references/jira.md) is enabled. +3. `git push -u {remote} {branch}`; a rejected push is reported and stops the phase. + +## Evidence + +The Evidence section is the PR's proof of the feature or fix, captured from a real run, never composed from memory. +`scripts/pr-evidence.py check` is the gate: loop on its output until it passes before creating or updating the PR. + +- **An e2e report exists** (`/tmp/{project-slug}/reports/{plan-name}-e2e-report.html`, refreshed by phase 1 when fixes landed): run `pr-evidence.py extract` on it. + Frontend reports yield one screenshot per meaningful step; `pr-evidence.py publish` pushes the PNGs to the `pr-evidence` branch so they render inline, because data URIs do not render on a PR. + Non-frontend reports yield labeled Before/After state per step, plus captured output. +- **No e2e report** (no spec, no e2e suite): capture the evidence in this phase, the same way build would. + A system with a UI: launch it in the mocked environment via the `run` skill or the repo's documented command, drive the changed behavior with browser automation, save one screenshot per meaningful step under the evidence directory, then publish them. + Anything else: the observable effect before and after, as a labeled pair of fenced blocks - a CLI transcript, an API response, a table row, a rendered file. +- **A bug fix** always adds the reproducing test as a labeled pair: `**On merge base**` shows it failing in an isolated worktree at `git merge-base HEAD origin/{default}`, `**On this branch**` shows the same test passing. + Red on base is what proves the bug was real; green on the branch is what proves it is gone. +- A screenshot proves the UI; when the real effect is a data change the screen does not show, add the Before/After pair for it as well. +- Never fabricate, reuse a screenshot from another run, or paste a data URI; a scenario that failed stays marked failed in the evidence. + +Evidence files live under `/tmp/{project-slug}/pr-evidence/`; the `pr-evidence` branch is an orphan branch that only ever receives images, one commit per run. +On a non-GitHub remote pass `--url-template` with wherever the images are hosted. + +## Body + +Write the body to `.dev/{plan-name}/pr.md` and hand it to the PR tool with `--body-file`; the file stays as the record of what was proposed. + +```markdown +## Summary + +{What changed and why, 2-4 sentences from the spec's research and scope sections.} +Plan `.dev/{plan-name}` | Review {N}: {verdict} | Jira {EPIC-KEY when enabled} + +## Evidence + +{pr-evidence.py extract output, or the hand-captured section per the rules above} + +## Quality + +| Check | Found | Fixed | Surviving | +| ----- | ----- | ----- | --------- | +{one row per gauntlet tool} + +Review panel: {lenses}; {blockers} blockers, {concerns} concerns ({review_N.md path}). +CI parity: {each reproducible required command and its result}; remote-only: {checks verified by the PR itself, or "none"}. + +## Open calls + +{Each surviving recorded decision from the gauntlet, each real blocker with its two-round history, and each concern from the review, one line each, or "none".} +``` + +Title: `{EPIC-KEY} ` prefix when Jira is enabled, then the change in imperative mood, under 70 characters. + +## Create or update + +- No PR for the branch: `gh pr create --title ... --body-file .dev/{plan-name}/pr.md` (`--draft` on BLOCK), base = the default branch. +- A PR already exists (build opened it on request, or a re-review): `gh pr edit --body-file ...`; `gh pr ready` when a BLOCK verdict has cleared, never the reverse. +- No `gh` and no equivalent host CLI: push, keep `pr.md`, print the compare URL, and say the PR must be opened by hand. + +Then follow the PR's required checks to a terminal state per [../../../references/ci-parity.md](../../../references/ci-parity.md). +Opening the PR is not the end of the phase; every required check green, or a recorded decision naming why one cannot be, is. diff --git a/factory/skills/ship/references/remediation.md b/factory/skills/ship/references/remediation.md new file mode 100644 index 0000000..36f8c47 --- /dev/null +++ b/factory/skills/ship/references/remediation.md @@ -0,0 +1,42 @@ +# Blocker Remediation + +How `ship` handles a BLOCK verdict from the review panel: two autonomous rounds of fix and re-review before a blocker is real. +A confirmed blocker is a defect with a triggering scenario, and a defect with a triggering scenario is work, not a question - the same reasoning that lets the gauntlet loop its fix agents. +Only a blocker that survives two rounds on the same root cause is recorded as a real blocker. + +## When it runs + +After phase 2 writes `review_N.md` with verdict BLOCK, in the default flow. +Never in "review only" mode: that mode exists for diffs whose owner did not ask for mutations. +CONCERNS and PASS skip straight to phase 3; concerns stay in the report and the PR's Open calls, they never trigger a round. + +## A round + +1. **Fix.** Take every Blocker from the current `review_N.md` - each carries a `file:line`, a title, and the triggering scenario verification confirmed. + Group blockers by independent area and dispatch one fresh-context fix agent per group, launched in a single message. + Each agent gets only its blockers, the relevant file paths, the brief phase 2 built (spec excerpts, contracts, decision rationale), and the fix vocabulary below. + Agents commit their fix on the work branch in `commit`'s message shape, naming the review index and finding in the Why. +2. **Re-harden.** Run phase 1's five read-only analyzers over the files the round touched, plus the flakiness check over the tests it added; dispatch fixes and loop per [gauntlet.md](gauntlet.md) as usual. + Then run the spec's Validation block (or the repo's test suite) and re-run the `[e2e]` scenarios when the round touched anything an e2e scenario exercises, overwriting the e2e report. +3. **Re-review.** Run phase 2 again as a re-review: `review_{N+1}.md`, same lenses, previous findings carried as verification items so the panel states whether each blocker is fixed or still open. + The summary line names the round: `Remediation round {r} of 2`. +4. **Decide.** PASS or CONCERNS ends the loop and phase 3 opens a ready PR. + BLOCK after round 1 starts round 2. + BLOCK after round 2 ends the loop: every surviving blocker is a real blocker. + +Round budget is two per ship run, not per blocker: a blocker the fixes introduced counts against the same budget, and a re-review that finds a new blocker after round 2 is a real blocker too. +A re-run of `ship` after a recorded real blocker starts a fresh budget. + +## Fix vocabulary + +- Fix the root cause the scenario names, never the symptom the lens observed; a fix that only makes the triggering scenario pass is a suppression. +- Add the test that reproduces the triggering scenario at its right layer, seen red before the fix and green after; a blocker with no test guarding it is not fixed. +- Stay inside the blocker's area: no refactors, no touching findings another agent owns, no concerns or nits. +- Never delete or weaken a test, a check, or a contract to make the blocker moot; never edit `docs/decisions.md` or `docs/contracts.md`. +- A blocker that cannot be fixed without changing specced behavior, a settled decision, or a contract the reliance sites still assume is marked `escalated` in the agent's result and left alone; escalations go straight to the real-blocker list without spending the second round. + +## Real blockers + +A real blocker is listed in the PR under Open calls with its history: the finding, the round-1 and round-2 attempts (commit and what each changed), and why it still stands - unfixable in scope, escalated, or a new blocker the fixes introduced. +Phase 3 opens the PR as a draft. +The wrap-up presents each real blocker as the recorded decision it is: a real defect the fixes could not reach, a decision worth reopening, or a spec the change outgrew - the same three outcomes the gauntlet's recorded decisions have. diff --git a/factory/skills/ship/references/report-format.md b/factory/skills/ship/references/report-format.md new file mode 100644 index 0000000..2f990bf --- /dev/null +++ b/factory/skills/ship/references/report-format.md @@ -0,0 +1,50 @@ +# Review Report Format + +The shape of `.dev/{plan-name}/review_N.md`, written by phase 2 at the next free index starting at 1. +`REVIEW_DATA` ([data-schema.md](data-schema.md)) mirrors this file section for section; the aggregator's JSON fills both. + +```markdown +# Review {N}: {title} + +Verdict: {PASS | CONCERNS | BLOCK} +Panel: {lenses run}, {failed lenses if any} +Base: {branch or PR}, {date} + +{1-2 sentence summary} + +## Spec conformance + +Tests: scenarios met/not met, invariants held, seams tested/untested. Omit if no spec. + +## Decision reconciliation + +{only when the repo keeps docs/decisions.md} Each colliding finding: still-holds (with the checked reopen condition), reopened, or diverged. + +## Previous findings + +{re-review only} Each finding from review_{N-1}: fixed or still open. + +## Blockers + +[{lenses}] {file:line} - {title} +{The scenario that triggers it, confirmed by verification.} + +## Concerns + +Same shape as blockers. A finding the verifier could not confirm belongs here, not in Blockers. + +## Nits + +| File | Lens | Issue | + +## What's good + +One line per lens; omit empty ones. + +## Next step + +What to fix first and why. +``` + +`scope` reads this file when the user accepts findings that need real work: its Blockers and Concerns open that run's interview, and its Decision reconciliation section is what gets applied to `docs/decisions.md`. +Write it for that reader - a finding with no triggering scenario cannot become a decision. diff --git a/factory/skills/ship/references/tools.md b/factory/skills/ship/references/tools.md new file mode 100644 index 0000000..64985db --- /dev/null +++ b/factory/skills/ship/references/tools.md @@ -0,0 +1,79 @@ +# Acquiring the Tools + +The gauntlet needs eight deterministic tools. +The acquisition ladder, per tool, is: the repo's existing tooling, else the ecosystem's established tool, else a small repo-fitted script an agent writes. +Never hand-roll what the ecosystem already maintains, and never download a generic harness wholesale - fit the tool to this repo, commit it, and reuse it on every later run. + +## Where tools live + +Repo-fitted scripts and configs go under `tools/harden/` in the consuming repository, committed with a README line saying what each does and how to invoke it. +The first run pays the acquisition cost; every later run - and any other agent - reuses them. +Check `tools/harden/` before acquiring anything. +Acquisitions are independent of each other: when several tools are missing, acquire them in parallel - one agent per tool, launched in a single message - rather than building them one after another. + +## 1. Project static analysis + +Run everything the repo already configures, not a subset you pick: linters, type checkers, format checkers, and any other analyzer wired into the project. +Discover the full set from where the repo declares it - package scripts, Makefile or task-runner targets, CI workflow steps, pre-commit hooks, and tool config files at the root (`.eslintrc*`, `pyproject.toml` tool sections, `clippy.toml`, and the like). +A configured tool that CI runs and the gauntlet skips is a check silently weakened; when unsure whether something counts, run it. +Prefer each tool's changed-files or per-path mode to scope to the diff; pre-existing findings elsewhere are noted, not fixed, unless the run is full-repo. +If the repo configures nothing, wire up the ecosystem's standard linter and type checker with their default configs as part of this step, committed like any other acquired tool. + +## 2. Security scan + +Three sub-scans, each zero-threshold: + +- **Secrets**: gitleaks or the ecosystem equivalent over the in-scope files. A found secret is never fix-agent work - stop the gauntlet and escalate immediately, because it needs rotation and possibly history rewriting, both human calls. +- **Dependency vulnerabilities**: the ecosystem's audit tool (osv-scanner, npm audit, pip-audit, cargo audit) over every manifest the diff touched. +- **Static security rules**: the repo's own SAST config if one exists, else semgrep with the ecosystem's default ruleset, scoped to in-scope files. + +## 3. Dead code detector + +The ecosystem's detector - knip or ts-prune (JS/TS), vulture (Python), staticcheck's unused checks (Go) - else a repo-fitted script that greps each added or touched export for references. +Scope: symbols the diff added that nothing references, plus symbols the diff orphaned by removing their last caller. +Declare entry points and deliberate public API surface as exclusions in committed config; a public export is not dead merely because the repo itself never calls it. + +## 4. Duplication detector + +jscpd or PMD CPD, comparing the in-scope files against the whole repo and reporting only clones that a diff-touched file participates in. +The threshold is token-based (default 50) so trivial similarity does not fire; where the line sits is a recorded decision like any other threshold. +Pre-existing clones between untouched files are noted, not fixed, unless the run is full-repo. + +## 5. Dependency checker + +Almost always a repo-fitted script: parse the `rules` block of `docs/dependencies.md` (format in [../../../references/dependency-rules.md](../../../references/dependency-rules.md)), glob-match the in-scope files to modules, extract static imports with the language's own tooling (a compiler API, an AST module, or a disciplined grep for import statements), and print each forbidden edge as `file -> file (module -> module)`. +Ecosystem tools exist for some stacks (dependency-cruiser for JS/TS, import-linter for Python, ArchUnit for JVM); prefer one when the repo already uses it or adoption is one config file that mirrors `docs/dependencies.md` - but `docs/dependencies.md` stays the single source of truth, so generate the tool's config from it rather than maintaining two rule sets. + +## 6. Coverage-weighted complexity + +The score per function combines cyclomatic complexity with test coverage so that only *uncovered* complexity fails: complexity squared, scaled down by the fraction of the function's paths the tests execute. A fully covered function scores its complexity; an uncovered one scores its complexity squared. + +- Coverage comes from the repo's existing coverage runner (jest/vitest `--coverage`, `coverage.py`, `go test -cover`, JaCoCo, tarpaulin). If the repo has none, wiring the standard one up is part of this step. +- Complexity comes from the ecosystem's standard analyzer (eslint `complexity` rule, `radon`, `gocyclo`, checkstyle) or, failing that, a small AST script. +- A repo-fitted script under `tools/harden/` joins the two reports and prints each function over threshold as `file:function score (complexity N, coverage P%)`. + +Scope the report to functions the diff touched; pre-existing offenders elsewhere are noted, not fixed, unless the run is full-repo. + +## 7. Flakiness detector + +Almost always a repo-fitted script: run the tests the diff added or touched N times (default 5), shuffling order where the runner supports it (vitest `sequence.shuffle`, pytest-randomly, `go test -shuffle`, JUnit's method orderer). +Any run disagreeing with any other marks that test flaky - a violation like a failure, because a test that passes inconsistently proves nothing. +Keep it affordable: only diff-touched tests, reuse build caches between repeats, and run repeats inside one runner invocation where the framework allows. + +## 8. Mutation testing + +Prefer the ecosystem's mutation framework - Stryker (JS/TS), mutmut or cosmic-ray (Python), PIT (JVM), cargo-mutants (Rust), go-mutesting (Go). +These handle mutant generation, test selection, and reporting far better than a hand-rolled loop; write only the thin config that scopes them. +Hand-roll only when the ecosystem has nothing: an agent-written script that applies one mutation at a time (flip `<` to `<=`, `==` to `!=`, `+` to `-`, negate conditions, drop return values), runs the narrowest relevant test command, and records survivors. + +Keeping it affordable: + +- **Scope to the diff**: mutate only in-scope files, run only the tests that cover them (most frameworks do incremental or per-file runs; use that). +- Set a per-mutant test timeout so an infinite-loop mutant cannot hang the run. +- Equivalent mutants (mutations that provably cannot change behavior) are the one legitimate survivor category: mark them as such in the report with the reasoning, don't chase them forever - two fix rounds, then escalate per the loop rule. + +## Fix-agent prompts + +Keep them minimal - the violation list, the file paths, the fix vocabulary, nothing else. +The agents inherit no conversation and need none: a surviving mutant at `cart.ts:41 (< flipped to <=)` plus "write the test that kills it, at the layer the surrounding tests use" is a complete task. +Do not paste tool source, spec prose, or this reference into fix prompts; long prompts are how rules soften. diff --git a/factory/skills/ship/scripts/aggregate-findings.py b/factory/skills/ship/scripts/aggregate-findings.py new file mode 100755 index 0000000..81d86d9 --- /dev/null +++ b/factory/skills/ship/scripts/aggregate-findings.py @@ -0,0 +1,185 @@ +#!/usr/bin/env python3 +"""Number the panel's findings for verification, then apply the verdicts. + + aggregate-findings.py plan + aggregate-findings.py aggregate + [--expected lens,lens] + +`plan` prints the BLOCK and CONCERN findings as one JSON array with stable ids +(f1, f2, ...) to hand to the verifiers. `aggregate` applies their verdicts and +prints the report data: refuted findings dropped, an unconfirmed BLOCK demoted +to CONCERN, duplicates merged by file:line, and the overall verdict. +""" + +from __future__ import annotations + +import json +import re +import sys +from pathlib import Path + +FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S) +RESULT_SUFFIXES = (".json", ".out") + + +def load_json(path: Path): + """Result files are JSON, or a final message wrapping one fenced block.""" + text = path.read_text(encoding="utf-8", errors="replace").strip() + for candidate in [text, *FENCE.findall(text)]: + candidate = candidate.strip() + if not candidate: + continue + try: + return json.loads(candidate) + except json.JSONDecodeError: + continue + return None + + +def read_results(directory: Path) -> tuple[dict[str, object], list[str]]: + """Every parseable result file by agent name, plus the names that failed.""" + parsed: dict[str, object] = {} + unreadable: list[str] = [] + for path in sorted(directory.iterdir()): + if not path.is_file() or path.suffix not in RESULT_SUFFIXES: + continue + payload = load_json(path) + if payload is None: + unreadable.append(path.stem) + else: + parsed[path.stem] = payload + return parsed, unreadable + + +def collect_findings(lens_results: dict[str, object]) -> list[dict]: + findings = [] + for lens, payload in lens_results.items(): + if not isinstance(payload, dict): + continue + for raw in payload.get("findings", []) or []: + finding = dict(raw) + finding["lens"] = lens + findings.append(finding) + return findings + + +def numbered(findings: list[dict]) -> list[dict]: + """BLOCK and CONCERN findings get an id; NITs skip verification.""" + out = [] + for finding in findings: + if str(finding.get("severity", "")).upper() not in ("BLOCK", "CONCERN"): + continue + entry = dict(finding) + entry["id"] = f"f{len(out) + 1}" + out.append(entry) + return out + + +def merge(findings: list[dict]) -> list[dict]: + """One entry per file:line; keep the fullest detail, credit every lens.""" + merged: dict[tuple, dict] = {} + for finding in findings: + key = (finding.get("file"), finding.get("line"), ) + existing = merged.get(key) + if existing is None: + entry = dict(finding) + entry["lenses"] = [finding.get("lens")] if finding.get("lens") else [] + entry.pop("lens", None) + merged[key] = entry + continue + if finding.get("lens") and finding["lens"] not in existing["lenses"]: + existing["lenses"].append(finding["lens"]) + if len(str(finding.get("detail", ""))) > len(str(existing.get("detail", ""))): + existing["detail"] = finding["detail"] + existing["title"] = finding.get("title", existing.get("title")) + return list(merged.values()) + + +def cmd_plan(lens_dir: Path) -> int: + lens_results, unreadable = read_results(lens_dir) + findings = numbered(collect_findings(lens_results)) + print(json.dumps(findings, indent=2, ensure_ascii=False)) + print( + f"\n{len(findings)} finding(s) to verify from {len(lens_results)} lens result(s)" + + (f"; unparseable: {', '.join(unreadable)}" if unreadable else ""), + file=sys.stderr, + ) + return 0 + + +def cmd_aggregate(lens_dir: Path, verifier_dir: Path, expected: list[str]) -> int: + lens_results, unreadable = read_results(lens_dir) + all_findings = collect_findings(lens_results) + to_verify = numbered(all_findings) + + verdicts: dict[str, dict] = {} + verifier_results, verifier_unreadable = read_results(verifier_dir) + for payload in verifier_results.values(): + for entry in payload if isinstance(payload, list) else [payload]: + if isinstance(entry, dict) and entry.get("id"): + verdicts[entry["id"]] = entry + + blockers, concerns = [], [] + for finding in to_verify: + verdict = verdicts.get(finding["id"], {}) + status = str(verdict.get("status", "")).upper() + if status == "REFUTED": + continue + finding = dict(finding) + finding["verification"] = status or "UNVERIFIED" + finding["reportedSeverity"] = str(finding.get("severity", "")).upper() + if finding["reportedSeverity"] == "BLOCK" and status == "CONFIRMED": + blockers.append(finding) + else: + # An unconfirmed BLOCK carries on as a CONCERN, not as a BLOCK. + finding["severity"] = "CONCERN" + concerns.append(finding) + + nits = [f for f in all_findings if str(f.get("severity", "")).upper() == "NIT"] + failed = sorted(set(expected) - set(lens_results)) + unreadable + + report = { + "verdict": "BLOCK" if blockers else "CONCERNS" if concerns else "PASS", + "panel": sorted(lens_results), + "failedLenses": failed, + "blockers": merge(blockers), + "concerns": merge(concerns), + "nits": merge(nits), + "good": {lens: (p.get("good", []) if isinstance(p, dict) else []) + for lens, p in sorted(lens_results.items())}, + } + print(json.dumps(report, indent=2, ensure_ascii=False)) + + unverified = [f["id"] for f in to_verify if f["id"] not in verdicts] + notes = [] + if failed: + notes.append(f"failed lenses: {', '.join(failed)}") + if unverified: + notes.append(f"unverified findings kept as concerns: {', '.join(unverified)}") + if verifier_unreadable: + notes.append(f"unparseable verifier files: {', '.join(verifier_unreadable)}") + if notes: + print("\n" + "; ".join(notes), file=sys.stderr) + return 0 + + +def main(argv: list[str]) -> int: + args, expected = [], [] + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--expected" and rest: + expected = [n.strip() for n in rest.pop(0).split(",") if n.strip()] + else: + args.append(value) + + if len(args) == 2 and args[0] == "plan": + return cmd_plan(Path(args[1])) + if len(args) == 3 and args[0] == "aggregate": + return cmd_aggregate(Path(args[1]), Path(args[2]), expected) + print(__doc__, file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/factory/skills/ship/scripts/pr-evidence.py b/factory/skills/ship/scripts/pr-evidence.py new file mode 100755 index 0000000..63a448f --- /dev/null +++ b/factory/skills/ship/scripts/pr-evidence.py @@ -0,0 +1,311 @@ +#!/usr/bin/env python3 +"""Turn e2e evidence into a PR-ready Evidence section, publish its images, and gate the body. + + pr-evidence.py extract --out [--branch pr-evidence] + [--url-template URL] + pr-evidence.py publish [--branch pr-evidence] [--remote origin] + pr-evidence.py check [--kind frontend|non-frontend] + +`extract` reads the E2E_DATA block of a rendered e2e report, decodes every +screenshot to `/{plan}/{scenario}/{step}.png`, and writes +`/evidence.md`: one "## Evidence" section with the images (frontend) or +before/after state tables (non-frontend) a reviewer can read inline on the PR. +Image links use --url-template, where `{path}` is the file's path under ; +on a GitHub remote the template is derived from --branch when omitted. + +`publish` commits everything under onto the assets branch and pushes it, +using a temporary index and plumbing commands only: the working tree, the +current branch, and the index are never touched. Data URIs do not render on a +PR, so this is how screenshots become visible. + +`check` exits non-zero unless the body has a non-empty "## Evidence" section +holding real proof: an image served over http(s), or a labeled before/after +(or merge-base/branch) pair of fenced blocks. Placeholders fail it. +""" + +from __future__ import annotations + +import argparse +import base64 +import json +import os +import re +import subprocess +import sys +import tempfile +from datetime import datetime, timezone +from pathlib import Path + +DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.S) +IDENT = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*") +PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.I) +IMAGE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)") +PAIR_LABELS = { + "before": "before", "after": "after", + "on merge base": "base", "on the merge base": "base", "merge base": "base", + "on this branch": "branch", "on the branch": "branch", "this branch": "branch", +} +LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.I) +FENCE_OPEN = re.compile(r"^\s*```") +IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".webp"} + + +def fail(message: str) -> None: + print(f"pr-evidence: {message}", file=sys.stderr) + sys.exit(1) + + +def slug(text: str, fallback: str) -> str: + cleaned = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-") + return cleaned[:60] or fallback + + +def load_e2e_data(report: Path) -> dict: + """The data block is a JS object literal; accept JSON, then a lightly normalized form.""" + match = DATA_BLOCK.search(report.read_text(encoding="utf-8", errors="replace")) + if not match: + fail(f"no E2E_DATA block between the markers in {report}") + literal = match.group(1) + for candidate in (literal, node_json(literal), js_to_json(literal)): + if candidate is None: + continue + try: + return json.loads(candidate) + except json.JSONDecodeError: + continue + fail("E2E_DATA is not parseable as JSON, through node, or as a plain JS object literal") + return {} + + +def js_to_json(literal: str) -> str: + """Rewrite a plain JS object literal as JSON without touching string contents. + + Handles bare keys, single-quoted strings, trailing commas, and comments; + string bodies (where prose can contain colons and commas) pass through as-is. + """ + out: list[str] = [] + i, n = 0, len(literal) + while i < n: + ch = literal[i] + if ch in "\"'": + quote, j, body = ch, i + 1, [] + while j < n and literal[j] != quote: + if literal[j] == "\\" and j + 1 < n: + body.append(literal[j:j + 2]) + j += 2 + else: + body.append(literal[j]) + j += 1 + text = "".join(body) + if quote == "'": + text = text.replace("\\'", "'").replace('"', '\\"') + out.append(f'"{text}"') + i = j + 1 + elif literal.startswith("//", i): + i = literal.find("\n", i) if literal.find("\n", i) != -1 else n + elif literal.startswith("/*", i): + end = literal.find("*/", i + 2) + i = n if end == -1 else end + 2 + elif ch == ",": + j = i + 1 + while j < n and literal[j] in " \t\r\n": + j += 1 + if j < n and literal[j] in "}]": + i += 1 + else: + out.append(ch) + i += 1 + else: + ident = IDENT.match(literal, i) + if ident: + j = ident.end() + while j < n and literal[j] in " \t\r\n": + j += 1 + if j < n and literal[j] == ":": + out.append(f'"{ident.group(0)}"') + else: + out.append(ident.group(0)) + i = ident.end() + else: + out.append(ch) + i += 1 + return "".join(out) + + +def node_json(literal: str) -> str | None: + """A JS object literal is exactly what node parses; use it when installed.""" + with tempfile.NamedTemporaryFile("w", suffix=".js", delete=False, encoding="utf-8") as handle: + handle.write(literal) + script = f"process.stdout.write(JSON.stringify(eval('(' + require('fs').readFileSync({handle.name!r}, 'utf8') + ')')))" + try: + result = subprocess.run(["node", "-e", script], capture_output=True, text=True) + return result.stdout if result.returncode == 0 else None + except OSError: + return None + finally: + os.unlink(handle.name) + + +def git(*args: str, env: dict | None = None, check: bool = True) -> str: + result = subprocess.run(["git", *args], capture_output=True, text=True, env=env) + if check and result.returncode != 0: + fail(f"git {' '.join(args)} failed: {result.stderr.strip()}") + return result.stdout.strip() + + +def github_slug(remote: str) -> str | None: + url = git("remote", "get-url", remote, check=False) + match = re.search(r"github\.com[:/]([^/]+)/([^/\s]+?)(?:\.git)?$", url) + return f"{match.group(1)}/{match.group(2)}" if match else None + + +def url_template(args: argparse.Namespace) -> str | None: + if args.url_template: + if "{path}" not in args.url_template: + fail("--url-template must contain {path}") + return args.url_template + repo = github_slug(args.remote) + if repo: + return f"https://github.com/{repo}/blob/{args.branch}/{{path}}?raw=true" + return None + + +def fenced(value: object, language: str = "json") -> str: + text = value if isinstance(value, str) else json.dumps(value, indent=2, ensure_ascii=False) + return f"```{language}\n{text.rstrip()}\n```" + + +def extract(args: argparse.Namespace) -> None: + data = load_e2e_data(Path(args.report)) + kind = data.get("kind", "non-frontend") + plan = data.get("planName") or slug(data.get("title", ""), "plan") + scenarios = data.get("scenarios", []) or [] + summary = data.get("summary") or {} + out = Path(args.out) + template = url_template(args) if kind == "frontend" else None + if kind == "frontend" and template is None: + fail("frontend report but no --url-template and the remote is not GitHub; pass --url-template") + + lines = ["## Evidence", "", + f"Captured from the e2e run of `{plan}` at {data.get('generatedAt', 'unknown time')}: " + f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] + written: list[str] = [] + for index, scenario in enumerate(scenarios, 1): + scenario_id = slug(str(scenario.get("id") or scenario.get("title") or index), f"scenario-{index}") + lines += [f"### {scenario.get('title', scenario_id)} - {scenario.get('status', 'unknown')}", "", + f"Given {scenario.get('given', '?')}, when {scenario.get('when', '?')}, then {scenario.get('then', '?')}.", ""] + for step_index, shot in enumerate(scenario.get("screenshots", []) or [], 1): + uri = shot.get("dataUri", "") + if not uri.startswith("data:image/"): + fail(f"{scenario_id} step {step_index} has no data URI screenshot") + ext = "." + uri[len("data:image/"):uri.index(";")].replace("jpeg", "jpg") + relative = Path(plan) / scenario_id / f"{step_index:02d}-{slug(shot.get('step', ''), 'step')}{ext}" + target = out / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(base64.b64decode(uri.split(",", 1)[1])) + written.append(relative.as_posix()) + url = template.replace("{path}", relative.as_posix()) + lines += [f"**{shot.get('step', 'step')}** - {shot.get('caption', '')}", "", + f"![{shot.get('caption') or shot.get('step', 'screenshot')}]({url})", ""] + for state in scenario.get("dataModelState", []) or []: + lines += [f"**{state.get('step', 'step')}** - {state.get('caption', '')} (`{state.get('entity', '?')}`)", "", + "**Before**", "", fenced(state.get("before", "")), "", + "**After**", "", fenced(state.get("after", "")), ""] + output = scenario.get("logsOrOutput") + if output: + lines += ["
Captured output", "", fenced(output, ""), "", "
", ""] + out.mkdir(parents=True, exist_ok=True) + (out / "evidence.md").write_text("\n".join(lines).rstrip() + "\n", encoding="utf-8") + print(json.dumps({"kind": kind, "plan": plan, "scenarios": len(scenarios), "images": written, + "evidence": str(out / "evidence.md"), "urlTemplate": template}, indent=2)) + + +def publish(args: argparse.Namespace) -> None: + root = Path(args.dir).resolve() + files = sorted(p for p in root.rglob("*") if p.is_file() and p.suffix.lower() in IMAGE_SUFFIXES) + if not files: + fail(f"nothing to publish: no image files under {root}") + ref = f"refs/remotes/{args.remote}/{args.branch}" + subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True) + parent = git("rev-parse", "--verify", "--quiet", ref, check=False) or None + with tempfile.TemporaryDirectory() as tmp: + env = {**os.environ, "GIT_INDEX_FILE": str(Path(tmp) / "index")} + if parent: + git("read-tree", parent, env=env) + for path in files: + blob = git("hash-object", "-w", str(path)) + git("update-index", "--add", "--cacheinfo", f"100644,{blob},{path.relative_to(root).as_posix()}", env=env) + tree = git("write-tree", env=env) + message = f"evidence: {len(files)} screenshots captured {datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')}" + commit = git("commit-tree", tree, *(["-p", parent] if parent else []), "-m", message) + git("push", args.remote, f"{commit}:refs/heads/{args.branch}") + print(json.dumps({"branch": args.branch, "commit": commit, "files": [p.relative_to(root).as_posix() for p in files]}, indent=2)) + + +def evidence_section(body: str) -> str | None: + match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.M | re.S) + return match.group(1) if match else None + + +def labeled_pairs(section: str) -> set[str]: + """Labels (before/after/base/branch) that head a fenced block within the next two lines.""" + found: set[str] = set() + lines = section.splitlines() + for i, line in enumerate(lines): + match = LABEL_LINE.match(line) + if not match: + continue + label = PAIR_LABELS.get(match.group(1).strip().lower()) + if label and any(FENCE_OPEN.match(nxt) for nxt in lines[i + 1:i + 3]): + found.add(label) + return found + + +def check(args: argparse.Namespace) -> None: + section = evidence_section(Path(args.body).read_text(encoding="utf-8")) + problems: list[str] = [] + if section is None or not section.strip(): + problems.append("no non-empty '## Evidence' section") + section = "" + placeholder = PLACEHOLDERS.search(section) + if placeholder: + problems.append(f"placeholder or inline data URI in Evidence: {placeholder.group(0)!r}") + images = IMAGE.findall(section) + labels = labeled_pairs(section) + pair = {"before", "after"} <= labels or {"base", "branch"} <= labels + if args.kind == "frontend" and not images: + problems.append("frontend change but no http(s) image in Evidence") + if args.kind == "non-frontend" and not pair: + problems.append("non-frontend change but no labeled Before/After or merge-base/branch pair in Evidence") + if not images and not pair: + problems.append("Evidence holds neither an http(s) image nor a labeled before/after pair") + if problems: + fail("PR body rejected:\n - " + "\n - ".join(problems)) + print(f"pr-evidence: ok ({len(images)} images, pairs: {', '.join(sorted(labels)) or 'none'})") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + sub = parser.add_subparsers(dest="command", required=True) + ex = sub.add_parser("extract") + ex.add_argument("report") + ex.add_argument("--out", required=True) + ex.add_argument("--branch", default="pr-evidence") + ex.add_argument("--remote", default="origin") + ex.add_argument("--url-template") + ex.set_defaults(run=extract) + pub = sub.add_parser("publish") + pub.add_argument("dir") + pub.add_argument("--branch", default="pr-evidence") + pub.add_argument("--remote", default="origin") + pub.set_defaults(run=publish) + ck = sub.add_parser("check") + ck.add_argument("body") + ck.add_argument("--kind", choices=["frontend", "non-frontend"]) + ck.set_defaults(run=check) + args = parser.parse_args() + args.run(args) + + +if __name__ == "__main__": + main() diff --git a/factory/skills/ship/scripts/run-pi-agents.sh b/factory/skills/ship/scripts/run-pi-agents.sh new file mode 100755 index 0000000..6abfd4d --- /dev/null +++ b/factory/skills/ship/scripts/run-pi-agents.sh @@ -0,0 +1,69 @@ +#!/usr/bin/env bash +set -uo pipefail + +if [ "$#" -ne 2 ]; then + echo "Usage: run-pi-agents.sh " >&2 + exit 2 +fi + +prompt_dir="$1" +output_dir="$2" +pi_bin="${PI_BIN:-pi}" + +if [ ! -d "$prompt_dir" ]; then + echo "Prompt directory not found: $prompt_dir" >&2 + exit 2 +fi +if ! command -v "$pi_bin" >/dev/null 2>&1; then + echo "Pi executable not found: $pi_bin" >&2 + exit 2 +fi + +mkdir -p "$output_dir" +prompts=("$prompt_dir"/*.md) +if [ ! -e "${prompts[0]}" ]; then + echo "No .md prompts found in: $prompt_dir" >&2 + exit 2 +fi + +args=( + --no-session + --no-skills + --no-extensions + --no-prompt-templates + --tools read,bash,grep,find,ls + --print +) +if [ -n "${PI_PROVIDER:-}" ]; then + args+=(--provider "$PI_PROVIDER") +fi +if [ -n "${PI_MODEL:-}" ]; then + args+=(--model "$PI_MODEL") +fi +if [ -n "${PI_REASONING_LEVEL:-}" ]; then + args+=(--thinking "$PI_REASONING_LEVEL") +fi + +pids=() +names=() + +for prompt in "${prompts[@]}"; do + name=$(basename "$prompt" .md) + output="$output_dir/$name.out" + error="$output_dir/$name.err" + "$pi_bin" "${args[@]}" < "$prompt" > "$output" 2> "$error" & + pids+=("$!") + names+=("$name") +done + +status=0 +for index in "${!pids[@]}"; do + if wait "${pids[$index]}"; then + rm -f "$output_dir/${names[$index]}.err" + else + echo "Pi agent failed: ${names[$index]}" >&2 + status=1 + fi +done + +exit "$status" diff --git a/factory/skills/ship/templates/review.html b/factory/skills/ship/templates/review.html new file mode 100644 index 0000000..31eb8d9 --- /dev/null +++ b/factory/skills/ship/templates/review.html @@ -0,0 +1,214 @@ + + + + + +Review + + + +
+ REVIEW + +
+ +
+
+

+ +
+
+
+ +
+

Spec conformance

+
+
+ +
+

Previous findings

+
+
+ +
+

Blockers

+
+
+ +
+

Concerns

+
+
+ +
+

Nits

+
FileLensIssue
+
+ +
+

What's good

+
+
+ +
+

Next step

+
+
+
+ + + + diff --git a/plugins/dev/skills/ship/references/orchestration.md b/plugins/dev/skills/ship/references/orchestration.md index c6aece4..57cc2ec 100644 --- a/plugins/dev/skills/ship/references/orchestration.md +++ b/plugins/dev/skills/ship/references/orchestration.md @@ -30,10 +30,10 @@ explicitly requested heavyweight runs only lives in ## Native transport Use the host's plain subagent tool exactly as provided (the Agent tool on -Claude Code, the `task` tool on opencode). Launch all agents of a batch in a -single parallel call: on opencode that means issuing every `task` call of the -batch in one message with the default general subagent. Native subagents -receive the prompt contracts below directly. +Claude Code, `spawn_agent` on Codex, the `task` tool on opencode). Launch all +agents of a batch in a single parallel call: on opencode that means issuing +every `task` call of the batch in one message with the default general +subagent. Native subagents receive the prompt contracts below directly. Result delivery is file-based, because on some hosts subagents run in the background and their final message never reaches the orchestrator: From 8d4b12c6c9112bcf3d28ca0e7c10414ece606e3e Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 11:15:30 +0200 Subject: [PATCH 04/30] feat(factory): the run skill and run-state.py What: add factory/skills/run/SKILL.md (transport and .dev/ preflight, init/resume, inline scope, handoff, per-phase launch/judge/act loop, closing report), references/judgment.md and references/launch.md, and scripts/run-state.py with init/handoff/diff-spec/record/check-result/show. Why: this is the orchestrator - the one skill that launches the phase copies, judges their results against evidence, and survives a closed session through the state file run-state.py owns. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/evals/tests/test_run_state.py | 211 +++++++++++++++++ factory/skills/run/SKILL.md | 51 ++++ factory/skills/run/references/judgment.md | 40 ++++ factory/skills/run/references/launch.md | 32 +++ factory/skills/run/scripts/run-state.py | 276 ++++++++++++++++++++++ 5 files changed, 610 insertions(+) create mode 100644 factory/evals/tests/test_run_state.py create mode 100644 factory/skills/run/SKILL.md create mode 100644 factory/skills/run/references/judgment.md create mode 100644 factory/skills/run/references/launch.md create mode 100644 factory/skills/run/scripts/run-state.py diff --git a/factory/evals/tests/test_run_state.py b/factory/evals/tests/test_run_state.py new file mode 100644 index 0000000..83a9ffa --- /dev/null +++ b/factory/evals/tests/test_run_state.py @@ -0,0 +1,211 @@ +"""Unit tests for factory/skills/run/scripts/run-state.py. + +Run from the repo root: python3 -m unittest discover -s factory/evals/tests -t . +Each test runs the script as a subprocess against a scratch .dev/ directory, +so it proves the real CLI contract the run skill depends on. +""" + +from __future__ import annotations + +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +SCRIPT = REPO_ROOT / "factory" / "skills" / "run" / "scripts" / "run-state.py" + +SPEC_TWO_CHANGE_SETS = """# Fixture plan + +## Research + +D-example: An example decision? + ✓ do it - reason + ✗ don't - reason + ⊘ not doing - reason one + +## Scope + +### Non-goals + +- ⊘ something else - reason two + +## Change plan + +1. Change set one + a. `a.py` - does a thing + tests: [unit] a -> b; [unit] c -> d + +2. Change set two + a. `b.py` - does another thing + tests: [unit] e -> f +""" + + +def run_state(cwd: Path, *args: str, stdin: str | None = None) -> subprocess.CompletedProcess: + return subprocess.run( + [sys.executable, str(SCRIPT), *args], + cwd=cwd, + input=stdin, + capture_output=True, + text=True, + ) + + +class RunStateTest(unittest.TestCase): + def setUp(self) -> None: + self.tmp = tempfile.TemporaryDirectory() + self.cwd = Path(self.tmp.name) + self.plan_dir = self.cwd / ".dev" / "fixture-plan" + self.plan_dir.mkdir(parents=True) + + def tearDown(self) -> None: + self.tmp.cleanup() + + def write_spec(self, text: str = SPEC_TWO_CHANGE_SETS) -> None: + (self.plan_dir / "spec.md").write_text(text, encoding="utf-8") + + def init_plan(self) -> None: + result = run_state(self.cwd, "init", "fixture-plan", "--request", "do it", "--base", "main") + self.assertEqual(result.returncode, 0, result.stderr) + + def test_init_twice_exits_1_naming_existing_state(self) -> None: + self.init_plan() + result = run_state(self.cwd, "init", "fixture-plan", "--request", "do it", "--base", "main") + self.assertEqual(result.returncode, 1) + self.assertIn("already exists", result.stdout) + + def test_handoff_records_scenarios_and_not_doing_lines(self) -> None: + self.init_plan() + self.write_spec() + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + self.assertEqual(result.returncode, 0, result.stderr) + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + self.assertIsNotNone(state["spec_sha256"]) + scenario_texts = state["scenario_texts"] + self.assertEqual(sorted(scenario_texts), ["1", "2"]) + total = sum(len(v) for v in scenario_texts.values()) + self.assertEqual(total, 3) + self.assertEqual(len(state["not_doing_lines"]), 2) + + def test_diff_spec_after_reworded_scenario_exits_1(self) -> None: + self.init_plan() + self.write_spec() + run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + reworded = SPEC_TWO_CHANGE_SETS.replace( + "tests: [unit] a -> b; [unit] c -> d", + "tests: [unit] a -> z; [unit] c -> d", + ) + self.write_spec(reworded) + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 1) + self.assertIn("change set 1", result.stdout) + self.assertIn("a -> b", result.stdout) + self.assertIn("a -> z", result.stdout) + + def test_diff_spec_unchanged_exits_0(self) -> None: + self.init_plan() + self.write_spec() + run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 0) + + def test_diff_spec_after_dropped_scenario_and_not_doing_line(self) -> None: + self.init_plan() + self.write_spec() + run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + dropped = SPEC_TWO_CHANGE_SETS.replace( + "tests: [unit] a -> b; [unit] c -> d", "tests: [unit] c -> d" + ).replace("- ⊘ something else - reason two\n", "") + self.write_spec(dropped) + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 1) + self.assertIn("change set 1", result.stdout) + self.assertIn("⊘ line dropped", result.stdout) + + def test_record_rejects_action_outside_the_four(self) -> None: + self.init_plan() + before = (self.plan_dir / "factory-run.json").read_text() + result = run_state( + self.cwd, + "record", + "fixture-plan", + stdin=json.dumps({"phase": "build", "attempt": 1, "action": "park"}), + ) + self.assertEqual(result.returncode, 1) + after = (self.plan_dir / "factory-run.json").read_text() + self.assertEqual(before, after) + + def test_record_a_repair_then_show(self) -> None: + self.init_plan() + result = run_state( + self.cwd, + "record", + "fixture-plan", + stdin=json.dumps( + { + "phase": "build", + "attempt": 1, + "action": "repair", + "rationale": "fixed a typo", + "evidence": ["ran tests"], + "repair": {"description": "fixed a typo", "files": ["a.py"]}, + } + ), + ) + self.assertEqual(result.returncode, 0, result.stderr) + result = run_state(self.cwd, "show", "fixture-plan") + self.assertIn("fixed a typo", result.stdout) + + def test_show_after_two_attempts_prints_two_rows(self) -> None: + self.init_plan() + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state["phases"]["build"] = [{"status": "failed"}, {"status": "done"}] + (self.plan_dir / "factory-run.json").write_text(json.dumps(state)) + result = run_state(self.cwd, "show", "fixture-plan") + self.assertIn("build attempt 1: failed", result.stdout) + self.assertIn("build attempt 2: done", result.stdout) + + def write_result(self, payload: dict) -> Path: + path = self.cwd / "result.json" + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + def test_check_result_done_exits_0(self) -> None: + path = self.write_result({"schema": "factory.result/1", "phase": "build", "status": "done"}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 0) + + def test_check_result_failed_exits_1_with_reason(self) -> None: + path = self.write_result({"status": "failed", "reason": "no result file"}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 1) + self.assertIn("no result file", result.stdout) + + def test_check_result_invalid_json_exits_3(self) -> None: + path = self.cwd / "bad.json" + path.write_text("{not json", encoding="utf-8") + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 3) + self.assertIn("unparseable", result.stdout) + + def test_check_result_stopped_exits_2_with_kind(self) -> None: + path = self.write_result({"status": "stopped", "stop": {"kind": "secret.found", "action": "purge it"}}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 2) + self.assertIn("secret.found", result.stdout) + + def test_check_result_missing_file_exits_3(self) -> None: + result = run_state(self.cwd, "check-result", str(self.cwd / "nope.json")) + self.assertEqual(result.returncode, 3) + + def test_check_result_unknown_extra_key_exits_0(self) -> None: + path = self.write_result({"status": "done", "totally_unknown_field": 42}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md new file mode 100644 index 0000000..a6ad360 --- /dev/null +++ b/factory/skills/run/SKILL.md @@ -0,0 +1,51 @@ +--- +name: run +description: Take a request through an interactive scope, then run an unattended scope-review, build, and ship as phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use to run the factory end to end on Claude Code or Codex. +disable-model-invocation: true +--- + +# Run + +You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase invoke another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. + +## 1. Preflight + +Confirm the host has a subagent tool per the transport ladder in [references/launch.md](references/launch.md); with none, stop before the interview so nobody pays for a scope that cannot run. +Check that `.dev/` is git-ignored in the consuming repository (`git check-ignore -q .dev/`); when it is not, append the entry to `.gitignore` yourself, and once the run branch exists at handoff, record the write as a repair with the check output as evidence (D-plan-files-ignored). The two hard stops in section 5 bind this write too. +Resolve your own base directory from the host's injected "Base directory for this skill" line. + + +If two unfinished state files exist under `.dev/` with no branch match, or a recorded run branch no longer exists, list the candidates (plan, branch, last phase) or name the missing branch and ask which to resume - a person is at the keyboard invoking `/factory:run`, so this is not a post-go touchpoint. + + +## 2. Init or resume + +No request and a run branch checked out: resume that plan. +No request and no branch: resume the single state file under `.dev/` not marked done. +A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it, create `.dev/{plan}/` with `request.md` holding the request, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch}` (D-plan-slug, D-request-input). + +## 3. Scope, inline + + +Launch the copied `scope` skill inline, in this session: read its `SKILL.md` at the absolute path `{run-skill-root}/../scope/SKILL.md`, telling it to read `{scope-skill-root}` as that path's parent directory. It keeps its interview and ends with one explicit go question you ask the user. + + +## 4. Handoff + +At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan}` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). + +## 5. Per phase: launch, judge, act + +For each phase in order (`scope-review`, `build`, `ship`): + +1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. +2. Wait for the host's completion signal - a batch in flight is not over. +3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). +4. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). +5. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. +6. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. +7. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. + +## 6. Closing report + +Print: phases, attempts, decisions, repairs, and the PR link or the exact action a person must take. Read only result files, check outputs, `git log`, and the state file for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). diff --git a/factory/skills/run/references/judgment.md b/factory/skills/run/references/judgment.md new file mode 100644 index 0000000..a800b8a --- /dev/null +++ b/factory/skills/run/references/judgment.md @@ -0,0 +1,40 @@ +# Judgment + +Per phase, the evidence commands the orchestrator re-runs as proof and what a `done` looks like. `run-state.py check-result` gives the result file's own claim; these checks are the independent evidence read before trusting it. + +## scope + +`done`: `lint-spec.py .dev/{plan}/spec.md` reports clean, and the go was recorded (the interview happened inline in this session, so the orchestrator already knows). + +## scope-review + +`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a `Verdict: APPROVED` line with a `Rounds:` line. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. + +## build + +`done`: `check-tests.py .dev/{plan}` clean, the spec's Validation block green, commits since handoff match the change plan's numbering, and (when the spec has `[e2e]` scenarios) the e2e report's data block shows every scenario passed. + +## ship + +`done` has two endings, decided by what `origin` resolves to: + +- **A GitHub remote**: `pr-evidence.py check` clean, `review_N.md` carries a verdict for HEAD, and `gh pr view` shows the PR open with head equal to HEAD. +- **A remote no GitHub host backs** (the fixture's bare origin): `gh pr create` cannot succeed, so ship cannot reach `done` at all. The evidence is `git ls-remote origin` showing the branch at HEAD and `pr.md` written to disk. The orchestrator ends the run with a report naming the pull request as the one unfinished step - this is the fixture's expected ending, not a factory fault, and is judged under the third `gh` fault category below, never as a park. + +## The failure ladder + +Cheapest sufficient step, recorded before it is acted on: + +1. **Repair.** Fix it yourself and re-run the phase's checks - at most two repairs per attempt; the third fix is a relaunch and counts as a new attempt. A repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches. +2. **Relaunch.** A fresh subagent, guidance naming what the previous attempt did and what is required instead - never "try again". +3. **End.** Three attempts of one phase without a `done` the orchestrator accepts: end the run with a report naming the last reason, the artifacts, and what a person could do. + +Every repair is committed on the run branch and bound by the two hard stops: never rewrite history, force-push, or delete outside the run branch. + +## Fault categories during ship + +Three, told apart by the error text, never guessed: + +1. **`gh` unauthenticated or GitHub unreachable** (token, credential helper, or sandbox network): an environment fault, the run ends naming the missing prerequisite. +2. **No GitHub remote behind `origin`**: `gh pr create` refuses with "none of the git remotes configured for this repository point to a known GitHub host" before any network call, even with `gh` installed and authenticated. Match this text first - its trailing `gh auth login` hint is never read as a credentials fault. The branch stays pushed, `pr.md` stays written, the run ends with a report naming the origin and the pull request as the one unfinished step. +3. **A secret found after build committed it**: the run ends without a pull request, the branch is left unpushed, the report names the commit to purge. diff --git a/factory/skills/run/references/launch.md b/factory/skills/run/references/launch.md new file mode 100644 index 0000000..6dbb186 --- /dev/null +++ b/factory/skills/run/references/launch.md @@ -0,0 +1,32 @@ +# Launch + +The host transport ladder for one phase agent, and the exact prompt template. Mirrors ship's `orchestration.md` native-transport rung, one agent at a time instead of a batch. + +## Transport ladder + +1. **Claude Code**: the Agent tool. +2. **Codex**: `spawn_agent`. +3. **opencode**: the `task` tool. +4. **Unavailable**: stop before the interview - nobody pays for a scope that cannot run. + +## Prompt template + +``` +You implement exactly one factory phase. Read {phase}-skill-root/SKILL.md at +the absolute path {skill_path}; treat {skill_path}'s parent directory as +{phase}-skill-root when the file uses that placeholder. Follow it exactly. + +Plan: {plan}, at .dev/{plan}/ (absolute: {plan_dir}). +Attempt: {attempt}. +{On a relaunch only: The previous attempt failed with reason "{reason}". +Guidance: {what to do differently, never "try again"}.} + +Scratch root for this attempt: /tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/. +Write your result to .dev/{plan}/results/{phase}-{attempt}.json as your last +action, per the schema in factory-run.md, echoing skill_path back exactly as +given above. + +This phase never launches another phase, and after the go it never asks a +person anything - decide under the unattended policy in factory-run.md and +record the decision in auto_decided. +``` diff --git a/factory/skills/run/scripts/run-state.py b/factory/skills/run/scripts/run-state.py new file mode 100644 index 0000000..fc9502a --- /dev/null +++ b/factory/skills/run/scripts/run-state.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +"""Own factory-run.json: init, handoff, diff-spec, record, check-result, show. + +Usage: + run-state.py init --request --base + run-state.py handoff --branch [--spec ] [--dirty ...] + run-state.py diff-spec [--spec ] + run-state.py record # reads one JSON decision object on stdin + run-state.py check-result + run-state.py show + +The orchestrator loops on the exit code of every subcommand, so a malformed +state file or a bad call never reaches a judgment. Every function here stays +at or under cyclomatic complexity 10 (D-complexity-threshold). +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import subprocess +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) +NOT_DOING = re.compile(r"⊘") + +ACTIONS = {"advance", "repair", "relaunch", "end"} +STATE_NAME = "factory-run.json" + + +def state_path(plan: str) -> Path: + return Path(".dev") / plan / STATE_NAME + + +def spec_path(plan: str, override: str | None) -> Path: + return Path(override) if override else Path(".dev") / plan / "spec.md" + + +def load_state(plan: str) -> dict: + path = state_path(plan) + return json.loads(path.read_text(encoding="utf-8")) + + +def save_state(plan: str, state: dict) -> None: + state_path(plan).write_text(json.dumps(state, indent=2) + "\n", encoding="utf-8") + + +def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list[str]]: + """Scenario texts per change set, and every ⊘ line, from the current spec.""" + texts: dict[str, list[str]] = {} + not_doing: list[str] = [] + in_plan = False + current: str | None = None + for line in spec.read_text(encoding="utf-8").splitlines(): + heading = HEADING.match(line) + if heading: + in_plan = heading.group(1).lower() == "change plan" + if NOT_DOING.search(line): + not_doing.append(line.strip()) + if not in_plan: + continue + change_set = CHANGE_SET.match(line) + if change_set: + current = change_set.group(1) + texts.setdefault(current, []) + continue + tests = TESTS.match(line) + if not tests or current is None: + continue + value = tests.group(1).strip() + if value.lower().startswith("none"): + continue + texts[current] = [s.strip() for s in value.split(";") if s.strip()] + return texts, not_doing + + +def git_dirty_files() -> list[str]: + try: + out = subprocess.run( + ["git", "status", "--porcelain"], capture_output=True, text=True, check=True + ).stdout + except (OSError, subprocess.CalledProcessError): + return [] + return [line[3:].strip() for line in out.splitlines() if line.strip()] + + +def cmd_init(args: argparse.Namespace) -> int: + path = state_path(args.plan) + if path.exists(): + print(f"factory-run.json already exists at {path}") + return 1 + path.parent.mkdir(parents=True, exist_ok=True) + state = { + "plan": args.plan, + "request": args.request, + "base": args.base, + "branch": None, + "spec_sha256": None, + "scenario_texts": {}, + "not_doing_lines": [], + "dirty_files": [], + "phases": {"scope": [], "scope-review": [], "build": [], "ship": []}, + "decisions": [], + "repairs": [], + } + save_state(args.plan, state) + print(f"initialized {path}") + return 0 + + +def cmd_handoff(args: argparse.Namespace) -> int: + spec = spec_path(args.plan, args.spec) + if not spec.is_file(): + print(f"no spec.md at {spec}") + return 1 + state = load_state(args.plan) + digest = hashlib.sha256(spec.read_bytes()).hexdigest() + texts, not_doing = scenario_texts_and_not_doing(spec) + state["branch"] = args.branch + state["spec_sha256"] = digest + state["scenario_texts"] = texts + state["not_doing_lines"] = not_doing + state["dirty_files"] = args.dirty if args.dirty else git_dirty_files() + save_state(args.plan, state) + print(f"handoff recorded: {sum(len(v) for v in texts.values())} scenario(s), {len(not_doing)} not-doing line(s)") + return 0 + + +def cmd_diff_spec(args: argparse.Namespace) -> int: + state = load_state(args.plan) + spec = spec_path(args.plan, args.spec) + texts, not_doing = scenario_texts_and_not_doing(spec) + problems: list[str] = [] + stored_texts: dict[str, list[str]] = state.get("scenario_texts", {}) + for change_set, before in stored_texts.items(): + after = texts.get(change_set, []) + dropped = [t for t in before if t not in after] + added = [t for t in after if t not in before] + if dropped: + problems.append( + f"change set {change_set}: scenario dropped or reworded: before {dropped!r}, now {added!r}" + ) + stored_not_doing = state.get("not_doing_lines", []) + for line in stored_not_doing: + if line not in not_doing: + problems.append(f"⊘ line dropped: {line!r}") + for message in problems: + print(message) + return 1 if problems else 0 + + +def cmd_record(args: argparse.Namespace) -> int: + raw = sys.stdin.read() + try: + decision = json.loads(raw) + except json.JSONDecodeError as exc: + print(f"invalid JSON on stdin: {exc}") + return 1 + action = decision.get("action") + if action not in ACTIONS: + print(f"action {action!r} is not one of {sorted(ACTIONS)}") + return 1 + state = load_state(args.plan) + entry = { + "phase": decision.get("phase"), + "attempt": decision.get("attempt"), + "action": action, + "rationale": decision.get("rationale", ""), + "evidence": decision.get("evidence", []), + } + state.setdefault("decisions", []).append(entry) + if action == "repair": + repair = decision.get("repair", {}) + state.setdefault("repairs", []).append( + { + "phase": entry["phase"], + "attempt": entry["attempt"], + "description": repair.get("description", ""), + "files": repair.get("files", []), + "evidence": entry["evidence"], + } + ) + save_state(args.plan, state) + print(f"recorded {action} for {entry['phase']} attempt {entry['attempt']}") + return 0 + + +def cmd_check_result(args: argparse.Namespace) -> int: + path = Path(args.path) + if not path.is_file(): + print(f"missing result file: {path}") + return 3 + try: + data = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError as exc: + print(f"unparseable result file {path}: {exc}") + return 3 + status = data.get("status") + if status == "done": + print("done") + return 0 + if status == "failed": + print(f"failed: {data.get('reason', '')}") + return 1 + if status == "stopped": + kind = data.get("stop", {}).get("kind", "") + print(f"stopped: {kind}") + return 2 + print(f"failed: unrecognized status {status!r}") + return 1 + + +def cmd_show(args: argparse.Namespace) -> int: + state = load_state(args.plan) + for phase, attempts in state.get("phases", {}).items(): + if not attempts: + print(f"{phase}: no attempts") + continue + for index, attempt in enumerate(attempts, start=1): + print(f"{phase} attempt {index}: {attempt.get('status', 'unknown')}") + for repair in state.get("repairs", []): + print(f"repair on {repair['phase']} attempt {repair['attempt']}: {repair['description']}") + return 0 + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + + init_p = sub.add_parser("init") + init_p.add_argument("plan") + init_p.add_argument("--request", required=True) + init_p.add_argument("--base", required=True) + + handoff_p = sub.add_parser("handoff") + handoff_p.add_argument("plan") + handoff_p.add_argument("--branch", required=True) + handoff_p.add_argument("--spec") + handoff_p.add_argument("--dirty", action="append") + + diff_p = sub.add_parser("diff-spec") + diff_p.add_argument("plan") + diff_p.add_argument("--spec") + + record_p = sub.add_parser("record") + record_p.add_argument("plan") + + check_p = sub.add_parser("check-result") + check_p.add_argument("path") + + show_p = sub.add_parser("show") + show_p.add_argument("plan") + + return parser + + +def main(argv: list[str]) -> int: + args = build_parser().parse_args(argv[1:]) + handlers = { + "init": cmd_init, + "handoff": cmd_handoff, + "diff-spec": cmd_diff_spec, + "record": cmd_record, + "check-result": cmd_check_result, + "show": cmd_show, + } + return handlers[args.command](args) + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) From c6af66bec623b6a69616d20102a29ff0865700e0 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 11:15:47 +0200 Subject: [PATCH 05/30] feat(factory): validate.sh unattended, protocol, and script checks What: add check_factory_unattended (F03, scoped to the four phase copies plus factory-run.md, with a regex self-test), check_factory_protocol (F02, the result-protocol literals and a guard against a phase invoking another), and check_factory_script (F01, runs factory/evals/tests/ excluding the nested-validate harness by name so the gate cannot recurse into itself). Resolve C-factory-result's open verify mark in docs/contracts.md now that check_factory_protocol exists to guarantee it. Why: these are the deterministic guards the factory's unattended policy and result protocol depend on; without check_factory_script excluding its own harness by name, validate.sh would recurse into itself unboundedly. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- docs/contracts.md | 16 +++++ scripts/validate.sh | 138 ++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 154 insertions(+) create mode 100644 docs/contracts.md diff --git a/docs/contracts.md b/docs/contracts.md new file mode 100644 index 0000000..e66bc3f --- /dev/null +++ b/docs/contracts.md @@ -0,0 +1,16 @@ +# Contracts + +## Factory plugin + +C-factory-result: every factory phase skill writes the result path the launch prompt names (.dev/{plan}/results/{phase}-{attempt}.json) as its last action, with schema factory.result/1, and never launches the next phase + guaranteed by: the Factory context section of each phase SKILL.md, checked by validate.sh check_factory_protocol (scripts/validate.sh, factory-plugin/spec.md change set 5) + relied on by: the run skill's judgment loop, run-state.py check-result + (2026-09-20, D-result-envelope) +C-factory-unattended: after the go, no factory phase skill routes a decision to a person - scope is the one interactive phase and is outside the check's scan root, and run/SKILL.md's person-routing lines live in blocks + guaranteed by: the validate.sh check over factory/skills/{scope-review,build,ship,run} and factory/references/factory-run.md, whose regex self-tests against four sample phrases and fails validation on a failed self-test + relied on by: the orchestrator's judgment loop, which never waits on a person; a run out of options ends with a written report instead of a question + (2026-09-20, D-unattended-wording-check, D-human-touchpoints) +C-factory-plan-files: plan files under .dev/ are never committed by a factory run; the run branch carries code, tests, and docs/ only + guaranteed by: the run skill's preflight, which runs git check-ignore -q .dev/ in the consuming repository and appends the entry to .gitignore as a recorded repair when it is missing, so the rule has a carrier in any consuming repository and not only in the fixture ? verify: no free scenario reaches the missing-entry branch, because the fixture's setup.sh writes the entry itself (D-verification names this one of the six knowingly unproven behaviors) + relied on by: any consuming repository that has not ignored .dev/ itself; git log --name-only on the run branch is the proof the closing report cites + (2026-09-20, D-plan-files-ignored) diff --git a/scripts/validate.sh b/scripts/validate.sh index d6332d7..dee0410 100755 --- a/scripts/validate.sh +++ b/scripts/validate.sh @@ -614,6 +614,141 @@ check_pi() { fi } +# =========================================================================== +# F03: unattended factory material never routes a decision to a person +# +# Scoped to factory/skills/{scope-review,build,ship,run} (SKILL.md and their +# references) plus factory/references/factory-run.md only - scope keeps its +# interview, and ci-parity.md, contracts.md, and jira.md keep their dev +# wording because the phases read them for notation, not for who decides. +# =========================================================================== +check_factory_unattended() { + local found + found=$(python3 - <<'PYEOF' +import pathlib, re, sys + +PATTERN = re.compile( + r"\b(ask(s|ed|ing)?|confirm(s|ed|ing)?\s+with|wait(s|ing)?\s+for|check(s|ing)?\s+with)\s+(the\s+)?(user|operator|human)s?\b" + r"|\b(user|operator|human)\s+(decides|approves|authorizes|chooses|confirms|answers)\b" + r"|structured user-input tool|\bhuman call\b|\bconfirm (with|before)\b", + re.I, +) +for sample in ("ask the user which one", "a human call", "the user decides", "confirm with the operator"): + if not PATTERN.search(sample): + print(f"scripts/validate.sh: F03 pattern no longer matches {sample!r}") + sys.exit(0) + +paths = [] +for phase in ("scope-review", "build", "ship", "run"): + root = pathlib.Path("factory/skills") / phase + if root.is_dir(): + paths.extend(sorted(root.rglob("*.md"))) +run_md = pathlib.Path("factory/references/factory-run.md") +if run_md.is_file(): + paths.append(run_md) + +for path in paths: + inside = False + for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1): + if "" in line: + inside = True + elif "" in line: + inside = False + elif not inside and PATTERN.search(line): + print(f"{path}:{number}: routes a decision to a person; decide under factory policy or mark the block ") +PYEOF +) + if [ -n "$found" ]; then + while IFS= read -r line; do + fail "F03" "${line%%: *}" "${line#*: }" + done <<< "$found" + fi +} + +# =========================================================================== +# F02: every factory phase writes the result protocol and never invokes +# another phase +# =========================================================================== +check_factory_protocol() { + local found + found=$(python3 - <<'PYEOF' +import pathlib, re, sys + +PHASES = ("scope", "scope-review", "build", "ship") +INVOKE = re.compile(r"\$factory:([a-z-]+)|/SKILL\.md\b") + +for phase in PHASES: + sf = pathlib.Path("factory/skills") / phase / "SKILL.md" + if not sf.is_file(): + print(f"{sf}: Factory phase skill not found") + continue + text = sf.read_text(encoding="utf-8") + if "factory-run.json" not in text: + print(f"{sf}: Must mention factory-run.json") + if "results/{phase}-{attempt}.json" not in text: + print(f"{sf}: Must mention the literal results/{{phase}}-{{attempt}}.json") + for match in INVOKE.finditer(text): + skill = match.group(1) + if skill is None or skill != phase: + print(f"{sf}: invokes another phase ({match.group(0)}); a phase skill never launches another") + +run_md = pathlib.Path("factory/skills/run/SKILL.md") +if not run_md.is_file(): + print(f"{run_md}: run orchestrator skill not found") + sys.exit(0) +run_text = run_md.read_text(encoding="utf-8") +for phase in PHASES: + if phase not in run_text: + print(f"{run_md}: must name phase '{phase}'") +if "run-state.py" not in run_text: + print(f"{run_md}: must name run-state.py") +PYEOF +) + if [ -n "$found" ]; then + while IFS= read -r line; do + fail "F02" "${line%%: *}" "${line#*: }" + done <<< "$found" + fi +} + +# =========================================================================== +# F01: run-state.py's unit tests pass, excluding the harness that would +# otherwise recurse into this very check (D-nested-validate) +# =========================================================================== +check_factory_script() { + local dir="factory/evals/tests" + [ -d "$dir" ] || return + if ! python3 - "$dir" >"$LOG_DIR/factory-script-tests.log" 2>&1 <<'PYEOF' +import sys, unittest + +EXCLUDE = {"test_factory_checks"} + + +def leaves(suite): + for item in suite: + if isinstance(item, unittest.TestSuite): + yield from leaves(item) + else: + yield item + + +directory = sys.argv[1] +loader = unittest.TestLoader() +discovered = loader.discover(start_dir=directory, top_level_dir=".") +cases = [t for t in leaves(discovered) if t.__class__.__module__.rsplit(".", 1)[-1] not in EXCLUDE] +modules = sorted({t.__class__.__module__.rsplit(".", 1)[-1] for t in cases}) +print(f"discovered modules: {modules}") +suite = unittest.TestSuite(cases) +result = unittest.TextTestRunner(verbosity=2).run(suite) +sys.exit(0 if result.wasSuccessful() and result.testsRun > 0 else 1) +PYEOF + then + tail -60 "$LOG_DIR/factory-script-tests.log" >&2 + fail "F01" "$dir" \ + "Factory script tests failed (full output: $LOG_DIR/factory-script-tests.log); run: python3 -m unittest discover -s factory/evals/tests -t ." + fi +} + # =========================================================================== # Main # =========================================================================== @@ -635,6 +770,9 @@ check_skill_length check_links check_codex check_pi +check_factory_unattended +check_factory_protocol +check_factory_script if [ -s "$ERROR_FILE" ]; then echo "" From aeca92ad54c9d74995b33cf766f5bc69d1d785b7 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 11:16:10 +0200 Subject: [PATCH 06/30] feat(factory): fixture and end-to-end procedure What: add the webhook delivery fixture (an http.server endpoint, two unit tests, a launch command), setup.sh (a fresh git repo with .dev/ ignored and a bare origin so a factory run can push offline), the recorded manual end-to-end procedure per host with pass criteria and a real high-entropy planted-secret variant, and results.md for recording outcomes. Why: the fixture is small enough that a real factory run costs minutes and real enough for build's unit layer; the bare origin lets the paid host runs prove everything except the pull-request step, which stays honestly unproven per D-codex-gh. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/evals/README.md | 52 +++++++++++++++ factory/evals/fixture/README.md | 8 +++ factory/evals/fixture/setup.sh | 31 +++++++++ factory/evals/fixture/tests/test_webhook.py | 51 +++++++++++++++ factory/evals/fixture/webhook/__init__.py | 42 ++++++++++++ factory/evals/fixture/webhook/__main__.py | 10 +++ factory/evals/results.md | 7 ++ factory/evals/tests/test_fixture_setup.py | 72 +++++++++++++++++++++ 8 files changed, 273 insertions(+) create mode 100644 factory/evals/README.md create mode 100644 factory/evals/fixture/README.md create mode 100755 factory/evals/fixture/setup.sh create mode 100644 factory/evals/fixture/tests/test_webhook.py create mode 100644 factory/evals/fixture/webhook/__init__.py create mode 100644 factory/evals/fixture/webhook/__main__.py create mode 100644 factory/evals/results.md create mode 100644 factory/evals/tests/test_fixture_setup.py diff --git a/factory/evals/README.md b/factory/evals/README.md new file mode 100644 index 0000000..345fa9c --- /dev/null +++ b/factory/evals/README.md @@ -0,0 +1,52 @@ +# Factory evals + +Three levels of verification (D-verification): + +1. **Structural, free**: `scripts/validate.sh`'s `check_factory_unattended`, `check_factory_protocol`, and `check_factory_script` - run on every `scripts/validate.sh` invocation. +2. **Unit tests, free**: `python3 -m unittest discover -s factory/evals/tests -t .` - `test_run_state.py`, `test_factory_checks.py`, `test_fixture_setup.py`. +3. **One real end-to-end run per host, paid and manual**: this file's procedure, recorded in `results.md`. + +## The end-to-end procedure + +Per host (Claude Code, Codex), because these are paid host sessions build's mocked `[e2e]` environment cannot execute: + +1. Set up the fixture: `bash factory/evals/fixture/setup.sh` and note the destination it prints. +2. Before opening the host, check the per-host setup below is in place, the secret scanner included - the gauntlet's acquire-a-missing-tool step needs a network this fixture run does not have. +3. Open the host in the fixture's destination directory. +4. Invoke `/factory:run "Add idempotency keys to webhook delivery"`. +5. Answer the interview, say go, then leave - no question should come after the go. +6. Once the run ends, record the outcome in `results.md`. + +### Pass criteria + +- The branch `factory/{plan}` pushed to the fixture's bare origin, with one commit per change set. +- `pr.md` written. +- `factory-run.json` recording `scope`, `scope-review`, and `build` as `done`, with every decision, and `ship`'s last attempt `failed` with the no-GitHub-remote reason. +- No question asked after the go. +- Every result file echoing `skill_path` as the path the launch prompt passed. +- `git log --name-only` on the run branch names no `.dev/` path. + +Ship reaching `gh pr create` and failing there is part of the pass, not an exception to it: with `gh` installed and authenticated the command still refuses with "none of the git remotes configured for this repository point to a known GitHub host" before any network call, the orchestrator judges that under the third `gh` fault category in [`judgment.md`](../skills/run/references/judgment.md), and the run ends with a report naming the origin and the pull request as the one unfinished step. There is no run in which all four phases are `done` on this fixture; that ending leaves the pull-request step and the Codex GitHub prerequisites unproven. + +### Planted-secret variant + +One per host, before build's first commit: write into `webhook/__init__.py` a `GITHUB_TOKEN = "ghp_{36 random alphanumerics}"`, generated at setup time as one contiguous literal: + +```bash +LC_ALL=C tr -dc 'A-Za-z0-9' .gitignore <<'EOF' +.dev/ +__pycache__/ +*.pyc +EOF +git add -A +git commit -q -m "chore: initial commit of the webhook fixture" + +git init -q --bare "$ORIGIN" +git remote add origin "$ORIGIN" +git push -q -u origin main + +echo "$DEST" diff --git a/factory/evals/fixture/tests/test_webhook.py b/factory/evals/fixture/tests/test_webhook.py new file mode 100644 index 0000000..bb0731b --- /dev/null +++ b/factory/evals/fixture/tests/test_webhook.py @@ -0,0 +1,51 @@ +"""Two unit tests for the webhook fixture: run with python3 -m unittest.""" + +from __future__ import annotations + +import json +import threading +import unittest +import urllib.error +import urllib.request + +import webhook +from webhook import serve + + +class WebhookTest(unittest.TestCase): + def setUp(self) -> None: + webhook.DELIVERIES.clear() + self.server = serve(port=0) + self.port = self.server.server_address[1] + self.thread = threading.Thread(target=self.server.serve_forever, daemon=True) + self.thread.start() + + def tearDown(self) -> None: + self.server.shutdown() + self.thread.join() + self.server.server_close() + + def post(self, body: bytes) -> tuple[int, bytes]: + request = urllib.request.Request( + f"http://127.0.0.1:{self.port}/webhook", data=body, method="POST" + ) + try: + with urllib.request.urlopen(request) as response: + return response.status, response.read() + except urllib.error.HTTPError as error: + return error.code, error.read() + + def test_a_delivery_is_stored(self) -> None: + status, _ = self.post(json.dumps({"event": "ping"}).encode()) + self.assertEqual(status, 200) + self.assertEqual(webhook.DELIVERIES, [{"event": "ping"}]) + + def test_a_malformed_body_returns_400(self) -> None: + status, body = self.post(b"not json") + self.assertEqual(status, 400) + self.assertIn(b"malformed body", body) + self.assertEqual(webhook.DELIVERIES, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/factory/evals/fixture/webhook/__init__.py b/factory/evals/fixture/webhook/__init__.py new file mode 100644 index 0000000..66b5171 --- /dev/null +++ b/factory/evals/fixture/webhook/__init__.py @@ -0,0 +1,42 @@ +"""A tiny webhook delivery service, the factory's end-to-end fixture. + +Accepts POST /webhook with a JSON body and stores each delivery in memory. +Small enough that a factory run against it costs minutes, real enough to +give build's unit and e2e layers something genuine to exercise. +""" + +from __future__ import annotations + +import json +from http.server import BaseHTTPRequestHandler, HTTPServer + +DELIVERIES: list[dict] = [] + + +class WebhookHandler(BaseHTTPRequestHandler): + def do_POST(self) -> None: # noqa: N802 (BaseHTTPRequestHandler's own naming) + if self.path != "/webhook": + self.send_response(404) + self.end_headers() + return + length = int(self.headers.get("Content-Length", 0)) + raw = self.rfile.read(length) + try: + payload = json.loads(raw) + except json.JSONDecodeError: + self.send_response(400) + self.end_headers() + self.wfile.write(b'{"error": "malformed body"}') + return + DELIVERIES.append(payload) + self.send_response(200) + self.end_headers() + self.wfile.write(b'{"stored": true}') + + def log_message(self, format: str, *args: object) -> None: # noqa: A002 + pass # keep the fixture's stdout quiet during a factory run + + +def serve(host: str = "127.0.0.1", port: int = 8765) -> HTTPServer: + server = HTTPServer((host, port), WebhookHandler) + return server diff --git a/factory/evals/fixture/webhook/__main__.py b/factory/evals/fixture/webhook/__main__.py new file mode 100644 index 0000000..db07210 --- /dev/null +++ b/factory/evals/fixture/webhook/__main__.py @@ -0,0 +1,10 @@ +"""Launch command for the fixture: python3 -m webhook.""" + +from __future__ import annotations + +from webhook import serve + +if __name__ == "__main__": + server = serve() + print(f"webhook fixture listening on {server.server_address}") + server.serve_forever() diff --git a/factory/evals/results.md b/factory/evals/results.md new file mode 100644 index 0000000..5cac480 --- /dev/null +++ b/factory/evals/results.md @@ -0,0 +1,7 @@ +# Factory end-to-end run results + +One row per host run, per `README.md`'s procedure. A PR link only ever appears for a run against a real GitHub remote - the fixture's bare origin ends in a pushed branch and a report instead. + +| Date | Host | Phases | Attempts | Repairs | Wall clock | Branch / PR | What parked or nearly parked | +| ---- | ---- | ------ | -------- | ------- | ---------- | ------------ | ----------------------------- | +| (none yet) | | | | | | | | diff --git a/factory/evals/tests/test_fixture_setup.py b/factory/evals/tests/test_fixture_setup.py new file mode 100644 index 0000000..719ed5f --- /dev/null +++ b/factory/evals/tests/test_fixture_setup.py @@ -0,0 +1,72 @@ +"""Unit tests for factory/evals/fixture/setup.sh - offline and deterministic. + +setup.sh is the fixture's own carrier of the .dev/ never-committed invariant +(the run skill's preflight is the carrier in any other consuming repository). +The two paid host runs and the planted-secret variant stay a recorded manual +procedure in factory/evals/README.md; no mocked layer can drive a real host. +""" + +from __future__ import annotations + +import shutil +import subprocess +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +SETUP_SH = REPO_ROOT / "factory" / "evals" / "fixture" / "setup.sh" + + +class FixtureSetupTest(unittest.TestCase): + def setUp(self) -> None: + self.scratch = Path(tempfile.mkdtemp(prefix="fixture-setup-test-")) + self.addCleanup(shutil.rmtree, self.scratch, ignore_errors=True) + dest = self.scratch / "fixture" + result = subprocess.run( + ["bash", str(SETUP_SH), str(dest)], + capture_output=True, + text=True, + check=True, + ) + self.dest = Path(result.stdout.strip()) + + def git(self, *args: str) -> subprocess.CompletedProcess: + return subprocess.run( + ["git", *args], cwd=self.dest, capture_output=True, text=True + ) + + def test_setup_produces_a_main_repo_with_one_commit_and_a_bare_origin(self) -> None: + branch = self.git("branch", "--show-current") + self.assertEqual(branch.stdout.strip(), "main") + + log = self.git("log", "--oneline") + self.assertEqual(len(log.stdout.strip().splitlines()), 1) + + gitignore = (self.dest / ".gitignore").read_text(encoding="utf-8") + self.assertIn(".dev/", gitignore) + + ls_remote = self.git("ls-remote", "origin") + self.assertEqual(ls_remote.returncode, 0, ls_remote.stderr) + self.assertIn("refs/heads/main", ls_remote.stdout) + + def test_push_after_writing_dev_plan_file_leaves_no_dev_path(self) -> None: + plan_dir = self.dest / ".dev" / "plan" + plan_dir.mkdir(parents=True) + (plan_dir / "spec.md").write_text("spec\n", encoding="utf-8") + with (self.dest / "README.md").open("a", encoding="utf-8") as handle: + handle.write("\nextra\n") + + add = self.git("add", "-A") + self.assertEqual(add.returncode, 0, add.stderr) + commit = self.git("commit", "-m", "test commit") + self.assertEqual(commit.returncode, 0, commit.stderr) + push = self.git("push", "origin", "main") + self.assertEqual(push.returncode, 0, push.stderr) + + log = self.git("log", "--name-only", "--pretty=format:") + self.assertNotIn(".dev/", log.stdout) + + +if __name__ == "__main__": + unittest.main() From f52285f2db90b4049bb1baf74986ffdfd20bdb7c Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 12:00:05 +0200 Subject: [PATCH 07/30] fix(harden): declare framework entry points in a vulture whitelist do_POST and log_message are BaseHTTPRequestHandler overrides that http.server calls, so they are deliberate API surface rather than dead code. Record them in tools/harden/vulture_whitelist.py and drop the noqa directives, which named rules this repo does not enable. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/evals/fixture/webhook/__init__.py | 4 ++-- factory/skills/run/scripts/run-state.py | 0 tools/harden/README.md | 4 +++- tools/harden/vulture_whitelist.py | 19 +++++++++++++++++++ 4 files changed, 24 insertions(+), 3 deletions(-) mode change 100644 => 100755 factory/skills/run/scripts/run-state.py create mode 100644 tools/harden/vulture_whitelist.py diff --git a/factory/evals/fixture/webhook/__init__.py b/factory/evals/fixture/webhook/__init__.py index 66b5171..3fd33f0 100644 --- a/factory/evals/fixture/webhook/__init__.py +++ b/factory/evals/fixture/webhook/__init__.py @@ -14,7 +14,7 @@ class WebhookHandler(BaseHTTPRequestHandler): - def do_POST(self) -> None: # noqa: N802 (BaseHTTPRequestHandler's own naming) + def do_POST(self) -> None: if self.path != "/webhook": self.send_response(404) self.end_headers() @@ -33,7 +33,7 @@ def do_POST(self) -> None: # noqa: N802 (BaseHTTPRequestHandler's own naming) self.end_headers() self.wfile.write(b'{"stored": true}') - def log_message(self, format: str, *args: object) -> None: # noqa: A002 + def log_message(self, _format: str, *_args: object) -> None: pass # keep the fixture's stdout quiet during a factory run diff --git a/factory/skills/run/scripts/run-state.py b/factory/skills/run/scripts/run-state.py old mode 100644 new mode 100755 diff --git a/tools/harden/README.md b/tools/harden/README.md index 7c2b354..20b7d46 100644 --- a/tools/harden/README.md +++ b/tools/harden/README.md @@ -7,6 +7,7 @@ Repo-fitted pieces the `ship` gauntlet reuses on every run; the analyzers themse - `complexity_coverage.py RADON_JSON COVERAGE_JSON [--prefix pkg/]` - coverage-weighted complexity (complexity squared scaled down by line coverage) for functions the branch added; the line is 10 per D-complexity-threshold in `docs/decisions.md`. - `flaky.py --runs 5 MODULE ...` - runs unittest modules repeatedly in shuffled order and names any test whose outcome disagrees between runs. - `mutate.py FILE FUNC ... --tests MODULE ...` - function-scoped mutation testing: mutates named functions one change at a time and runs their fast tests. +- `vulture_whitelist.py` - the committed list of names a framework calls rather than repository code, such as `BaseHTTPRequestHandler` overrides; vulture reads it as an ordinary input file, so pass it alongside the scoped files and record deliberate API surface there instead of adding per-line ignores. The gauntlet, from the repository root: @@ -15,7 +16,8 @@ files=$(tools/harden/scope.sh --python) uvx ruff check --output-format concise $files | python3 tools/harden/added_lines.py # static analysis uvx semgrep scan --config p/python --metrics off --quiet $files # static security rules uvx detect-secrets scan $(tools/harden/scope.sh) # secrets -uvx vulture --min-confidence 60 $files | python3 tools/harden/added_lines.py # dead code +uvx vulture --min-confidence 60 $files tools/harden/vulture_whitelist.py | + python3 tools/harden/added_lines.py # dead code npx --yes jscpd@4 --min-tokens 50 --format python scripts # clones ``` diff --git a/tools/harden/vulture_whitelist.py b/tools/harden/vulture_whitelist.py new file mode 100644 index 0000000..184f8dc --- /dev/null +++ b/tools/harden/vulture_whitelist.py @@ -0,0 +1,19 @@ +"""Deliberate API surface that vulture cannot see a caller for. + +Every name below is invoked by a framework or a base class rather than by +repository code, so the dead-code pass would otherwise report it. Vulture +parses this file, it never runs it, so referencing a name here is the whole +mechanism. Add one only when a framework really owns the call, and name that +framework in the comment. +""" + +from factory.evals.fixture.webhook import WebhookHandler + +# http.server dispatches these on a BaseHTTPRequestHandler subclass: do_POST +# per request verb, log_message for every access-log line. +FRAMEWORK_ENTRY_POINTS = ( + WebhookHandler.do_POST, + WebhookHandler.log_message, +) + +__all__ = ["FRAMEWORK_ENTRY_POINTS"] From ab98facdd45ea0ddfefde6b72849ffc4243c7f1b Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 12:52:46 +0200 Subject: [PATCH 08/30] fix(factory): make run-state.py exit codes safe and record attempts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Why: the four confirmed blockers in .dev/factory-plugin/review_1.md. B1: check-result read `stop.kind` with no type guard, so a `stop` that was a string or null crashed and exited 1 - a retryable failure - turning a never-overridable hard stop into a repair and relaunch. A stopped result now always exits 2, however loosely the phase wrote `stop`. B2: nothing ever wrote `phases`, so resume could not tell a crashed attempt from an unstarted one. Adds the `attempt` subcommand, the missing writer for the attempt schema the run protocol documents, and calls it from run/SKILL.md at launch and at close. B3: the not-doing match was a bare `⊘` search, so any line citing the character counted as an entry (17 instead of 12 on this repo's own spec) and diff-spec invented `⊘ line dropped` drift. It now matches an entry opening with the mark, as lint-spec.py parses the same notation. B4: load_state, a missing spec.md, and record's stdin parse crashed or returned 1, colliding with diff-spec's "drift found" code that the orchestrator folds into a judgment. Bad calls now return 3, and every subcommand's exit codes are documented in the module docstring. --- factory/evals/tests/test_run_state.py | 212 +++++++++++++++++- factory/skills/run/SKILL.md | 4 +- factory/skills/run/scripts/run-state.py | 183 ++++++++++++--- plugins/factory/skills/run/SKILL.md | 4 +- .../factory/skills/run/scripts/run-state.py | 183 ++++++++++++--- 5 files changed, 510 insertions(+), 76 deletions(-) mode change 100644 => 100755 plugins/factory/skills/run/scripts/run-state.py diff --git a/factory/evals/tests/test_run_state.py b/factory/evals/tests/test_run_state.py index 83a9ffa..1cd6a6a 100644 --- a/factory/evals/tests/test_run_state.py +++ b/factory/evals/tests/test_run_state.py @@ -8,6 +8,7 @@ from __future__ import annotations import json +import os import subprocess import sys import tempfile @@ -44,13 +45,56 @@ """ -def run_state(cwd: Path, *args: str, stdin: str | None = None) -> subprocess.CompletedProcess: +SPEC_NUMBERED_PREAMBLE = """# Fixture plan + +7. A numbered line before any section + tests: [unit] preamble -> ignored + +## Change plan + +1. Change set one + a. `a.py` - does a thing + tests: [unit] a -> b; [unit] c -> d + +2. Change set two + a. `b.py` - does another thing + tests: [unit] e -> f +""" + + +SPEC_CITING_THE_CHARACTER = """# Fixture plan + +## Research + +D-example: An example decision? + ✓ do it - the state records every `⊘` line at handoff + ⊘ not doing - reason one + +## Scope + +### Non-goals + +- ⊘ something else - reason two + +## Change plan + +1. Change set one + a. `a.py` - does a thing + tests: [unit] a -> b +""" + + +def run_state( + cwd: Path, *args: str, stdin: str | None = None, env: dict[str, str] | None = None +) -> subprocess.CompletedProcess: return subprocess.run( [sys.executable, str(SCRIPT), *args], cwd=cwd, + check=False, input=stdin, capture_output=True, text=True, + env=env, ) @@ -71,6 +115,12 @@ def init_plan(self) -> None: result = run_state(self.cwd, "init", "fixture-plan", "--request", "do it", "--base", "main") self.assertEqual(result.returncode, 0, result.stderr) + def handoff_state(self) -> dict: + """Run a successful handoff on the fixture plan and return the state it wrote.""" + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + self.assertEqual(result.returncode, 0, result.stderr) + return json.loads((self.plan_dir / "factory-run.json").read_text()) + def test_init_twice_exits_1_naming_existing_state(self) -> None: self.init_plan() result = run_state(self.cwd, "init", "fixture-plan", "--request", "do it", "--base", "main") @@ -80,9 +130,7 @@ def test_init_twice_exits_1_naming_existing_state(self) -> None: def test_handoff_records_scenarios_and_not_doing_lines(self) -> None: self.init_plan() self.write_spec() - result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") - self.assertEqual(result.returncode, 0, result.stderr) - state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state = self.handoff_state() self.assertIsNotNone(state["spec_sha256"]) scenario_texts = state["scenario_texts"] self.assertEqual(sorted(scenario_texts), ["1", "2"]) @@ -206,6 +254,162 @@ def test_check_result_unknown_extra_key_exits_0(self) -> None: result = run_state(self.cwd, "check-result", str(path)) self.assertEqual(result.returncode, 0) + def test_check_result_unrecognized_status_exits_1(self) -> None: + path = self.write_result({"status": "in-progress"}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 1) + self.assertIn("unrecognized status", result.stdout) + + def test_init_creates_the_whole_dev_tree(self) -> None: + with tempfile.TemporaryDirectory() as bare: + cwd = Path(bare) + result = run_state(cwd, "init", "fresh-plan", "--request", "do it", "--base", "main") + self.assertEqual(result.returncode, 0, result.stderr) + self.assertTrue((cwd / ".dev" / "fresh-plan" / "factory-run.json").is_file()) + + def test_state_file_is_written_as_indented_json(self) -> None: + self.init_plan() + text = (self.plan_dir / "factory-run.json").read_text(encoding="utf-8") + self.assertIn('\n "plan": "fixture-plan"', text) + self.assertIn('\n "scope": []', text) + self.assertTrue(text.endswith("}\n")) + + def test_handoff_ignores_numbered_lines_outside_the_change_plan(self) -> None: + self.init_plan() + self.write_spec(SPEC_NUMBERED_PREAMBLE) + state = self.handoff_state() + self.assertEqual(sorted(state["scenario_texts"]), ["1", "2"]) + self.assertEqual(sum(len(v) for v in state["scenario_texts"].values()), 3) + + def test_handoff_without_a_spec_exits_1(self) -> None: + self.init_plan() + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + self.assertEqual(result.returncode, 1) + self.assertIn("no spec.md", result.stdout) + + def test_handoff_without_dirty_flag_records_git_status_paths(self) -> None: + subprocess.run(["git", "init", "-q"], cwd=self.cwd, check=True, capture_output=True) + (self.cwd / "foo.py").write_text("x = 1\n", encoding="utf-8") + self.init_plan() + self.write_spec() + state = self.handoff_state() + self.assertEqual(sorted(state["dirty_files"]), [".dev/", "foo.py"]) + + def test_handoff_records_no_dirty_files_when_git_fails(self) -> None: + bin_dir = self.cwd / "fakebin" + bin_dir.mkdir() + fake_git = bin_dir / "git" + fake_git.write_text('#!/bin/sh\necho " M fake.py"\nexit 128\n', encoding="utf-8") + fake_git.chmod(0o755) + env = {**os.environ, "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}"} + self.init_plan() + self.write_spec() + result = run_state( + self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan", env=env + ) + self.assertEqual(result.returncode, 0, result.stderr) + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + self.assertEqual(state["dirty_files"], []) + + def test_record_rejects_malformed_stdin_json(self) -> None: + self.init_plan() + before = (self.plan_dir / "factory-run.json").read_text() + result = run_state(self.cwd, "record", "fixture-plan", stdin="{not json") + self.assertEqual(result.returncode, 3) + self.assertIn("invalid JSON", result.stdout) + self.assertEqual(before, (self.plan_dir / "factory-run.json").read_text()) + + def test_check_result_stopped_with_a_string_stop_still_exits_2(self) -> None: + path = self.write_result({"status": "stopped", "stop": "secret.found"}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 2) + self.assertIn("secret.found", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_check_result_stopped_without_a_stop_object_still_exits_2(self) -> None: + path = self.write_result({"status": "stopped", "stop": None, "reason": "destructive"}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 2) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_records_a_launch_then_its_result(self) -> None: + self.init_plan() + launched = run_state(self.cwd, "attempt", "fixture-plan", "build") + self.assertEqual(launched.returncode, 0, launched.stderr) + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + self.assertEqual(state["phases"]["build"], [{"status": "launched", "result": None}]) + path = self.write_result({"schema": "factory.result/1", "status": "done"}) + closed = run_state( + self.cwd, "attempt", "fixture-plan", "build", "--status", "done", "--result", str(path) + ) + self.assertEqual(closed.returncode, 0, closed.stderr) + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + self.assertEqual(state["phases"]["build"][0]["status"], "done") + self.assertEqual(state["phases"]["build"][0]["result"]["status"], "done") + shown = run_state(self.cwd, "show", "fixture-plan") + self.assertIn("build attempt 1: done", shown.stdout) + + def test_attempt_without_a_launch_to_close_exits_3(self) -> None: + self.init_plan() + result = run_state(self.cwd, "attempt", "fixture-plan", "build", "--status", "failed") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_a_line_citing_the_not_doing_character_is_not_an_entry(self) -> None: + self.init_plan() + self.write_spec(SPEC_CITING_THE_CHARACTER) + state = self.handoff_state() + self.assertEqual(len(state["not_doing_lines"]), 2) + reworded = SPEC_CITING_THE_CHARACTER.replace( + "the state records every `⊘` line at handoff", "the state records the `⊘` lines" + ) + self.write_spec(reworded) + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 0, result.stdout) + + def test_diff_spec_without_a_state_file_exits_3(self) -> None: + self.write_spec() + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_diff_spec_with_a_truncated_state_file_exits_3(self) -> None: + self.init_plan() + self.write_spec() + (self.plan_dir / "factory-run.json").write_text('{"plan": "fix', encoding="utf-8") + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_diff_spec_without_a_spec_exits_3(self) -> None: + self.init_plan() + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_record_with_a_corrupt_state_file_exits_3(self) -> None: + self.init_plan() + (self.plan_dir / "factory-run.json").write_text("{", encoding="utf-8") + result = run_state( + self.cwd, + "record", + "fixture-plan", + stdin=json.dumps({"phase": "build", "attempt": 1, "action": "advance"}), + ) + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_show_with_a_missing_state_file_exits_3(self) -> None: + result = run_state(self.cwd, "show", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_show_exits_0(self) -> None: + self.init_plan() + result = run_state(self.cwd, "show", "fixture-plan") + self.assertEqual(result.returncode, 0, result.stderr) + self.assertIn("scope: no attempts", result.stdout) + if __name__ == "__main__": unittest.main() diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index a6ad360..3605e02 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -38,9 +38,9 @@ At the go: create branch `factory/{plan}` from the base branch. Run `run-state.p For each phase in order (`scope-review`, `build`, `ship`): -1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. +1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. Record the launch with `run-state.py attempt {plan} {phase}` so a crashed attempt is distinguishable from an unstarted one on resume. 2. Wait for the host's completion signal - a batch in flight is not over. -3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). +3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`. 4. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). 5. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. 6. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. diff --git a/factory/skills/run/scripts/run-state.py b/factory/skills/run/scripts/run-state.py index fc9502a..f6c2fc0 100755 --- a/factory/skills/run/scripts/run-state.py +++ b/factory/skills/run/scripts/run-state.py @@ -1,17 +1,28 @@ #!/usr/bin/env python3 -"""Own factory-run.json: init, handoff, diff-spec, record, check-result, show. +"""Own factory-run.json: init, handoff, diff-spec, attempt, record, check-result, show. Usage: run-state.py init --request --base run-state.py handoff --branch [--spec ] [--dirty ...] run-state.py diff-spec [--spec ] + run-state.py attempt [--status launched|done|failed|stopped] [--result ] run-state.py record # reads one JSON decision object on stdin run-state.py check-result run-state.py show -The orchestrator loops on the exit code of every subcommand, so a malformed -state file or a bad call never reaches a judgment. Every function here stays -at or under cyclomatic complexity 10 (D-complexity-threshold). +Exit codes, which the orchestrator branches on: + init 0 written; 1 a state file already exists + handoff 0 recorded; 1 no spec.md at the path; 3 unusable state file + diff-spec 0 no drift; 1 drift found; 3 unusable state file or no spec.md + attempt 0 recorded; 3 unusable state file, or no launched attempt to close + record 0 recorded; 1 action outside the four; 3 unreadable stdin JSON + or unusable state file + check-result 0 done; 1 failed; 2 stopped; 3 missing or unparseable result file + show 0 printed; 3 unusable state file + +3 always means the call itself could not be carried out, never a phase outcome, +so a malformed state file or a bad call never reaches a judgment. Every function +here stays at or under cyclomatic complexity 10 (D-complexity-threshold). """ from __future__ import annotations @@ -27,10 +38,13 @@ HEADING = re.compile(r"^##\s+(.*?)\s*$") CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) -NOT_DOING = re.compile(r"⊘") +NOT_DOING = re.compile(r"^\s*(?:[-*]\s+)?⊘\s") ACTIONS = {"advance", "repair", "relaunch", "end"} +PHASES = ("scope", "scope-review", "build", "ship") +ATTEMPT_STATUSES = ("launched", "done", "failed", "stopped") STATE_NAME = "factory-run.json" +BAD_CALL = 3 def state_path(plan: str) -> Path: @@ -41,27 +55,51 @@ def spec_path(plan: str, override: str | None) -> Path: return Path(override) if override else Path(".dev") / plan / "spec.md" -def load_state(plan: str) -> dict: +def load_state(plan: str) -> dict | None: + """The parsed state file, or None (with the reason printed) when it is unusable.""" path = state_path(plan) - return json.loads(path.read_text(encoding="utf-8")) + try: + state = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + print(f"unusable state file {path}: {exc}") + return None + if not isinstance(state, dict): + print(f"unusable state file {path}: not a JSON object") + return None + return state def save_state(plan: str, state: dict) -> None: state_path(plan).write_text(json.dumps(state, indent=2) + "\n", encoding="utf-8") -def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list[str]]: - """Scenario texts per change set, and every ⊘ line, from the current spec.""" +def parse_scenario_list(value: str) -> list[str] | None: + """The scenarios named on a `tests:` line, or None when it names none.""" + value = value.strip() + if value.lower().startswith("none"): + return None + return [s.strip() for s in value.split(";") if s.strip()] + + +def not_doing_lines(lines: list[str]) -> list[str]: + """Every ⊘ entry in the spec, stripped, in order. + + An entry opens with the mark, as a decision alternative or a list bullet + (the notation lint-spec.py parses); a line that merely cites the character + in prose is not one. + """ + return [line.strip() for line in lines if NOT_DOING.match(line)] + + +def scenario_texts(lines: list[str]) -> dict[str, list[str]]: + """Scenario texts per change set, read from the Change plan section.""" texts: dict[str, list[str]] = {} - not_doing: list[str] = [] in_plan = False current: str | None = None - for line in spec.read_text(encoding="utf-8").splitlines(): + for line in lines: heading = HEADING.match(line) if heading: in_plan = heading.group(1).lower() == "change plan" - if NOT_DOING.search(line): - not_doing.append(line.strip()) if not in_plan: continue change_set = CHANGE_SET.match(line) @@ -72,11 +110,16 @@ def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list tests = TESTS.match(line) if not tests or current is None: continue - value = tests.group(1).strip() - if value.lower().startswith("none"): - continue - texts[current] = [s.strip() for s in value.split(";") if s.strip()] - return texts, not_doing + scenarios = parse_scenario_list(tests.group(1)) + if scenarios is not None: + texts[current] = scenarios + return texts + + +def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list[str]]: + """Scenario texts per change set, and every ⊘ line, from the current spec.""" + lines = spec.read_text(encoding="utf-8").splitlines() + return scenario_texts(lines), not_doing_lines(lines) def git_dirty_files() -> list[str]: @@ -119,6 +162,8 @@ def cmd_handoff(args: argparse.Namespace) -> int: print(f"no spec.md at {spec}") return 1 state = load_state(args.plan) + if state is None: + return BAD_CALL digest = hashlib.sha256(spec.read_bytes()).hexdigest() texts, not_doing = scenario_texts_and_not_doing(spec) state["branch"] = args.branch @@ -131,13 +176,10 @@ def cmd_handoff(args: argparse.Namespace) -> int: return 0 -def cmd_diff_spec(args: argparse.Namespace) -> int: - state = load_state(args.plan) - spec = spec_path(args.plan, args.spec) - texts, not_doing = scenario_texts_and_not_doing(spec) +def scenario_drift(stored: dict[str, list[str]], texts: dict[str, list[str]]) -> list[str]: + """One message per change set whose approved scenarios were dropped or reworded.""" problems: list[str] = [] - stored_texts: dict[str, list[str]] = state.get("scenario_texts", {}) - for change_set, before in stored_texts.items(): + for change_set, before in stored.items(): after = texts.get(change_set, []) dropped = [t for t in before if t not in after] added = [t for t in after if t not in before] @@ -145,10 +187,25 @@ def cmd_diff_spec(args: argparse.Namespace) -> int: problems.append( f"change set {change_set}: scenario dropped or reworded: before {dropped!r}, now {added!r}" ) - stored_not_doing = state.get("not_doing_lines", []) - for line in stored_not_doing: - if line not in not_doing: - problems.append(f"⊘ line dropped: {line!r}") + return problems + + +def not_doing_drift(stored: list[str], not_doing: list[str]) -> list[str]: + """One message per stored ⊘ line that is no longer in the spec.""" + return [f"⊘ line dropped: {line!r}" for line in stored if line not in not_doing] + + +def cmd_diff_spec(args: argparse.Namespace) -> int: + state = load_state(args.plan) + if state is None: + return BAD_CALL + spec = spec_path(args.plan, args.spec) + if not spec.is_file(): + print(f"no spec.md at {spec}") + return BAD_CALL + texts, not_doing = scenario_texts_and_not_doing(spec) + problems = scenario_drift(state.get("scenario_texts", {}), texts) + problems += not_doing_drift(state.get("not_doing_lines", []), not_doing) for message in problems: print(message) return 1 if problems else 0 @@ -160,12 +217,17 @@ def cmd_record(args: argparse.Namespace) -> int: decision = json.loads(raw) except json.JSONDecodeError as exc: print(f"invalid JSON on stdin: {exc}") - return 1 + return BAD_CALL + if not isinstance(decision, dict): + print("invalid JSON on stdin: not a decision object") + return BAD_CALL action = decision.get("action") if action not in ACTIONS: print(f"action {action!r} is not one of {sorted(ACTIONS)}") return 1 state = load_state(args.plan) + if state is None: + return BAD_CALL entry = { "phase": decision.get("phase"), "attempt": decision.get("attempt"), @@ -190,16 +252,42 @@ def cmd_record(args: argparse.Namespace) -> int: return 0 +def read_result(path: Path) -> dict | None: + """The parsed result file, or None when it is missing, unreadable, or not an object.""" + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None + return data if isinstance(data, dict) else None + + +def stop_kind(data: dict) -> str: + """The stop kind of a stopped result, however loosely the phase wrote `stop`. + + A hard stop is never overridable, so a `stop` that is not the documented + object still ends as a stop, never as a retryable failure. + """ + stop = data.get("stop") + if isinstance(stop, dict): + return str(stop.get("kind", "")) + if isinstance(stop, str): + return stop + return "" + + def cmd_check_result(args: argparse.Namespace) -> int: path = Path(args.path) if not path.is_file(): print(f"missing result file: {path}") - return 3 + return BAD_CALL try: data = json.loads(path.read_text(encoding="utf-8")) - except json.JSONDecodeError as exc: + except (OSError, json.JSONDecodeError) as exc: print(f"unparseable result file {path}: {exc}") - return 3 + return BAD_CALL + if not isinstance(data, dict): + print(f"unparseable result file {path}: not a JSON object") + return BAD_CALL status = data.get("status") if status == "done": print("done") @@ -208,15 +296,35 @@ def cmd_check_result(args: argparse.Namespace) -> int: print(f"failed: {data.get('reason', '')}") return 1 if status == "stopped": - kind = data.get("stop", {}).get("kind", "") - print(f"stopped: {kind}") + print(f"stopped: {stop_kind(data)}") return 2 print(f"failed: unrecognized status {status!r}") return 1 +def cmd_attempt(args: argparse.Namespace) -> int: + state = load_state(args.plan) + if state is None: + return BAD_CALL + attempts = state.setdefault("phases", {}).setdefault(args.phase, []) + if args.status == "launched": + attempts.append({"status": "launched", "result": None}) + elif not attempts: + print(f"no launched attempt of {args.phase} to close") + return BAD_CALL + else: + attempts[-1]["status"] = args.status + if args.result: + attempts[-1]["result"] = read_result(Path(args.result)) + save_state(args.plan, state) + print(f"{args.phase} attempt {len(attempts)}: {args.status}") + return 0 + + def cmd_show(args: argparse.Namespace) -> int: state = load_state(args.plan) + if state is None: + return BAD_CALL for phase, attempts in state.get("phases", {}).items(): if not attempts: print(f"{phase}: no attempts") @@ -247,6 +355,12 @@ def build_parser() -> argparse.ArgumentParser: diff_p.add_argument("plan") diff_p.add_argument("--spec") + attempt_p = sub.add_parser("attempt") + attempt_p.add_argument("plan") + attempt_p.add_argument("phase", choices=PHASES) + attempt_p.add_argument("--status", choices=ATTEMPT_STATUSES, default="launched") + attempt_p.add_argument("--result") + record_p = sub.add_parser("record") record_p.add_argument("plan") @@ -265,6 +379,7 @@ def main(argv: list[str]) -> int: "init": cmd_init, "handoff": cmd_handoff, "diff-spec": cmd_diff_spec, + "attempt": cmd_attempt, "record": cmd_record, "check-result": cmd_check_result, "show": cmd_show, diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index 05dd6ab..f4e424b 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -37,9 +37,9 @@ At the go: create branch `factory/{plan}` from the base branch. Run `run-state.p For each phase in order (`scope-review`, `build`, `ship`): -1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. +1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. Record the launch with `run-state.py attempt {plan} {phase}` so a crashed attempt is distinguishable from an unstarted one on resume. 2. Wait for the host's completion signal - a batch in flight is not over. -3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). +3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`. 4. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). 5. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. 6. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. diff --git a/plugins/factory/skills/run/scripts/run-state.py b/plugins/factory/skills/run/scripts/run-state.py old mode 100644 new mode 100755 index fc9502a..f6c2fc0 --- a/plugins/factory/skills/run/scripts/run-state.py +++ b/plugins/factory/skills/run/scripts/run-state.py @@ -1,17 +1,28 @@ #!/usr/bin/env python3 -"""Own factory-run.json: init, handoff, diff-spec, record, check-result, show. +"""Own factory-run.json: init, handoff, diff-spec, attempt, record, check-result, show. Usage: run-state.py init --request --base run-state.py handoff --branch [--spec ] [--dirty ...] run-state.py diff-spec [--spec ] + run-state.py attempt [--status launched|done|failed|stopped] [--result ] run-state.py record # reads one JSON decision object on stdin run-state.py check-result run-state.py show -The orchestrator loops on the exit code of every subcommand, so a malformed -state file or a bad call never reaches a judgment. Every function here stays -at or under cyclomatic complexity 10 (D-complexity-threshold). +Exit codes, which the orchestrator branches on: + init 0 written; 1 a state file already exists + handoff 0 recorded; 1 no spec.md at the path; 3 unusable state file + diff-spec 0 no drift; 1 drift found; 3 unusable state file or no spec.md + attempt 0 recorded; 3 unusable state file, or no launched attempt to close + record 0 recorded; 1 action outside the four; 3 unreadable stdin JSON + or unusable state file + check-result 0 done; 1 failed; 2 stopped; 3 missing or unparseable result file + show 0 printed; 3 unusable state file + +3 always means the call itself could not be carried out, never a phase outcome, +so a malformed state file or a bad call never reaches a judgment. Every function +here stays at or under cyclomatic complexity 10 (D-complexity-threshold). """ from __future__ import annotations @@ -27,10 +38,13 @@ HEADING = re.compile(r"^##\s+(.*?)\s*$") CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) -NOT_DOING = re.compile(r"⊘") +NOT_DOING = re.compile(r"^\s*(?:[-*]\s+)?⊘\s") ACTIONS = {"advance", "repair", "relaunch", "end"} +PHASES = ("scope", "scope-review", "build", "ship") +ATTEMPT_STATUSES = ("launched", "done", "failed", "stopped") STATE_NAME = "factory-run.json" +BAD_CALL = 3 def state_path(plan: str) -> Path: @@ -41,27 +55,51 @@ def spec_path(plan: str, override: str | None) -> Path: return Path(override) if override else Path(".dev") / plan / "spec.md" -def load_state(plan: str) -> dict: +def load_state(plan: str) -> dict | None: + """The parsed state file, or None (with the reason printed) when it is unusable.""" path = state_path(plan) - return json.loads(path.read_text(encoding="utf-8")) + try: + state = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + print(f"unusable state file {path}: {exc}") + return None + if not isinstance(state, dict): + print(f"unusable state file {path}: not a JSON object") + return None + return state def save_state(plan: str, state: dict) -> None: state_path(plan).write_text(json.dumps(state, indent=2) + "\n", encoding="utf-8") -def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list[str]]: - """Scenario texts per change set, and every ⊘ line, from the current spec.""" +def parse_scenario_list(value: str) -> list[str] | None: + """The scenarios named on a `tests:` line, or None when it names none.""" + value = value.strip() + if value.lower().startswith("none"): + return None + return [s.strip() for s in value.split(";") if s.strip()] + + +def not_doing_lines(lines: list[str]) -> list[str]: + """Every ⊘ entry in the spec, stripped, in order. + + An entry opens with the mark, as a decision alternative or a list bullet + (the notation lint-spec.py parses); a line that merely cites the character + in prose is not one. + """ + return [line.strip() for line in lines if NOT_DOING.match(line)] + + +def scenario_texts(lines: list[str]) -> dict[str, list[str]]: + """Scenario texts per change set, read from the Change plan section.""" texts: dict[str, list[str]] = {} - not_doing: list[str] = [] in_plan = False current: str | None = None - for line in spec.read_text(encoding="utf-8").splitlines(): + for line in lines: heading = HEADING.match(line) if heading: in_plan = heading.group(1).lower() == "change plan" - if NOT_DOING.search(line): - not_doing.append(line.strip()) if not in_plan: continue change_set = CHANGE_SET.match(line) @@ -72,11 +110,16 @@ def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list tests = TESTS.match(line) if not tests or current is None: continue - value = tests.group(1).strip() - if value.lower().startswith("none"): - continue - texts[current] = [s.strip() for s in value.split(";") if s.strip()] - return texts, not_doing + scenarios = parse_scenario_list(tests.group(1)) + if scenarios is not None: + texts[current] = scenarios + return texts + + +def scenario_texts_and_not_doing(spec: Path) -> tuple[dict[str, list[str]], list[str]]: + """Scenario texts per change set, and every ⊘ line, from the current spec.""" + lines = spec.read_text(encoding="utf-8").splitlines() + return scenario_texts(lines), not_doing_lines(lines) def git_dirty_files() -> list[str]: @@ -119,6 +162,8 @@ def cmd_handoff(args: argparse.Namespace) -> int: print(f"no spec.md at {spec}") return 1 state = load_state(args.plan) + if state is None: + return BAD_CALL digest = hashlib.sha256(spec.read_bytes()).hexdigest() texts, not_doing = scenario_texts_and_not_doing(spec) state["branch"] = args.branch @@ -131,13 +176,10 @@ def cmd_handoff(args: argparse.Namespace) -> int: return 0 -def cmd_diff_spec(args: argparse.Namespace) -> int: - state = load_state(args.plan) - spec = spec_path(args.plan, args.spec) - texts, not_doing = scenario_texts_and_not_doing(spec) +def scenario_drift(stored: dict[str, list[str]], texts: dict[str, list[str]]) -> list[str]: + """One message per change set whose approved scenarios were dropped or reworded.""" problems: list[str] = [] - stored_texts: dict[str, list[str]] = state.get("scenario_texts", {}) - for change_set, before in stored_texts.items(): + for change_set, before in stored.items(): after = texts.get(change_set, []) dropped = [t for t in before if t not in after] added = [t for t in after if t not in before] @@ -145,10 +187,25 @@ def cmd_diff_spec(args: argparse.Namespace) -> int: problems.append( f"change set {change_set}: scenario dropped or reworded: before {dropped!r}, now {added!r}" ) - stored_not_doing = state.get("not_doing_lines", []) - for line in stored_not_doing: - if line not in not_doing: - problems.append(f"⊘ line dropped: {line!r}") + return problems + + +def not_doing_drift(stored: list[str], not_doing: list[str]) -> list[str]: + """One message per stored ⊘ line that is no longer in the spec.""" + return [f"⊘ line dropped: {line!r}" for line in stored if line not in not_doing] + + +def cmd_diff_spec(args: argparse.Namespace) -> int: + state = load_state(args.plan) + if state is None: + return BAD_CALL + spec = spec_path(args.plan, args.spec) + if not spec.is_file(): + print(f"no spec.md at {spec}") + return BAD_CALL + texts, not_doing = scenario_texts_and_not_doing(spec) + problems = scenario_drift(state.get("scenario_texts", {}), texts) + problems += not_doing_drift(state.get("not_doing_lines", []), not_doing) for message in problems: print(message) return 1 if problems else 0 @@ -160,12 +217,17 @@ def cmd_record(args: argparse.Namespace) -> int: decision = json.loads(raw) except json.JSONDecodeError as exc: print(f"invalid JSON on stdin: {exc}") - return 1 + return BAD_CALL + if not isinstance(decision, dict): + print("invalid JSON on stdin: not a decision object") + return BAD_CALL action = decision.get("action") if action not in ACTIONS: print(f"action {action!r} is not one of {sorted(ACTIONS)}") return 1 state = load_state(args.plan) + if state is None: + return BAD_CALL entry = { "phase": decision.get("phase"), "attempt": decision.get("attempt"), @@ -190,16 +252,42 @@ def cmd_record(args: argparse.Namespace) -> int: return 0 +def read_result(path: Path) -> dict | None: + """The parsed result file, or None when it is missing, unreadable, or not an object.""" + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None + return data if isinstance(data, dict) else None + + +def stop_kind(data: dict) -> str: + """The stop kind of a stopped result, however loosely the phase wrote `stop`. + + A hard stop is never overridable, so a `stop` that is not the documented + object still ends as a stop, never as a retryable failure. + """ + stop = data.get("stop") + if isinstance(stop, dict): + return str(stop.get("kind", "")) + if isinstance(stop, str): + return stop + return "" + + def cmd_check_result(args: argparse.Namespace) -> int: path = Path(args.path) if not path.is_file(): print(f"missing result file: {path}") - return 3 + return BAD_CALL try: data = json.loads(path.read_text(encoding="utf-8")) - except json.JSONDecodeError as exc: + except (OSError, json.JSONDecodeError) as exc: print(f"unparseable result file {path}: {exc}") - return 3 + return BAD_CALL + if not isinstance(data, dict): + print(f"unparseable result file {path}: not a JSON object") + return BAD_CALL status = data.get("status") if status == "done": print("done") @@ -208,15 +296,35 @@ def cmd_check_result(args: argparse.Namespace) -> int: print(f"failed: {data.get('reason', '')}") return 1 if status == "stopped": - kind = data.get("stop", {}).get("kind", "") - print(f"stopped: {kind}") + print(f"stopped: {stop_kind(data)}") return 2 print(f"failed: unrecognized status {status!r}") return 1 +def cmd_attempt(args: argparse.Namespace) -> int: + state = load_state(args.plan) + if state is None: + return BAD_CALL + attempts = state.setdefault("phases", {}).setdefault(args.phase, []) + if args.status == "launched": + attempts.append({"status": "launched", "result": None}) + elif not attempts: + print(f"no launched attempt of {args.phase} to close") + return BAD_CALL + else: + attempts[-1]["status"] = args.status + if args.result: + attempts[-1]["result"] = read_result(Path(args.result)) + save_state(args.plan, state) + print(f"{args.phase} attempt {len(attempts)}: {args.status}") + return 0 + + def cmd_show(args: argparse.Namespace) -> int: state = load_state(args.plan) + if state is None: + return BAD_CALL for phase, attempts in state.get("phases", {}).items(): if not attempts: print(f"{phase}: no attempts") @@ -247,6 +355,12 @@ def build_parser() -> argparse.ArgumentParser: diff_p.add_argument("plan") diff_p.add_argument("--spec") + attempt_p = sub.add_parser("attempt") + attempt_p.add_argument("plan") + attempt_p.add_argument("phase", choices=PHASES) + attempt_p.add_argument("--status", choices=ATTEMPT_STATUSES, default="launched") + attempt_p.add_argument("--result") + record_p = sub.add_parser("record") record_p.add_argument("plan") @@ -265,6 +379,7 @@ def main(argv: list[str]) -> int: "init": cmd_init, "handoff": cmd_handoff, "diff-spec": cmd_diff_spec, + "attempt": cmd_attempt, "record": cmd_record, "check-result": cmd_check_result, "show": cmd_show, From c0d50b2428db7debc03b11569b3b3ea83157f4d0 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 12:53:08 +0200 Subject: [PATCH 09/30] fix(factory): drop deferrals from the scope-review copy Why: review_1.md blockers B5 and B6. The factory scope-review copy kept dev's report template and authority rules verbatim, so it still offered "APPROVED WITH DEFERRALS" and a "## Deferred" section that change set 3b removed and D-scope-review-deferrals rejects. That verdict also contains the substring the run skill's done criterion matched, so a deferrals verdict would have advanced the run to build on an unsettled spec (B5). The same copy queued escalations for an "interview below" that this copy replaced with the unattended section 4 (B6). What: the report template now writes only "Verdict: APPROVED", records escalations as decided under policy, and carries a "## Rescope" section for the failed/rescope ending instead of "## Deferred". judgment.md now requires a whole line exactly equal to "Verdict: APPROVED", checked with grep -x, so a longer verdict cannot pass. The frontmatter, overview and authority rules repoint at section 4 and factory policy; the constraint that refine agents never auto-apply an escalation is kept. --- factory/skills/run/references/judgment.md | 2 +- factory/skills/scope-review/SKILL.md | 23 +++++++++++-------- .../factory/skills/run/references/judgment.md | 2 +- plugins/factory/skills/scope-review/SKILL.md | 23 +++++++++++-------- 4 files changed, 28 insertions(+), 22 deletions(-) diff --git a/factory/skills/run/references/judgment.md b/factory/skills/run/references/judgment.md index a800b8a..f30a2e0 100644 --- a/factory/skills/run/references/judgment.md +++ b/factory/skills/run/references/judgment.md @@ -8,7 +8,7 @@ Per phase, the evidence commands the orchestrator re-runs as proof and what a `d ## scope-review -`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a `Verdict: APPROVED` line with a `Rounds:` line. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. +`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a line whose whole text is exactly `Verdict: APPROVED` - nothing after it - with a `Rounds:` line. Check it as a whole line, `grep -cx 'Verdict: APPROVED'`, never as a substring: a longer verdict such as `Verdict: APPROVED WITH DEFERRALS` is not a `done`. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. ## build diff --git a/factory/skills/scope-review/SKILL.md b/factory/skills/scope-review/SKILL.md index 728dba1..0ba8b55 100644 --- a/factory/skills/scope-review/SKILL.md +++ b/factory/skills/scope-review/SKILL.md @@ -1,6 +1,6 @@ --- name: scope-review -description: Review and auto-refine a settled spec before build starts - a fresh-context agent panel checks the plan against the actual repo for infeasible change sets, missing failure paths, semantic contradictions, and untestable scenarios, then verified findings are applied to spec.md (and, where they touch settled decisions or cross-boundary invariants, to docs/decisions.md and docs/contracts.md) by refine agents and the panel re-runs; findings only the user can decide are asked as questions at the end and their answers applied and promoted the same way, so a finished run hands build a spec - and ledger - ready to implement, with no separate scope pass needed to promote the changes. Use after scope settles a spec and before build implements it. +description: Review and auto-refine a settled spec before build starts - a fresh-context agent panel checks the plan against the actual repo for infeasible change sets, missing failure paths, semantic contradictions, and untestable scenarios, then verified findings are applied to spec.md (and, where they touch settled decisions or cross-boundary invariants, to docs/decisions.md and docs/contracts.md) by refine agents and the panel re-runs; findings that exceed refine authority are decided at the end of the run under factory policy and their answers applied and promoted the same way, so a finished run hands build a spec - and ledger - ready to implement, with no separate scope pass needed to promote the changes. Use after scope settles a spec and before build implements it. disable-model-invocation: true --- @@ -8,7 +8,7 @@ disable-model-invocation: true Review the spec with agents that did not write it, then refine it in place - before any implementation exists, and without a human in the loop. A defect caught here costs a spec edit; the same defect after build costs a re-implementation, so this loop runs to completion on its own and ends with a spec build can start on - not a findings list to triage, and not a handoff back to `scope`. -The few findings only the user can decide are asked as questions at the end of the run, and the answers are applied before it finishes. +The few findings that exceed refine authority are answered in section 4 under factory policy, and those answers are applied before the run finishes. The panel judges the plan against the actual repo, not against the conversation that produced it. You are the orchestrator: run tools, dispatch agents, apply the loop, report - your own reading of the spec is not a lens, and findings reach the spec only through verification. This skill edits `spec.md`, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. @@ -39,7 +39,7 @@ A third panel means the refinements are churning, not converging - stop and esca 5. **Re-gate.** Loop `lint-spec.py` until clean, fixing mechanical fallout with the same refine agents; then start the next round so fresh eyes judge the refined spec. Exit the loop when a panel raises nothing refinable; two rounds of the same finding surviving refinement is itself an escalation. -Escalations collected across the rounds go to the interview below, not to a handoff. +Escalations collected across the rounds go to section 4 below, not to a handoff. ## 3. Authority and refinement rules @@ -51,10 +51,10 @@ Refine agents resolve conflicts by this order - each level beats everything belo 4. The spec's prose. Refinable: false premises about the repo (rewrite the entry against the real code, including the extra work that reveals), change sets contradicting their linked decisions, missing test scenarios for stated invariants and failure paths, untestable scenarios (replace with one provable at that layer), and gaps whose resolution is forced once the repo is consulted. -Escalations - never auto-applied, queued for the interview instead: anything that would flip a `✓` decision to a rejected alternative, change the user-visible scope or behavior, add or drop a dependency, or contradict the user's recorded intent. +Escalations - never auto-applied by a refine agent, queued for section 4 instead: anything that would flip a `✓` decision to a rejected alternative, change the user-visible scope or behavior, add or drop a dependency, or contradict the user's recorded intent. A refine agent that cannot fix its finding without crossing that line marks it escalated and leaves the spec alone. Refinements follow `scope`'s notation: decision entries keep their slugs and marks, change sets keep their numbering, new scenarios carry layer tags. -A refinement that adds or rewrites a decision entry, or a cross-boundary invariant, is promoted to the ledger immediately per step 5 - it does not wait for the interview. +A refinement that adds or rewrites a decision entry, or a cross-boundary invariant, is promoted to the ledger immediately per step 5 - it does not wait for section 4. ## 4. Resolve escalations under factory policy @@ -100,7 +100,7 @@ Write `.dev/{plan-name}/spec-review_N.md` at the next free index, one per run, c ```markdown # Spec review N - {plan-name} - {date} -Verdict: APPROVED | APPROVED WITH DEFERRALS +Verdict: APPROVED Rounds: {R} - {finding counts per round} ## Refinements applied @@ -109,19 +109,22 @@ Rounds: {R} - {finding counts per round} ## Escalations resolved ### E1 - {lens} - {one-line title} -Asked: {the question} - Answered: {the user's decision} -> {what the spec says now} +Asked: {the question} - Decided under policy: {the option chosen} -> {what the spec says now} ## Promoted to the ledger {decision slug or contract entry} -> `docs/decisions.md` | `docs/contracts.md` -## Deferred -### D1 - {lens} - {one-line title} -{the question still open, and why it exceeded this run: premise invalidated, new effort, or unanswered} +## Rescope +### R1 - {lens} - {one-line title} +{what a future `scope` run must revisit, and why policy could not decide it: premise invalidated or a new effort} ## Strengths {the good notes worth keeping, deduplicated} ``` +`APPROVED` is the only verdict this copy writes, and the verdict line carries nothing after it. +The `## Rescope` section appears only on a `failed` run with reason `rescope`, and that report omits the verdict line entirely because nothing was approved. + ## Wrap up Open the chat summary with the table from `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py end scope-review --count findings_verified=N --count findings_refuted=N --count refinements_applied=N --count escalated=N`, pasted verbatim. diff --git a/plugins/factory/skills/run/references/judgment.md b/plugins/factory/skills/run/references/judgment.md index a800b8a..f30a2e0 100644 --- a/plugins/factory/skills/run/references/judgment.md +++ b/plugins/factory/skills/run/references/judgment.md @@ -8,7 +8,7 @@ Per phase, the evidence commands the orchestrator re-runs as proof and what a `d ## scope-review -`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a `Verdict: APPROVED` line with a `Rounds:` line. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. +`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a line whose whole text is exactly `Verdict: APPROVED` - nothing after it - with a `Rounds:` line. Check it as a whole line, `grep -cx 'Verdict: APPROVED'`, never as a substring: a longer verdict such as `Verdict: APPROVED WITH DEFERRALS` is not a `done`. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. ## build diff --git a/plugins/factory/skills/scope-review/SKILL.md b/plugins/factory/skills/scope-review/SKILL.md index 3a16fa0..9e16d87 100644 --- a/plugins/factory/skills/scope-review/SKILL.md +++ b/plugins/factory/skills/scope-review/SKILL.md @@ -1,13 +1,13 @@ --- name: scope-review -description: Review and auto-refine a settled spec before build starts - a fresh-context agent panel checks the plan against the actual repo for infeasible change sets, missing failure paths, semantic contradictions, and untestable scenarios, then verified findings are applied to spec.md (and, where they touch settled decisions or cross-boundary invariants, to docs/decisions.md and docs/contracts.md) by refine agents and the panel re-runs; findings only the user can decide are asked as questions at the end and their answers applied and promoted the same way, so a finished run hands build a spec - and ledger - ready to implement, with no separate scope pass needed to promote the changes. Use after scope settles a spec and before build implements it. +description: Review and auto-refine a settled spec before build starts - a fresh-context agent panel checks the plan against the actual repo for infeasible change sets, missing failure paths, semantic contradictions, and untestable scenarios, then verified findings are applied to spec.md (and, where they touch settled decisions or cross-boundary invariants, to docs/decisions.md and docs/contracts.md) by refine agents and the panel re-runs; findings that exceed refine authority are decided at the end of the run under factory policy and their answers applied and promoted the same way, so a finished run hands build a spec - and ledger - ready to implement, with no separate scope pass needed to promote the changes. Use after scope settles a spec and before build implements it. --- # Scope Review Review the spec with agents that did not write it, then refine it in place - before any implementation exists, and without a human in the loop. A defect caught here costs a spec edit; the same defect after build costs a re-implementation, so this loop runs to completion on its own and ends with a spec build can start on - not a findings list to triage, and not a handoff back to `scope`. -The few findings only the user can decide are asked as questions at the end of the run, and the answers are applied before it finishes. +The few findings that exceed refine authority are answered in section 4 under factory policy, and those answers are applied before the run finishes. The panel judges the plan against the actual repo, not against the conversation that produced it. You are the orchestrator: run tools, dispatch agents, apply the loop, report - your own reading of the spec is not a lens, and findings reach the spec only through verification. This skill edits `spec.md`, and promotes settled changes to `docs/decisions.md` and `docs/contracts.md` per step 5 below - never code, and never any other file. @@ -38,7 +38,7 @@ A third panel means the refinements are churning, not converging - stop and esca 5. **Re-gate.** Loop `lint-spec.py` until clean, fixing mechanical fallout with the same refine agents; then start the next round so fresh eyes judge the refined spec. Exit the loop when a panel raises nothing refinable; two rounds of the same finding surviving refinement is itself an escalation. -Escalations collected across the rounds go to the interview below, not to a handoff. +Escalations collected across the rounds go to section 4 below, not to a handoff. ## 3. Authority and refinement rules @@ -50,10 +50,10 @@ Refine agents resolve conflicts by this order - each level beats everything belo 4. The spec's prose. Refinable: false premises about the repo (rewrite the entry against the real code, including the extra work that reveals), change sets contradicting their linked decisions, missing test scenarios for stated invariants and failure paths, untestable scenarios (replace with one provable at that layer), and gaps whose resolution is forced once the repo is consulted. -Escalations - never auto-applied, queued for the interview instead: anything that would flip a `✓` decision to a rejected alternative, change the user-visible scope or behavior, add or drop a dependency, or contradict the user's recorded intent. +Escalations - never auto-applied by a refine agent, queued for section 4 instead: anything that would flip a `✓` decision to a rejected alternative, change the user-visible scope or behavior, add or drop a dependency, or contradict the user's recorded intent. A refine agent that cannot fix its finding without crossing that line marks it escalated and leaves the spec alone. Refinements follow `scope`'s notation: decision entries keep their slugs and marks, change sets keep their numbering, new scenarios carry layer tags. -A refinement that adds or rewrites a decision entry, or a cross-boundary invariant, is promoted to the ledger immediately per step 5 - it does not wait for the interview. +A refinement that adds or rewrites a decision entry, or a cross-boundary invariant, is promoted to the ledger immediately per step 5 - it does not wait for section 4. ## 4. Resolve escalations under factory policy @@ -99,7 +99,7 @@ Write `.dev/{plan-name}/spec-review_N.md` at the next free index, one per run, c ```markdown # Spec review N - {plan-name} - {date} -Verdict: APPROVED | APPROVED WITH DEFERRALS +Verdict: APPROVED Rounds: {R} - {finding counts per round} ## Refinements applied @@ -108,19 +108,22 @@ Rounds: {R} - {finding counts per round} ## Escalations resolved ### E1 - {lens} - {one-line title} -Asked: {the question} - Answered: {the user's decision} -> {what the spec says now} +Asked: {the question} - Decided under policy: {the option chosen} -> {what the spec says now} ## Promoted to the ledger {decision slug or contract entry} -> `docs/decisions.md` | `docs/contracts.md` -## Deferred -### D1 - {lens} - {one-line title} -{the question still open, and why it exceeded this run: premise invalidated, new effort, or unanswered} +## Rescope +### R1 - {lens} - {one-line title} +{what a future `scope` run must revisit, and why policy could not decide it: premise invalidated or a new effort} ## Strengths {the good notes worth keeping, deduplicated} ``` +`APPROVED` is the only verdict this copy writes, and the verdict line carries nothing after it. +The `## Rescope` section appears only on a `failed` run with reason `rescope`, and that report omits the verdict line entirely because nothing was approved. + ## Wrap up Open the chat summary with the table from `python3 {scope-review-skill-root}/../../scripts/skill-metrics.py end scope-review --count findings_verified=N --count findings_refuted=N --count refinements_applied=N --count escalated=N`, pasted verbatim. From 196ac73f7b99fe614012aac9b4dd487bcbf4f0f7 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 13:00:57 +0200 Subject: [PATCH 10/30] fix(factory): enforce C-factory-unattended in the F03 pattern What: widen scripts/validate.sh's F03 pattern to catch the person-routing phrasings it missed (plural "human calls", "question left for the user", "the user asks/says/accepts", "prompt the user", "surface to the maintainer", "AskUserQuestion"), guard it with positive and negative self-test samples, skip YAML frontmatter (a description says when a person invokes the skill, which is before the go), and rewrite every line the widened pattern now catches in the factory phase copies to the unattended policy. Adds two eval tests planting newly covered phrasings, and corrects the scope comment's wrong justification for excluding ci-parity.md and jira.md. Why: review_1.md blocker B7 - C-factory-unattended was stated as a guarantee but its guarantor regex missed live person-routing text already inside its own scan root, so validate.sh stayed green while phases still routed decisions to a person. --- factory/evals/tests/test_factory_checks.py | 57 +++++++++++----- factory/references/factory-run.md | 2 +- factory/skills/build/SKILL.md | 2 +- factory/skills/ship/SKILL.md | 2 +- factory/skills/ship/references/gauntlet.md | 2 +- .../ship/references/orchestration-heavy.md | 2 +- .../skills/ship/references/orchestration.md | 6 +- .../skills/ship/references/pull-request.md | 2 +- .../skills/ship/references/report-format.md | 2 +- factory/skills/ship/references/tools.md | 2 +- plugins/factory/references/factory-run.md | 2 +- plugins/factory/skills/build/SKILL.md | 2 +- plugins/factory/skills/ship/SKILL.md | 2 +- .../skills/ship/references/gauntlet.md | 2 +- .../ship/references/orchestration-heavy.md | 2 +- .../skills/ship/references/orchestration.md | 6 +- .../skills/ship/references/pull-request.md | 2 +- .../skills/ship/references/report-format.md | 2 +- .../factory/skills/ship/references/tools.md | 2 +- scripts/validate.sh | 67 +++++++++++++++++-- 20 files changed, 123 insertions(+), 45 deletions(-) diff --git a/factory/evals/tests/test_factory_checks.py b/factory/evals/tests/test_factory_checks.py index f3dba60..42b44f2 100644 --- a/factory/evals/tests/test_factory_checks.py +++ b/factory/evals/tests/test_factory_checks.py @@ -51,6 +51,7 @@ def run_validate(copy_root: Path) -> subprocess.CompletedProcess: return subprocess.run( ["bash", "scripts/validate.sh"], cwd=copy_root, + check=False, capture_output=True, text=True, env=full_env, @@ -70,6 +71,15 @@ def append_to(self, relative: str, text: str) -> Path: handle.write(text) return target + def assert_rebuilt_copy_fails(self, tag: str, target: Path) -> None: + """Rebuild the mutated copy, run its validate.sh, expect `tag` at `target`.""" + rebuild(self.copy_root) + result = run_validate(self.copy_root) + self.assertEqual(result.returncode, 1) + self.assertIn(tag, result.stdout) + self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) + self.assertNotIn("[C01]", result.stdout) + def test_unmutated_copy_passes_and_excludes_the_harness(self) -> None: result = run_validate(self.copy_root) self.assertEqual(result.returncode, 0, result.stdout + result.stderr) @@ -83,12 +93,32 @@ def test_decision_routed_to_a_person_fails_f03(self) -> None: target = self.append_to( "factory/skills/build/SKILL.md", "\n\nAsk the user which one to pick.\n" ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_human_call_wording_fails_f03(self) -> None: + """A phrasing the first F03 pattern missed: the plural defeated `human call`.""" + target = self.append_to( + "factory/skills/ship/SKILL.md", + "\n\nRotation and history rewriting are both human calls.\n", + ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_question_left_for_the_user_fails_f03(self) -> None: + target = self.append_to( + "factory/skills/build/SKILL.md", + "\n\nThen any question left for the user - a blocked gate.\n", + ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_prose_about_people_does_not_fire_f03(self) -> None: + """The widened pattern must not fire on possessives or compounds.""" + self.append_to( + "factory/skills/build/SKILL.md", + "\n\nA user-facing change in the user's repository still needs an e2e scenario.\n", + ) rebuild(self.copy_root) result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 1) - self.assertIn("[F03]", result.stdout) - self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) - self.assertNotIn("[C01]", result.stdout) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) def test_same_line_inside_interactive_only_passes(self) -> None: self.append_to( @@ -104,8 +134,8 @@ def test_broken_self_test_regex_fails_f03(self) -> None: text = validate_sh.read_text(encoding="utf-8") # Break the F03 pattern so it no longer matches its own self-test samples. broken = text.replace( - r"ask(s|ed|ing)?|confirm(s|ed|ing)?\s+with", - r"askXXX(s|ed|ing)?|confirm(s|ed|ing)?\s+withXXX", + r'ASK = r"ask(?:s|ed|ing)?|prompt(?:s|ed|ing)?', + r'ASK = r"askXXX(?:s|ed|ing)?|promptXXX(?:s|ed|ing)?', ) self.assertNotEqual(text, broken, "expected the F03 pattern text to be present") validate_sh.write_text(broken, encoding="utf-8") @@ -119,22 +149,13 @@ def test_missing_result_protocol_literal_fails_f02(self) -> None: text = target.read_text(encoding="utf-8") text = text.replace("results/{phase}-{attempt}.json", "some-other-path.json") target.write_text(text, encoding="utf-8") - rebuild(self.copy_root) - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 1) - self.assertIn("[F02]", result.stdout) - self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) - self.assertNotIn("[C01]", result.stdout) + self.assert_rebuilt_copy_fails("[F02]", target) def test_invoking_another_phase_fails_f02(self) -> None: target = self.append_to( "factory/skills/build/SKILL.md", "\n\nOn success, invoke $factory:ship.\n" ) - rebuild(self.copy_root) - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 1) - self.assertIn("[F02]", result.stdout) - self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) + self.assert_rebuilt_copy_fails("[F02]", target) def test_unattended_wording_check_is_clean_over_the_four_phase_copies(self) -> None: result = run_validate(self.copy_root) @@ -151,6 +172,7 @@ def test_build_codex_plugin_check_reports_up_to_date(self) -> None: result = subprocess.run( [sys.executable, "scripts/build_codex_plugin.py", "--check", "--plugin", "factory"], cwd=self.copy_root, + check=False, capture_output=True, text=True, ) @@ -161,6 +183,7 @@ def test_architecture_check_is_clean_on_the_updated_overview(self) -> None: result = subprocess.run( [sys.executable, "dev/scripts/architecture-check.py", "docs/architecture.md", "--root", "."], cwd=self.copy_root, + check=False, capture_output=True, text=True, ) diff --git a/factory/references/factory-run.md b/factory/references/factory-run.md index db3ecb6..4436a29 100644 --- a/factory/references/factory-run.md +++ b/factory/references/factory-run.md @@ -42,7 +42,7 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las ## The unattended policy -After the go, no phase skill asks a person anything. An escalation that dev's version would ask a human about is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. +After the go, no phase skill asks a person anything. An escalation that dev's version would route to a person is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. `scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. Every other phase copy, and `run/SKILL.md`'s own person-routing lines, live inside `` / `` blocks or outside the scan root entirely. diff --git a/factory/skills/build/SKILL.md b/factory/skills/build/SKILL.md index 84685ae..797a9a1 100644 --- a/factory/skills/build/SKILL.md +++ b/factory/skills/build/SKILL.md @@ -48,7 +48,7 @@ python3 {build-skill-root}/../../scripts/skill-metrics.py end build --count chan Then it ends with the same two lines, in this order - a green run, a blocked gate, and a run with open deviations all get both: 1. `Next step: run ship over this work, pointed at .dev/{plan-name}/implementation-notes.md and the {plan-name}-e2e-report.html.` Recommend it; never launch it yourself. -2. Then any question left for the user - a blocked gate, an unresolved deviation. +2. Then every escalation this run settled itself - a blocked gate, an unresolved deviation - decided under factory policy and recorded in `auto_decided`. A blocked gate never replaces line 1. Neither does a failed e2e loop: say what is blocked, then still point at `ship`. diff --git a/factory/skills/ship/SKILL.md b/factory/skills/ship/SKILL.md index afe8cf2..a6f212d 100644 --- a/factory/skills/ship/SKILL.md +++ b/factory/skills/ship/SKILL.md @@ -27,7 +27,7 @@ Both phases share one scope. - PR number given -> target that PR; else the host's PR tooling, when it has any, for an open PR; else the local branch against the default branch. - Gather the diff per the diff-scope rules in [../../references/plan-layout.md](../../references/plan-layout.md): local git only, standard exclusions. - Locate the plan directory by the same reference's convention and read its `spec.md`; degrade gracefully without one. -- Full-repo runs only when the user asks - they are expensive, and the loops are the same. +- Full-repo runs only when the launch prompt names one - they are expensive, and the loops are the same. - No reviewable files: report `failed` with that reason and stop. Diff is tiny (1-2 files) or huge (>25k lines): proceed and note it in `auto_decided` rather than spending extra agents confirming. ## Phase 1: the gauntlet diff --git a/factory/skills/ship/references/gauntlet.md b/factory/skills/ship/references/gauntlet.md index 1e7270b..a8f4a6d 100644 --- a/factory/skills/ship/references/gauntlet.md +++ b/factory/skills/ship/references/gauntlet.md @@ -52,5 +52,5 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. -When the user accepts a different threshold, record it in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. +When a run settles on a different threshold under factory policy, record it in `auto_decided` and in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. Never adjust a threshold silently to make a run pass. diff --git a/factory/skills/ship/references/orchestration-heavy.md b/factory/skills/ship/references/orchestration-heavy.md index 3fa9a76..cae42bb 100644 --- a/factory/skills/ship/references/orchestration-heavy.md +++ b/factory/skills/ship/references/orchestration-heavy.md @@ -1,6 +1,6 @@ # Heavy Mode: Workflow Tool (explicit opt-in only) -Use this only when the user has explicitly asked for a workflow or an exhaustive run; it triggers the dynamic-workflow confirmation dialog. +This variant belongs to the interactive dev flow only: it triggers the dynamic-workflow confirmation dialog, which routes a decision outside the run, so an unattended phase never uses it. It buys schema-validated outputs, pipelining (findings verify while other lenses still review), and `/workflows` progress. Pass `diffFile`, `brief`, `planContext`, `lenses` (as `{key, prompt}` with the full assembled lens prompt), and `priorFindings` via `args`. diff --git a/factory/skills/ship/references/orchestration.md b/factory/skills/ship/references/orchestration.md index 57cc2ec..f304bb5 100644 --- a/factory/skills/ship/references/orchestration.md +++ b/factory/skills/ship/references/orchestration.md @@ -23,9 +23,9 @@ a fresh spawn has none of the original context and its output must not be used. 3. **Unavailable:** if neither transport exists, stop. The orchestrator must not impersonate the panel or review the code itself. -No dynamic workflows are involved by default. A Workflow-tool variant for -explicitly requested heavyweight runs only lives in -[orchestration-heavy.md](orchestration-heavy.md); do not read it otherwise. +No dynamic workflows are involved. The Workflow-tool variant in +[orchestration-heavy.md](orchestration-heavy.md) belongs to the interactive +dev flow only; an unattended phase never reads it. ## Native transport diff --git a/factory/skills/ship/references/pull-request.md b/factory/skills/ship/references/pull-request.md index 9c0b6ea..e317f98 100644 --- a/factory/skills/ship/references/pull-request.md +++ b/factory/skills/ship/references/pull-request.md @@ -6,7 +6,7 @@ A PR without evidence asks the reviewer to trust the description; this phase mak ## When it runs Phase 3 runs in the default flow, after the review report is written and the remediation loop in [remediation.md](remediation.md) has run its course: PASS and CONCERNS open a PR ready for review, a real blocker - one that survived both remediation rounds - opens a draft PR with the blockers listed first, so the work is preserved and CI runs while the recorded decision stands. -It is skipped in "gauntlet only" and "review only" runs and when the user says "no PR" or "local only"; a review-only run on someone else's PR never pushes anything. +It is skipped in "gauntlet only" and "review only" runs and when the launch prompt says "no PR" or "local only"; a review-only run on someone else's PR never pushes anything. Never force-push, never rebase, and never touch a branch other than the work branch and the evidence branch. ## Commit and push diff --git a/factory/skills/ship/references/report-format.md b/factory/skills/ship/references/report-format.md index 2f990bf..c188f9e 100644 --- a/factory/skills/ship/references/report-format.md +++ b/factory/skills/ship/references/report-format.md @@ -46,5 +46,5 @@ One line per lens; omit empty ones. What to fix first and why. ``` -`scope` reads this file when the user accepts findings that need real work: its Blockers and Concerns open that run's interview, and its Decision reconciliation section is what gets applied to `docs/decisions.md`. +`scope` reads this file when a later run takes up findings that need real work: its Blockers and Concerns open that run's interview, and its Decision reconciliation section is what gets applied to `docs/decisions.md`. Write it for that reader - a finding with no triggering scenario cannot become a decision. diff --git a/factory/skills/ship/references/tools.md b/factory/skills/ship/references/tools.md index 64985db..2e4a157 100644 --- a/factory/skills/ship/references/tools.md +++ b/factory/skills/ship/references/tools.md @@ -23,7 +23,7 @@ If the repo configures nothing, wire up the ecosystem's standard linter and type Three sub-scans, each zero-threshold: -- **Secrets**: gitleaks or the ecosystem equivalent over the in-scope files. A found secret is never fix-agent work - stop the gauntlet and escalate immediately, because it needs rotation and possibly history rewriting, both human calls. +- **Secrets**: gitleaks or the ecosystem equivalent over the in-scope files. A found secret is never fix-agent work - stop the gauntlet and report `stopped` with kind `secret.found`, because it needs rotation and possibly history rewriting, and the run performs neither itself. - **Dependency vulnerabilities**: the ecosystem's audit tool (osv-scanner, npm audit, pip-audit, cargo audit) over every manifest the diff touched. - **Static security rules**: the repo's own SAST config if one exists, else semgrep with the ecosystem's default ruleset, scoped to in-scope files. diff --git a/plugins/factory/references/factory-run.md b/plugins/factory/references/factory-run.md index db3ecb6..4436a29 100644 --- a/plugins/factory/references/factory-run.md +++ b/plugins/factory/references/factory-run.md @@ -42,7 +42,7 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las ## The unattended policy -After the go, no phase skill asks a person anything. An escalation that dev's version would ask a human about is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. +After the go, no phase skill asks a person anything. An escalation that dev's version would route to a person is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. `scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. Every other phase copy, and `run/SKILL.md`'s own person-routing lines, live inside `` / `` blocks or outside the scan root entirely. diff --git a/plugins/factory/skills/build/SKILL.md b/plugins/factory/skills/build/SKILL.md index abd6cfa..45d2736 100644 --- a/plugins/factory/skills/build/SKILL.md +++ b/plugins/factory/skills/build/SKILL.md @@ -47,7 +47,7 @@ python3 {build-skill-root}/../../scripts/skill-metrics.py end build --count chan Then it ends with the same two lines, in this order - a green run, a blocked gate, and a run with open deviations all get both: 1. `Next step: run ship over this work, pointed at .dev/{plan-name}/implementation-notes.md and the {plan-name}-e2e-report.html.` Recommend it; never launch it yourself. -2. Then any question left for the user - a blocked gate, an unresolved deviation. +2. Then every escalation this run settled itself - a blocked gate, an unresolved deviation - decided under factory policy and recorded in `auto_decided`. A blocked gate never replaces line 1. Neither does a failed e2e loop: say what is blocked, then still point at `ship`. diff --git a/plugins/factory/skills/ship/SKILL.md b/plugins/factory/skills/ship/SKILL.md index 7ac4945..a004142 100644 --- a/plugins/factory/skills/ship/SKILL.md +++ b/plugins/factory/skills/ship/SKILL.md @@ -26,7 +26,7 @@ Both phases share one scope. - PR number given -> target that PR; else the host's PR tooling, when it has any, for an open PR; else the local branch against the default branch. - Gather the diff per the diff-scope rules in [../../references/plan-layout.md](../../references/plan-layout.md): local git only, standard exclusions. - Locate the plan directory by the same reference's convention and read its `spec.md`; degrade gracefully without one. -- Full-repo runs only when the user asks - they are expensive, and the loops are the same. +- Full-repo runs only when the launch prompt names one - they are expensive, and the loops are the same. - No reviewable files: report `failed` with that reason and stop. Diff is tiny (1-2 files) or huge (>25k lines): proceed and note it in `auto_decided` rather than spending extra agents confirming. ## Phase 1: the gauntlet diff --git a/plugins/factory/skills/ship/references/gauntlet.md b/plugins/factory/skills/ship/references/gauntlet.md index 1e7270b..a8f4a6d 100644 --- a/plugins/factory/skills/ship/references/gauntlet.md +++ b/plugins/factory/skills/ship/references/gauntlet.md @@ -52,5 +52,5 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. -When the user accepts a different threshold, record it in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. +When a run settles on a different threshold under factory policy, record it in `auto_decided` and in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. Never adjust a threshold silently to make a run pass. diff --git a/plugins/factory/skills/ship/references/orchestration-heavy.md b/plugins/factory/skills/ship/references/orchestration-heavy.md index 3fa9a76..cae42bb 100644 --- a/plugins/factory/skills/ship/references/orchestration-heavy.md +++ b/plugins/factory/skills/ship/references/orchestration-heavy.md @@ -1,6 +1,6 @@ # Heavy Mode: Workflow Tool (explicit opt-in only) -Use this only when the user has explicitly asked for a workflow or an exhaustive run; it triggers the dynamic-workflow confirmation dialog. +This variant belongs to the interactive dev flow only: it triggers the dynamic-workflow confirmation dialog, which routes a decision outside the run, so an unattended phase never uses it. It buys schema-validated outputs, pipelining (findings verify while other lenses still review), and `/workflows` progress. Pass `diffFile`, `brief`, `planContext`, `lenses` (as `{key, prompt}` with the full assembled lens prompt), and `priorFindings` via `args`. diff --git a/plugins/factory/skills/ship/references/orchestration.md b/plugins/factory/skills/ship/references/orchestration.md index 57cc2ec..f304bb5 100644 --- a/plugins/factory/skills/ship/references/orchestration.md +++ b/plugins/factory/skills/ship/references/orchestration.md @@ -23,9 +23,9 @@ a fresh spawn has none of the original context and its output must not be used. 3. **Unavailable:** if neither transport exists, stop. The orchestrator must not impersonate the panel or review the code itself. -No dynamic workflows are involved by default. A Workflow-tool variant for -explicitly requested heavyweight runs only lives in -[orchestration-heavy.md](orchestration-heavy.md); do not read it otherwise. +No dynamic workflows are involved. The Workflow-tool variant in +[orchestration-heavy.md](orchestration-heavy.md) belongs to the interactive +dev flow only; an unattended phase never reads it. ## Native transport diff --git a/plugins/factory/skills/ship/references/pull-request.md b/plugins/factory/skills/ship/references/pull-request.md index 9c0b6ea..e317f98 100644 --- a/plugins/factory/skills/ship/references/pull-request.md +++ b/plugins/factory/skills/ship/references/pull-request.md @@ -6,7 +6,7 @@ A PR without evidence asks the reviewer to trust the description; this phase mak ## When it runs Phase 3 runs in the default flow, after the review report is written and the remediation loop in [remediation.md](remediation.md) has run its course: PASS and CONCERNS open a PR ready for review, a real blocker - one that survived both remediation rounds - opens a draft PR with the blockers listed first, so the work is preserved and CI runs while the recorded decision stands. -It is skipped in "gauntlet only" and "review only" runs and when the user says "no PR" or "local only"; a review-only run on someone else's PR never pushes anything. +It is skipped in "gauntlet only" and "review only" runs and when the launch prompt says "no PR" or "local only"; a review-only run on someone else's PR never pushes anything. Never force-push, never rebase, and never touch a branch other than the work branch and the evidence branch. ## Commit and push diff --git a/plugins/factory/skills/ship/references/report-format.md b/plugins/factory/skills/ship/references/report-format.md index 2f990bf..c188f9e 100644 --- a/plugins/factory/skills/ship/references/report-format.md +++ b/plugins/factory/skills/ship/references/report-format.md @@ -46,5 +46,5 @@ One line per lens; omit empty ones. What to fix first and why. ``` -`scope` reads this file when the user accepts findings that need real work: its Blockers and Concerns open that run's interview, and its Decision reconciliation section is what gets applied to `docs/decisions.md`. +`scope` reads this file when a later run takes up findings that need real work: its Blockers and Concerns open that run's interview, and its Decision reconciliation section is what gets applied to `docs/decisions.md`. Write it for that reader - a finding with no triggering scenario cannot become a decision. diff --git a/plugins/factory/skills/ship/references/tools.md b/plugins/factory/skills/ship/references/tools.md index 64985db..2e4a157 100644 --- a/plugins/factory/skills/ship/references/tools.md +++ b/plugins/factory/skills/ship/references/tools.md @@ -23,7 +23,7 @@ If the repo configures nothing, wire up the ecosystem's standard linter and type Three sub-scans, each zero-threshold: -- **Secrets**: gitleaks or the ecosystem equivalent over the in-scope files. A found secret is never fix-agent work - stop the gauntlet and escalate immediately, because it needs rotation and possibly history rewriting, both human calls. +- **Secrets**: gitleaks or the ecosystem equivalent over the in-scope files. A found secret is never fix-agent work - stop the gauntlet and report `stopped` with kind `secret.found`, because it needs rotation and possibly history rewriting, and the run performs neither itself. - **Dependency vulnerabilities**: the ecosystem's audit tool (osv-scanner, npm audit, pip-audit, cargo audit) over every manifest the diff touched. - **Static security rules**: the repo's own SAST config if one exists, else semgrep with the ecosystem's default ruleset, scoped to in-scope files. diff --git a/scripts/validate.sh b/scripts/validate.sh index dee0410..3ae5d18 100755 --- a/scripts/validate.sh +++ b/scripts/validate.sh @@ -620,23 +620,66 @@ check_pi() { # Scoped to factory/skills/{scope-review,build,ship,run} (SKILL.md and their # references) plus factory/references/factory-run.md only - scope keeps its # interview, and ci-parity.md, contracts.md, and jira.md keep their dev -# wording because the phases read them for notation, not for who decides. +# wording because they are dev-shared references whose human-call branches are +# overridden for a factory run by factory-run.md's unattended policy. +# +# Each file's YAML frontmatter is skipped: a description states when a person +# invokes the skill, which is before the go, not a decision routed after it. # =========================================================================== check_factory_unattended() { local found found=$(python3 - <<'PYEOF' import pathlib, re, sys +# A person noun, never the possessive ("the user's repo") or a compound +# ("a user-facing change"): those are prose about people, not routing to one. +PERSON = r"(?:user|operator|human|maintainer)s?\b(?![-'\u2019])" +ASK = r"ask(?:s|ed|ing)?|prompt(?:s|ed|ing)?|poll(?:s|ed|ing)?|quer(?:y|ies|ied|ying)|consult(?:s|ed|ing)?" +ROUTE = (r"confirm(?:s|ed|ing)?|check(?:s|ed|ing)?|wait(?:s|ed|ing)?\s+for|escalat(?:e|es|ed|ing)\s+to" + r"|surfac(?:e|es|ed|ing)\s+to|defer(?:s|red|ring)?\s+to|hand(?:s|ed|ing)?(?:\s+off)?\s+to" + r"|rout(?:e|es|ed|ing)\s+to") +DECIDES = (r"decides?|approves?|authorizes?|chooses?|confirms?|answers?|asks?|says?|accepts?|picks?" + r"|selects?|responds?|has\s+(?:explicitly\s+)?asked|must\s+(?:decide|choose|answer|confirm)") +ARTICLE = r"(?:the\s+|a\s+|an\s+|each\s+|every\s+)?" PATTERN = re.compile( - r"\b(ask(s|ed|ing)?|confirm(s|ed|ing)?\s+with|wait(s|ing)?\s+for|check(s|ing)?\s+with)\s+(the\s+)?(user|operator|human)s?\b" - r"|\b(user|operator|human)\s+(decides|approves|authorizes|chooses|confirms|answers)\b" - r"|structured user-input tool|\bhuman call\b|\bconfirm (with|before)\b", + rf"\b(?:{ASK})\s+{ARTICLE}{PERSON}" + rf"|\b(?:{ROUTE})\s+{ARTICLE}{PERSON}" + rf"|\b{ARTICLE}{PERSON}\s+(?:{DECIDES})\b" + rf"|\b(?:question|decision|choice|call)s?\b[^.]{{0,40}}?\b(?:for|to|from)\s+{ARTICLE}{PERSON}" + r"|\bhuman\s+calls?\b|\bstructured user-input tool\b|\bAskUserQuestion\b" + r"|\bconfirm (?:with|before)\b", re.I, ) -for sample in ("ask the user which one", "a human call", "the user decides", "confirm with the operator"): +SAMPLES = ( + "ask the user which one", + "a human call", + "both human calls", + "the user decides", + "confirm with the operator", + "then any question left for the user", + "full-repo runs only when the user asks", + "prompt the user for a threshold", + "surface to the maintainer", + "offer the choice with AskUserQuestion", + "wait for the human to answer", + "escalate to a human", + "the user accepts a different threshold", +) +NON_SAMPLES = ( + "the user's repository stays untouched", + "a user-facing change needs an e2e scenario", + "a whole user journey actually works", + "no phase skill asks a person anything", + "report failed naming what a person must supply", +) +for sample in SAMPLES: if not PATTERN.search(sample): print(f"scripts/validate.sh: F03 pattern no longer matches {sample!r}") sys.exit(0) +for sample in NON_SAMPLES: + if PATTERN.search(sample): + print(f"scripts/validate.sh: F03 pattern now falsely matches {sample!r}") + sys.exit(0) paths = [] for phase in ("scope-review", "build", "ship", "run"): @@ -647,9 +690,21 @@ run_md = pathlib.Path("factory/references/factory-run.md") if run_md.is_file(): paths.append(run_md) +def body_lines(path): + """Every line after the YAML frontmatter, numbered from 1 in the file.""" + lines = path.read_text(encoding="utf-8").splitlines() + start = 0 + if lines and lines[0].strip() == "---": + for index in range(1, len(lines)): + if lines[index].strip() == "---": + start = index + 1 + break + return enumerate(lines[start:], start=start + 1) + + for path in paths: inside = False - for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1): + for number, line in body_lines(path): if "" in line: inside = True elif "" in line: From ecb5f3e5eca91ae6bcdb725765cd110dd1bb9449 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 13:27:01 +0200 Subject: [PATCH 11/30] test(factory): cover run-state BAD_CALL paths and empty stop kind Why: mutation testing on run-state.py left 11 survivors after the attempt subcommand and the new exit-code ladder landed. Five were real gaps: a `return None` mutant on `return BAD_CALL` in cmd_handoff, cmd_record, cmd_check_result and cmd_attempt turns exit 3 into exit 0, so a corrupt state file or an impossible call would be reported to the orchestrator as success, and stop_kind's `return ""` fallthrough was unobserved. No test covered those error paths. Adds seven subprocess tests: handoff over a truncated and over a non-object state file, record with stdin JSON that parses but is not a decision object, check-result on a JSON array, check-result on a stopped result whose `stop` is neither dict nor str (exact stdout `stopped: `), and attempt over a truncated and over a non-object state file. Survivors now 6, all `return 0` sites where `SystemExit(None)` also exits 0. --- factory/evals/tests/test_run_state.py | 53 +++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/factory/evals/tests/test_run_state.py b/factory/evals/tests/test_run_state.py index 1cd6a6a..57e971b 100644 --- a/factory/evals/tests/test_run_state.py +++ b/factory/evals/tests/test_run_state.py @@ -404,6 +404,59 @@ def test_show_with_a_missing_state_file_exits_3(self) -> None: self.assertEqual(result.returncode, 3) self.assertNotIn("Traceback", result.stderr) + def test_handoff_with_a_corrupt_state_file_exits_3(self) -> None: + self.init_plan() + self.write_spec() + (self.plan_dir / "factory-run.json").write_text('{"plan": "fix', encoding="utf-8") + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertIn("unusable state file", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_handoff_with_a_non_object_state_file_exits_3(self) -> None: + self.init_plan() + self.write_spec() + (self.plan_dir / "factory-run.json").write_text("[]", encoding="utf-8") + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertIn("not a JSON object", result.stdout) + + def test_record_rejects_stdin_json_that_is_not_an_object(self) -> None: + self.init_plan() + before = (self.plan_dir / "factory-run.json").read_text() + result = run_state(self.cwd, "record", "fixture-plan", stdin=json.dumps([{"action": "advance"}])) + self.assertEqual(result.returncode, 3) + self.assertIn("not a decision object", result.stdout) + self.assertEqual(before, (self.plan_dir / "factory-run.json").read_text()) + + def test_check_result_with_a_json_array_exits_3(self) -> None: + path = self.cwd / "array.json" + path.write_text(json.dumps([{"status": "done"}]), encoding="utf-8") + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 3) + self.assertIn("not a JSON object", result.stdout) + + def test_check_result_stopped_without_a_kind_prints_an_empty_kind(self) -> None: + path = self.write_result({"status": "stopped", "stop": ["secret.found"]}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 2) + self.assertEqual(result.stdout, "stopped: \n") + + def test_attempt_with_a_corrupt_state_file_exits_3(self) -> None: + self.init_plan() + (self.plan_dir / "factory-run.json").write_text('{"plan": "fix', encoding="utf-8") + result = run_state(self.cwd, "attempt", "fixture-plan", "build") + self.assertEqual(result.returncode, 3) + self.assertIn("unusable state file", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_with_a_non_object_state_file_exits_3(self) -> None: + self.init_plan() + (self.plan_dir / "factory-run.json").write_text('"not a state"', encoding="utf-8") + result = run_state(self.cwd, "attempt", "fixture-plan", "build", "--status", "done") + self.assertEqual(result.returncode, 3) + self.assertIn("not a JSON object", result.stdout) + def test_show_exits_0(self) -> None: self.init_plan() result = run_state(self.cwd, "show", "fixture-plan") From ac04e2d79d1522aae1e71bb75bc2f9fee7270802 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 14:00:41 +0200 Subject: [PATCH 12/30] fix(factory): make a crashed run resumable from factory-run.json alone Why: the ship re-review found round 1's `phases` writer inert - R7 nothing read it back, so a resume relaunched as attempt 1 and check-result'd the stale build-1.json left by the dead session; R8 the attempt was recorded after a launch call that blocks until the phase ends, so a mid-phase crash left the list empty and indistinguishable from never started; R9 section 2 resumed "the state file not marked done" when no writer produced any terminal marker. Section 2 now reads the history with `run-state.py show` before launching anything and branches on the last attempt's status: a dangling `launched` with its own result file present is picked up at check-result, one without is closed `failed` with no `--result` and relaunched with guidance to start from the artifacts on disk. Section 5 records the launch first and takes the attempt number from what `run-state.py attempt` prints, so the state file is the number's only owner. The run's terminal marker is the `decisions` list ending in `end` or an `advance` on `ship`, which the state can already answer and which factory-run.md now documents beside the attempt numbering. --- factory/evals/tests/test_factory_checks.py | 79 +++++++- factory/references/factory-run.md | 10 +- factory/skills/build/references/mocking.md | 2 +- factory/skills/run/SKILL.md | 28 ++- factory/skills/run/references/launch.md | 2 +- factory/skills/ship/references/gauntlet.md | 3 +- plugins/factory/references/factory-run.md | 10 +- .../skills/build/references/mocking.md | 2 +- plugins/factory/skills/run/SKILL.md | 28 ++- .../factory/skills/run/references/launch.md | 2 +- .../skills/ship/references/gauntlet.md | 3 +- scripts/validate.sh | 180 +++++++++++++++--- 12 files changed, 289 insertions(+), 60 deletions(-) diff --git a/factory/evals/tests/test_factory_checks.py b/factory/evals/tests/test_factory_checks.py index 42b44f2..2037c67 100644 --- a/factory/evals/tests/test_factory_checks.py +++ b/factory/evals/tests/test_factory_checks.py @@ -121,21 +121,94 @@ def test_prose_about_people_does_not_fire_f03(self) -> None: self.assertEqual(result.returncode, 0, result.stdout + result.stderr) def test_same_line_inside_interactive_only_passes(self) -> None: + """run/SKILL.md is the one file whose pre-go lines may reach a person.""" self.append_to( - "factory/skills/build/SKILL.md", + "factory/skills/run/SKILL.md", "\n\n\nAsk the user which one to pick.\n\n", ) rebuild(self.copy_root) result = run_validate(self.copy_root) self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + def test_interactive_only_marker_in_a_phase_copy_fails_f03(self) -> None: + """A phase copy has nothing to route, so it may not use the escape hatch.""" + target = self.append_to( + "factory/skills/build/SKILL.md", + "\n\n\nAsk the user which one to pick.\n\n", + ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_unclosed_interactive_only_block_fails_f03(self) -> None: + """An unclosed marker used to silence every line after it, to end of file.""" + target = self.append_to( + "factory/skills/run/SKILL.md", + "\n\n\nAsk the user which one to pick.\n", + ) + self.assert_rebuilt_copy_fails("[F03]", target) + result = run_validate(self.copy_root) + self.assertIn("never closed", result.stdout) + + def test_deleting_a_close_marker_fails_f03(self) -> None: + """Dropping one close marker leaves the rest of run/SKILL.md unscanned.""" + target = self.copy_root / "factory" / "skills" / "run" / "SKILL.md" + text = target.read_text(encoding="utf-8") + self.assertIn("", text) + target.write_text( + text.replace("", "", 1) + "\n\nAsk the user which one to pick.\n", + encoding="utf-8", + ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_stray_close_marker_fails_f03(self) -> None: + target = self.append_to( + "factory/skills/run/SKILL.md", "\n\n\n" + ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_marker_sharing_its_line_does_not_silence_the_line(self) -> None: + """Only a marker alone on its line opens a block; anything else is prose.""" + target = self.append_to( + "factory/skills/build/SKILL.md", + "\n\n Ask the user which one to pick.\n", + ) + self.assert_rebuilt_copy_fails("[F03]", target) + + def test_newly_covered_person_routing_shapes_fail_f03(self) -> None: + """Shapes the round-1 pattern missed: person, requester, owner, someone.""" + shapes = ( + "Surface it to the person and wait for the answer.", + "Escalate the threshold to the requester.", + "Hand it to the owner for a decision.", + "Ask someone on the team which one to pick.", + "Raising the gate is a human decision.", + "Get approval before opening the PR.", + "Check with them before proceeding.", + "In the end it is their call.", + "Seek sign-off from the maintainer.", + "This seam is worth agreeing on with the user.", + ) + for shape in shapes: + with self.subTest(shape=shape): + copy_root = copy_repo() + try: + target = copy_root / "factory" / "skills" / "build" / "SKILL.md" + with target.open("a", encoding="utf-8") as handle: + handle.write(f"\n\n{shape}\n") + rebuild(copy_root) + result = run_validate(copy_root) + self.assertEqual(result.returncode, 1, result.stdout) + self.assertIn("[F03]", result.stdout) + self.assertIn("factory/skills/build/SKILL.md", result.stdout) + finally: + shutil.rmtree(copy_root.parent, ignore_errors=True) + def test_broken_self_test_regex_fails_f03(self) -> None: validate_sh = self.copy_root / "scripts" / "validate.sh" text = validate_sh.read_text(encoding="utf-8") # Break the F03 pattern so it no longer matches its own self-test samples. broken = text.replace( - r'ASK = r"ask(?:s|ed|ing)?|prompt(?:s|ed|ing)?', - r'ASK = r"askXXX(?:s|ed|ing)?|promptXXX(?:s|ed|ing)?', + r'ASK = (r"ask|prompt|poll', + r'ASK = (r"askXXX|promptXXX|pollXXX', ) self.assertNotEqual(text, broken, "expected the F03 pattern text to be present") validate_sh.write_text(broken, encoding="utf-8") diff --git a/factory/references/factory-run.md b/factory/references/factory-run.md index 4436a29..36b0103 100644 --- a/factory/references/factory-run.md +++ b/factory/references/factory-run.md @@ -8,8 +8,8 @@ Owned by the `run` skill. Every phase copy (`scope`, `scope-review`, `build`, `s - `plan`, `request`, `base` (the default branch at `init`). - `branch`, `spec_sha256`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `dirty_files` - all written at `handoff`. -- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. -- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. +- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. The list is the attempt numbering: an attempt's number is its 1-based position, it is appended as `launched` before the phase is launched, and its result file is `results/{phase}-{that number}.json`. A trailing `launched` entry on resume therefore means a session died mid-phase, which is what tells that case apart from a phase never started (an empty list). +- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. The list's end is also the run's terminal marker: a run is finished when its last entry is an `end`, or an `advance` on `ship`. There is no separate done field. - `repairs`: an ordered list of `{phase, attempt, description, files: [], evidence: []}`. ## Result envelope: `factory.result/1` @@ -42,9 +42,9 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las ## The unattended policy -After the go, no phase skill asks a person anything. An escalation that dev's version would route to a person is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. +After the go, no phase skill asks a person anything. An escalation that dev's version would put outside the run is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. -`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. Every other phase copy, and `run/SKILL.md`'s own person-routing lines, live inside `` / `` blocks or outside the scan root entirely. +`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. `run/SKILL.md` is the only file in the scan root that may use `` / ``, for its pre-go lines; the markers must balance and each must stand alone on its line. No phase copy may use them, because no phase copy has anything to route. ## Launch prompt shape @@ -52,7 +52,7 @@ Each phase subagent's prompt carries, in order: 1. The phase's skill path: the absolute path to `factory/skills/{phase}/SKILL.md`, with the instruction to read `{phase}-skill-root` as that path's parent directory (a subagent reading a file cannot resolve `{phase}-skill-root}`-style placeholders on its own). 2. The plan name and the plan directory's absolute path. -3. The attempt number. +3. The attempt number, which is the one `run-state.py attempt` printed when it recorded this launch, never a number counted by hand or read off a file in `results/`. 4. On a relaunch: the previous attempt's reason and explicit guidance naming what it did and what is required instead - never "try again". 5. The scratch root for this attempt: `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/`. 6. The result path this attempt must write: `.dev/{plan}/results/{phase}-{attempt}.json`. diff --git a/factory/skills/build/references/mocking.md b/factory/skills/build/references/mocking.md index 26af070..3eb5b3d 100644 --- a/factory/skills/build/references/mocking.md +++ b/factory/skills/build/references/mocking.md @@ -17,7 +17,7 @@ If the internal wiring is wrong, a test full of internal mocks will still pass, ## Never mock internal collaborators If you feel the need to mock a class your own codebase owns, that is a design signal, not a testing problem. -Either test at a higher seam where the collaborator can just run, or the collaborator itself is a seam worth agreeing on with the user. +Either test at a higher seam where the collaborator can just run, or the collaborator itself is a seam this run settles under factory policy and records in `auto_decided`. The tell that a mock is wrong: the test asserts *that* a method was called (`toHaveBeenCalledWith`) instead of *what the outcome was*. Interaction assertions couple the test to the implementation; state and output assertions couple it to behavior. diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index 3605e02..b9b607e 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -20,9 +20,17 @@ If two unfinished state files exist under `.dev/` with no branch match, or a rec ## 2. Init or resume -No request and a run branch checked out: resume that plan. -No request and no branch: resume the single state file under `.dev/` not marked done. A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it, create `.dev/{plan}/` with `request.md` holding the request, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch}` (D-plan-slug, D-request-input). +No request and a run branch checked out: resume that plan. +No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when its `decisions` list ends in an `end` action, or in an `advance` on `ship`; nothing else marks a run finished, so every other state file is unfinished. + +Resuming, launch nothing before reading the history. Run `run-state.py show {plan}`: it prints each phase's attempts in order with their statuses, and those numbers are the run's only attempt numbers. Take the highest-numbered attempt of the earliest phase you have not accepted as `done`, and act on its status: + +- `launched`, and `.dev/{plan}/results/{phase}-{attempt}.json` exists for that attempt's own number: the phase finished and the session died before its result was read. Pick section 5 up at step 4, `check-result` on that exact path. +- `launched`, and no result file at that attempt's own number: the attempt died mid-phase. Close it with `run-state.py attempt {plan} {phase} --status failed` and no `--result`, so it is a failure with the reason "no result file" and no other attempt's result is read into it. Then take a new attempt from section 5 step 1, its guidance saying to start from the artifacts already on disk and not redo finished work. +- `done`, `failed`, or `stopped`: the attempt is closed already. Pick section 5 up at step 5 for it. + +A result file belongs to the attempt number in its name and to no other. Never infer an attempt number from what is in `results/`: after a crash, the newest file there is the last attempt that finished, not the one that was running. ## 3. Scope, inline @@ -38,14 +46,16 @@ At the go: create branch `factory/{plan}` from the base branch. Run `run-state.p For each phase in order (`scope-review`, `build`, `ship`): -1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. Record the launch with `run-state.py attempt {plan} {phase}` so a crashed attempt is distinguishable from an unstarted one on resume. -2. Wait for the host's completion signal - a batch in flight is not over. -3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`. -4. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). -5. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. -6. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. -7. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. +1. Take the attempt number first: `run-state.py attempt {plan} {phase}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. +2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. +3. Wait for the host's completion signal - a batch in flight is not over. +4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. +5. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). +6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. +7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. +8. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. ## 6. Closing report +The last decision is recorded before the report is printed - the `end`, or the `advance` that accepts `ship` - because that entry is the only thing that marks the run finished for a later resume. Print: phases, attempts, decisions, repairs, and the PR link or the exact action a person must take. Read only result files, check outputs, `git log`, and the state file for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). diff --git a/factory/skills/run/references/launch.md b/factory/skills/run/references/launch.md index 6dbb186..2e9b7ff 100644 --- a/factory/skills/run/references/launch.md +++ b/factory/skills/run/references/launch.md @@ -17,7 +17,7 @@ the absolute path {skill_path}; treat {skill_path}'s parent directory as {phase}-skill-root when the file uses that placeholder. Follow it exactly. Plan: {plan}, at .dev/{plan}/ (absolute: {plan_dir}). -Attempt: {attempt}. +Attempt: {attempt}, the number run-state.py just printed for this launch. {On a relaunch only: The previous attempt failed with reason "{reason}". Guidance: {what to do differently, never "try again"}.} diff --git a/factory/skills/ship/references/gauntlet.md b/factory/skills/ship/references/gauntlet.md index a8f4a6d..91a1911 100644 --- a/factory/skills/ship/references/gauntlet.md +++ b/factory/skills/ship/references/gauntlet.md @@ -52,5 +52,6 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. -When a run settles on a different threshold under factory policy, record it in `auto_decided` and in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. +A run never moves a threshold it is being judged by: the defaults above, plus any `D-` entry already in `docs/decisions.md` (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), are fixed for the whole run, and no phase writes a threshold decision into `docs/decisions.md`. +A threshold this run cannot meet is a `failed` result naming the finding, its score, and the threshold it missed; when the evidence argues the line itself sits wrong, record that argument in `auto_decided` as a proposal for review, and keep judging this run by the unchanged threshold. Never adjust a threshold silently to make a run pass. diff --git a/plugins/factory/references/factory-run.md b/plugins/factory/references/factory-run.md index 4436a29..36b0103 100644 --- a/plugins/factory/references/factory-run.md +++ b/plugins/factory/references/factory-run.md @@ -8,8 +8,8 @@ Owned by the `run` skill. Every phase copy (`scope`, `scope-review`, `build`, `s - `plan`, `request`, `base` (the default branch at `init`). - `branch`, `spec_sha256`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `dirty_files` - all written at `handoff`. -- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. -- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. +- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. The list is the attempt numbering: an attempt's number is its 1-based position, it is appended as `launched` before the phase is launched, and its result file is `results/{phase}-{that number}.json`. A trailing `launched` entry on resume therefore means a session died mid-phase, which is what tells that case apart from a phase never started (an empty list). +- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. The list's end is also the run's terminal marker: a run is finished when its last entry is an `end`, or an `advance` on `ship`. There is no separate done field. - `repairs`: an ordered list of `{phase, attempt, description, files: [], evidence: []}`. ## Result envelope: `factory.result/1` @@ -42,9 +42,9 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las ## The unattended policy -After the go, no phase skill asks a person anything. An escalation that dev's version would route to a person is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. +After the go, no phase skill asks a person anything. An escalation that dev's version would put outside the run is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. -`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. Every other phase copy, and `run/SKILL.md`'s own person-routing lines, live inside `` / `` blocks or outside the scan root entirely. +`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. `run/SKILL.md` is the only file in the scan root that may use `` / ``, for its pre-go lines; the markers must balance and each must stand alone on its line. No phase copy may use them, because no phase copy has anything to route. ## Launch prompt shape @@ -52,7 +52,7 @@ Each phase subagent's prompt carries, in order: 1. The phase's skill path: the absolute path to `factory/skills/{phase}/SKILL.md`, with the instruction to read `{phase}-skill-root` as that path's parent directory (a subagent reading a file cannot resolve `{phase}-skill-root}`-style placeholders on its own). 2. The plan name and the plan directory's absolute path. -3. The attempt number. +3. The attempt number, which is the one `run-state.py attempt` printed when it recorded this launch, never a number counted by hand or read off a file in `results/`. 4. On a relaunch: the previous attempt's reason and explicit guidance naming what it did and what is required instead - never "try again". 5. The scratch root for this attempt: `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/`. 6. The result path this attempt must write: `.dev/{plan}/results/{phase}-{attempt}.json`. diff --git a/plugins/factory/skills/build/references/mocking.md b/plugins/factory/skills/build/references/mocking.md index 26af070..3eb5b3d 100644 --- a/plugins/factory/skills/build/references/mocking.md +++ b/plugins/factory/skills/build/references/mocking.md @@ -17,7 +17,7 @@ If the internal wiring is wrong, a test full of internal mocks will still pass, ## Never mock internal collaborators If you feel the need to mock a class your own codebase owns, that is a design signal, not a testing problem. -Either test at a higher seam where the collaborator can just run, or the collaborator itself is a seam worth agreeing on with the user. +Either test at a higher seam where the collaborator can just run, or the collaborator itself is a seam this run settles under factory policy and records in `auto_decided`. The tell that a mock is wrong: the test asserts *that* a method was called (`toHaveBeenCalledWith`) instead of *what the outcome was*. Interaction assertions couple the test to the implementation; state and output assertions couple it to behavior. diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index f4e424b..6ec02e0 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -19,9 +19,17 @@ If two unfinished state files exist under `.dev/` with no branch match, or a rec ## 2. Init or resume -No request and a run branch checked out: resume that plan. -No request and no branch: resume the single state file under `.dev/` not marked done. A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it, create `.dev/{plan}/` with `request.md` holding the request, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch}` (D-plan-slug, D-request-input). +No request and a run branch checked out: resume that plan. +No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when its `decisions` list ends in an `end` action, or in an `advance` on `ship`; nothing else marks a run finished, so every other state file is unfinished. + +Resuming, launch nothing before reading the history. Run `run-state.py show {plan}`: it prints each phase's attempts in order with their statuses, and those numbers are the run's only attempt numbers. Take the highest-numbered attempt of the earliest phase you have not accepted as `done`, and act on its status: + +- `launched`, and `.dev/{plan}/results/{phase}-{attempt}.json` exists for that attempt's own number: the phase finished and the session died before its result was read. Pick section 5 up at step 4, `check-result` on that exact path. +- `launched`, and no result file at that attempt's own number: the attempt died mid-phase. Close it with `run-state.py attempt {plan} {phase} --status failed` and no `--result`, so it is a failure with the reason "no result file" and no other attempt's result is read into it. Then take a new attempt from section 5 step 1, its guidance saying to start from the artifacts already on disk and not redo finished work. +- `done`, `failed`, or `stopped`: the attempt is closed already. Pick section 5 up at step 5 for it. + +A result file belongs to the attempt number in its name and to no other. Never infer an attempt number from what is in `results/`: after a crash, the newest file there is the last attempt that finished, not the one that was running. ## 3. Scope, inline @@ -37,14 +45,16 @@ At the go: create branch `factory/{plan}` from the base branch. Run `run-state.p For each phase in order (`scope-review`, `build`, `ship`): -1. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, the attempt number, and (on a relaunch) the previous reason with explicit guidance. Record the launch with `run-state.py attempt {plan} {phase}` so a crashed attempt is distinguishable from an unstarted one on resume. -2. Wait for the host's completion signal - a batch in flight is not over. -3. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`. -4. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). -5. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. -6. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. -7. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. +1. Take the attempt number first: `run-state.py attempt {plan} {phase}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. +2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. +3. Wait for the host's completion signal - a batch in flight is not over. +4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. +5. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). +6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. +7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. +8. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. ## 6. Closing report +The last decision is recorded before the report is printed - the `end`, or the `advance` that accepts `ship` - because that entry is the only thing that marks the run finished for a later resume. Print: phases, attempts, decisions, repairs, and the PR link or the exact action a person must take. Read only result files, check outputs, `git log`, and the state file for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). diff --git a/plugins/factory/skills/run/references/launch.md b/plugins/factory/skills/run/references/launch.md index 6dbb186..2e9b7ff 100644 --- a/plugins/factory/skills/run/references/launch.md +++ b/plugins/factory/skills/run/references/launch.md @@ -17,7 +17,7 @@ the absolute path {skill_path}; treat {skill_path}'s parent directory as {phase}-skill-root when the file uses that placeholder. Follow it exactly. Plan: {plan}, at .dev/{plan}/ (absolute: {plan_dir}). -Attempt: {attempt}. +Attempt: {attempt}, the number run-state.py just printed for this launch. {On a relaunch only: The previous attempt failed with reason "{reason}". Guidance: {what to do differently, never "try again"}.} diff --git a/plugins/factory/skills/ship/references/gauntlet.md b/plugins/factory/skills/ship/references/gauntlet.md index a8f4a6d..91a1911 100644 --- a/plugins/factory/skills/ship/references/gauntlet.md +++ b/plugins/factory/skills/ship/references/gauntlet.md @@ -52,5 +52,6 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. -When a run settles on a different threshold under factory policy, record it in `auto_decided` and in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. +A run never moves a threshold it is being judged by: the defaults above, plus any `D-` entry already in `docs/decisions.md` (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), are fixed for the whole run, and no phase writes a threshold decision into `docs/decisions.md`. +A threshold this run cannot meet is a `failed` result naming the finding, its score, and the threshold it missed; when the evidence argues the line itself sits wrong, record that argument in `auto_decided` as a proposal for review, and keep judging this run by the unchanged threshold. Never adjust a threshold silently to make a run pass. diff --git a/scripts/validate.sh b/scripts/validate.sh index 3ae5d18..9686f8f 100755 --- a/scripts/validate.sh +++ b/scripts/validate.sh @@ -625,6 +625,11 @@ check_pi() { # # Each file's YAML frontmatter is skipped: a description states when a person # invokes the skill, which is before the go, not a decision routed after it. +# +# The pattern fires on a routing construction (ask/route/decide/gate), never +# on a bare mention of a person, and a construction inside a negated clause +# ("never ask the user", "no phase skill asks a person anything") is the +# policy being stated rather than a decision being routed. # =========================================================================== check_factory_unattended() { local found @@ -633,23 +638,75 @@ import pathlib, re, sys # A person noun, never the possessive ("the user's repo") or a compound # ("a user-facing change"): those are prose about people, not routing to one. -PERSON = r"(?:user|operator|human|maintainer)s?\b(?![-'\u2019])" -ASK = r"ask(?:s|ed|ing)?|prompt(?:s|ed|ing)?|poll(?:s|ed|ing)?|quer(?:y|ies|ied|ying)|consult(?:s|ed|ing)?" -ROUTE = (r"confirm(?:s|ed|ing)?|check(?:s|ed|ing)?|wait(?:s|ed|ing)?\s+for|escalat(?:e|es|ed|ing)\s+to" - r"|surfac(?:e|es|ed|ing)\s+to|defer(?:s|red|ring)?\s+to|hand(?:s|ed|ing)?(?:\s+off)?\s+to" - r"|rout(?:e|es|ed|ing)\s+to") -DECIDES = (r"decides?|approves?|authorizes?|chooses?|confirms?|answers?|asks?|says?|accepts?|picks?" - r"|selects?|responds?|has\s+(?:explicitly\s+)?asked|must\s+(?:decide|choose|answer|confirm)") -ARTICLE = r"(?:the\s+|a\s+|an\s+|each\s+|every\s+)?" +# "reviewer" and "author" are deliberately absent: in a code-review skill they +# name the downstream reader of a PR the run opens, not a gate inside the run. +PERSON = (r"(?:user|operator|human|maintainer|person|people|requester|requestor" + r"|owner|stakeholder|someone|somebody|anyone|anybody|whoever|team" + r"|them|they)s?\b(?![-'’])") +ARTICLE = r"(?:the\s+|a\s+|an\s+|each\s+|every\s+|any\s+|some\s+|another\s+|your\s+|their\s+)?" +ASK = (r"ask|prompt|poll|quer(?:y|ie)|consult|interview|question|survey|solicit" + r"|check\s+with|check\s+in\s+with") +ROUTE = (r"confirm|check|verify|escalat\w*|surfac\w*|defer|hand(?:\s+off)?|rout\w*" + r"|rais\w*|agree\w*|align|discuss|negotiat\w*|coordinat\w*|sync|refer" + r"|bring|take|leave|let|flag|wait|paus\w*|loop\s+in|circle\s+back" + r"|follow\s+up|report\s+back|put|send|forward|delegat\w*|punt|kick\s+up") +DECIDES = (r"decides?|decided|approves?|approved|authoriz\w+|chooses?|chose|confirms?" + r"|answers?|asks?|says?|accepts?|accepted|picks?|selects?|responds?|replies" + r"|weighs?\s+in|signs?\s+off|has\s+(?:explicitly\s+)?asked" + r"|must\s+(?:decide|choose|answer|confirm|approve|accept|sign|say|pick|select)" + r"|to\s+(?:decide|choose|confirm|approve|accept)") +GATE = (r"call|decision|choice|question|judgm?ent|approval|sign-?off|input|answer" + r"|consent|permission|go-?ahead|say-?so|blessing|verdict|guidance|steer|gate") PATTERN = re.compile( - rf"\b(?:{ASK})\s+{ARTICLE}{PERSON}" - rf"|\b(?:{ROUTE})\s+{ARTICLE}{PERSON}" - rf"|\b{ARTICLE}{PERSON}\s+(?:{DECIDES})\b" - rf"|\b(?:question|decision|choice|call)s?\b[^.]{{0,40}}?\b(?:for|to|from)\s+{ARTICLE}{PERSON}" - r"|\bhuman\s+calls?\b|\bstructured user-input tool\b|\bAskUserQuestion\b" - r"|\bconfirm (?:with|before)\b", + # 1. ask / prompt / consult a person + rf"\b(?:{ASK})(?:s|es|ed|ing)?\s+{ARTICLE}{PERSON}" + # 2. ask first, ask before doing it + rf"|\bask(?:s|ed|ing)?\s+(?:first|before|again)\b" + # 3. route it to / with / for a person, over a little filler + rf"|\b(?:{ROUTE})(?:s|es|ed|ing)?\b(?:\s+\w+){{0,3}}?\s+(?:to|with|for|from|by|on)\s+{ARTICLE}{PERSON}" + # 4. a person decides + rf"|\b{ARTICLE}{PERSON}(?:\s+\w+){{0,4}}?\s+(?:{DECIDES})\b" + # 5. a decision for / to / from a person + rf"|\b(?:{GATE})s?\b(?![-'’])[^.]{{0,40}}?\b(?:for|to|from|by)\s+{ARTICLE}{PERSON}" + # 6. a human call, manual approval, a user sign-off + rf"|\b(?:human|manual|person|people|user|operator)[-\s](?:{GATE})s?\b(?![-'’])" + # 7. their call, the user's say-so + r"|\btheir\s+(?:call|say-?so|blessing|approval|sign-?off|consent|permission|go-?ahead)s?\b" + rf"|\bthe\s+(?:user|human|person|operator|maintainer|owner|requester)'s\s+(?:{GATE})s?\b" + # 8. get approval, await sign-off, seek permission + r"|\b(?:get|obtain|seek|secure|request|await|need|require|ask\s+for)(?:s|ed|ing)?" + r"\s+(?:\w+\s+){0,2}?(?:approval|sign-?off|confirmation|permission|consent" + r"|the\s+go-?ahead|a\s+decision|an\s+answer)\b" + # 9. wait for an answer + r"|\bwait(?:s|ed|ing)?\s+(?:around\s+)?for\s+(?:\w+\s+){0,2}?" + r"(?:answer|reply|response|approval|decision|confirmation|go-?ahead|sign-?off|input)s?\b" + # 10. the structured user-input tools themselves + r"|\bstructured user-input tool\b|\bAskUserQuestion\b", + re.I, +) +# A negated clause states the policy ("never ask the user"); it routes nothing. +NEGATOR = re.compile( + r"\b(?:never|not|no|none|nothing|neither|nor|without|instead\s+of|rather\s+than" + r"|avoid(?:s|ed|ing)?|n't|cannot|can't)\b", re.I, ) +CLAUSE = re.compile(r"[.;:!?]|\s-\s") +OPEN, CLOSE = "", "" +# Only the orchestrator has pre-go lines that legitimately reach a person. +MARKER_ALLOWED = ("factory/skills/run/SKILL.md",) + + +def routes(line): + """The first person-routing match on the line, ignoring negated clauses.""" + for match in PATTERN.finditer(line): + prefix = line[: match.start()] + bounds = [m.end() for m in CLAUSE.finditer(prefix)] + clause = (prefix[bounds[-1]:] if bounds else prefix) + match.group(0) + if not NEGATOR.search(clause): + return match + return None + + SAMPLES = ( "ask the user which one", "a human call", @@ -664,6 +721,49 @@ SAMPLES = ( "wait for the human to answer", "escalate to a human", "the user accepts a different threshold", + "surface it to the person and wait for the answer", + "escalate to the requester", + "hand it to the owner for a decision", + "ask someone on the team", + "check with them before proceeding", + "this is a human decision", + "get approval before merging", + "seek sign-off from the maintainer", + "the collaborator itself is a seam worth agreeing on with the user", + "raise it with whoever owns the module", + "in the end it is their call", + "wait for their answer", + "the requester decides", + "a question for the owner", + "ask first, then proceed", + "pause for human input", + "leave the call to the person running the factory", + "defer to the stakeholder", + "the owner must approve", + "needs manual approval", + "route the choice to a person", + "the person running the run picks", + "await confirmation", + "the user's say-so", + "let the maintainer choose", + "put the question to the operator", + "poll the team", + "request permission first", + "this is a judgment call for the user", + "consult whoever owns it", + "send it to the owner for sign-off", + "ask them what they want", + "take it up with the requester", + "a decision that belongs to a person", + "coordinate with the maintainer on the threshold", + "delegate the call to a human", + "the user says no PR", + "interview the user", + "the operator answers", + "sync with the owner", + "requires sign-off from a human", + "a manual gate", + "get the go-ahead from the owner", ) NON_SAMPLES = ( "the user's repository stays untouched", @@ -671,13 +771,30 @@ NON_SAMPLES = ( "a whole user journey actually works", "no phase skill asks a person anything", "report failed naming what a person must supply", + "never ask the user; decide under factory policy", + "instead of asking the user, record it in auto_decided", + "rather than escalating to a person, report failed with the reason", + "the run never waits for an answer", + "decide it without asking the user", + "diffs whose owner did not ask for mutations", + "the merge happens after their verdicts", + "does it need a call-out to the reviewer", + "name it so the reviewer can decide", + "bring it to the author of the diff", + "record the reason a person would need", + "the exact step a person must take", + "a user story per scenario", + "the fix agent decides how", + "user-visible behavior changes", + "cannot ask the user, so it records auto_decided", + "the owner of the module is the module itself", ) for sample in SAMPLES: - if not PATTERN.search(sample): + if not routes(sample): print(f"scripts/validate.sh: F03 pattern no longer matches {sample!r}") sys.exit(0) for sample in NON_SAMPLES: - if PATTERN.search(sample): + if routes(sample): print(f"scripts/validate.sh: F03 pattern now falsely matches {sample!r}") sys.exit(0) @@ -703,14 +820,31 @@ def body_lines(path): for path in paths: - inside = False + opened_at = None for number, line in body_lines(path): - if "" in line: - inside = True - elif "" in line: - inside = False - elif not inside and PATTERN.search(line): - print(f"{path}:{number}: routes a decision to a person; decide under factory policy or mark the block ") + stripped = line.strip() + # A marker only silences what follows when it stands alone on its line: + # anything else on the line is prose, and prose is always scanned. + if stripped not in (OPEN, CLOSE): + if opened_at is None and routes(line): + print(f"{path}:{number}: routes a decision to a person; decide under factory policy") + continue + if path.as_posix() not in MARKER_ALLOWED: + print(f"{path}:{number}: interactive-only markers are allowed only in " + f"{', '.join(MARKER_ALLOWED)}; this file decides under factory policy instead") + continue + if stripped == OPEN: + if opened_at is not None: + print(f"{path}:{number}: interactive-only block opened again while the one " + f"opened at line {opened_at} is still open") + opened_at = number + else: + if opened_at is None: + print(f"{path}:{number}: interactive-only block closed but none was open") + opened_at = None + if opened_at is not None: + print(f"{path}:{opened_at}: interactive-only block is never closed; every line " + f"after it escapes the check") PYEOF ) if [ -n "$found" ]; then From 0e8f550c302960eeef4a398369c56375cd0b6503 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 14:03:18 +0200 Subject: [PATCH 13/30] fix(factory): keep an unattended run from moving its own gates Note: the content of this change is already in ac04e2d - a concurrently running agent staged and committed the whole tree, including these files, while this change was being staged. This commit records the intent and the Why; nothing is added on top. What: rewrite the factory gauntlet's threshold line so a run never amends `docs/decisions.md` or the threshold it is judged by - a threshold it cannot meet is a `failed` result naming the finding, the score, and the missed threshold, or an `auto_decided` proposal for review. Rewrite mocking.md's "a seam worth agreeing on with the user" line to the unattended policy. Restate factory-run.md's policy prose so it states the rule instead of reading as a routing instruction, and record that `run/SKILL.md` is the only file that may use the interactive-only markers. Widen the F03 pattern from an allow-list of its own samples into routing constructions over a much larger person vocabulary (person, requester, owner, stakeholder, someone, whoever, team, them), with a negated-clause test so "no phase skill asks a person anything" stays clean, plus branches for gate nouns, approval-seeking and waiting for an answer: 45 of 60 probed phrasings were missed before, 0 after. Make the markers checkable - unbalanced, stray, and out-of-file markers are F03 failures, and a marker sharing its line no longer silences that line. Extend SAMPLES and NON_SAMPLES to everything newly covered and newly excluded, and add eval tests for the new shapes and every marker failure. Why: the ship re-review, items R10-R13. Round 1 turned a human gate into self-acceptance (a run could raise a threshold it had failed and write the loosened gate into the committed decision ledger as precedent for later runs), left a person-routing line inside F03's own scan root, widened the pattern only as far as its own samples, and left an unclosed marker able to silence the rest of a file. From f056f3d5cc0dd97a96f3f2558289e558bb560ab7 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 14:20:50 +0200 Subject: [PATCH 14/30] fix(factory): reconcile run-state.py's exit ladder with the protocol Why: the ship re-review found six blockers in the exit ladder round 1 introduced. R1: argparse usage errors exited 2, the ladder's unappealable hard stop, so a mistyped phase or status ended an unattended run as if a secret had been found; a Parser subclass now reports usage errors as BAD_CALL. R2: check-result called a missing or unparseable result file a bad call while factory-run.md, run/SKILL.md and the spec's Error handling all call it a failed attempt; it is now exit 1 with the reason, since a subagent dying before writing its result is a phase outcome the orchestrator can relaunch. R3: load_state validated only the top level, so a null or list-shaped phases or scenario_texts crashed out as exit 1, colliding with diff-spec's drift code; the shape the commands rely on is validated and reported as BAD_CALL. R4: the hard-stop leniency covered stop but not status, so a secret reported alongside status done or failed was not read as a stop; a present stop kind now wins over status. R5: attempt silently rewrote a closed attempt and recorded an unreadable result as null while printing success; reclosing is refused and an unusable result file is recorded as a failed attempt with its reason. R6: tests for a second attempt on one phase and for every behavior above. --- factory/evals/tests/test_run_state.py | 177 +++++++++++++++++- factory/skills/run/SKILL.md | 2 +- factory/skills/run/scripts/run-state.py | 142 +++++++++++--- plugins/factory/skills/run/SKILL.md | 2 +- .../factory/skills/run/scripts/run-state.py | 142 +++++++++++--- 5 files changed, 395 insertions(+), 70 deletions(-) diff --git a/factory/evals/tests/test_run_state.py b/factory/evals/tests/test_run_state.py index 57e971b..e72511b 100644 --- a/factory/evals/tests/test_run_state.py +++ b/factory/evals/tests/test_run_state.py @@ -232,12 +232,13 @@ def test_check_result_failed_exits_1_with_reason(self) -> None: self.assertEqual(result.returncode, 1) self.assertIn("no result file", result.stdout) - def test_check_result_invalid_json_exits_3(self) -> None: + def test_check_result_invalid_json_is_a_failed_attempt(self) -> None: path = self.cwd / "bad.json" path.write_text("{not json", encoding="utf-8") result = run_state(self.cwd, "check-result", str(path)) - self.assertEqual(result.returncode, 3) + self.assertEqual(result.returncode, 1) self.assertIn("unparseable", result.stdout) + self.assertTrue(result.stdout.startswith("failed:"), result.stdout) def test_check_result_stopped_exits_2_with_kind(self) -> None: path = self.write_result({"status": "stopped", "stop": {"kind": "secret.found", "action": "purge it"}}) @@ -245,9 +246,10 @@ def test_check_result_stopped_exits_2_with_kind(self) -> None: self.assertEqual(result.returncode, 2) self.assertIn("secret.found", result.stdout) - def test_check_result_missing_file_exits_3(self) -> None: + def test_check_result_missing_file_is_a_failed_attempt(self) -> None: result = run_state(self.cwd, "check-result", str(self.cwd / "nope.json")) - self.assertEqual(result.returncode, 3) + self.assertEqual(result.returncode, 1) + self.assertIn("no result file", result.stdout) def test_check_result_unknown_extra_key_exits_0(self) -> None: path = self.write_result({"status": "done", "totally_unknown_field": 42}) @@ -429,11 +431,11 @@ def test_record_rejects_stdin_json_that_is_not_an_object(self) -> None: self.assertIn("not a decision object", result.stdout) self.assertEqual(before, (self.plan_dir / "factory-run.json").read_text()) - def test_check_result_with_a_json_array_exits_3(self) -> None: + def test_check_result_with_a_json_array_is_a_failed_attempt(self) -> None: path = self.cwd / "array.json" path.write_text(json.dumps([{"status": "done"}]), encoding="utf-8") result = run_state(self.cwd, "check-result", str(path)) - self.assertEqual(result.returncode, 3) + self.assertEqual(result.returncode, 1) self.assertIn("not a JSON object", result.stdout) def test_check_result_stopped_without_a_kind_prints_an_empty_kind(self) -> None: @@ -457,6 +459,169 @@ def test_attempt_with_a_non_object_state_file_exits_3(self) -> None: self.assertEqual(result.returncode, 3) self.assertIn("not a JSON object", result.stdout) + def test_check_result_on_a_directory_is_a_failed_attempt(self) -> None: + target = self.cwd / "results" + target.mkdir() + result = run_state(self.cwd, "check-result", str(target)) + self.assertEqual(result.returncode, 1) + self.assertIn("no result file", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_with_an_unknown_phase_exits_3(self) -> None: + self.init_plan() + result = run_state(self.cwd, "attempt", "fixture-plan", "deploy") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_with_an_unknown_status_exits_3(self) -> None: + self.init_plan() + result = run_state(self.cwd, "attempt", "fixture-plan", "build", "--status", "weird") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_an_unknown_subcommand_exits_3(self) -> None: + result = run_state(self.cwd, "frobnicate", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def write_state(self, state: dict) -> None: + (self.plan_dir / "factory-run.json").write_text(json.dumps(state), encoding="utf-8") + + def test_show_with_a_list_phases_state_exits_3(self) -> None: + self.init_plan() + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state["phases"] = [] + self.write_state(state) + result = run_state(self.cwd, "show", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertIn("unusable state file", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_with_a_null_phases_state_exits_3(self) -> None: + self.init_plan() + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state["phases"] = None + self.write_state(state) + result = run_state(self.cwd, "attempt", "fixture-plan", "build") + self.assertEqual(result.returncode, 3) + self.assertIn("unusable state file", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_diff_spec_with_a_null_scenario_texts_state_exits_3_not_1(self) -> None: + self.init_plan() + self.write_spec() + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state["scenario_texts"] = None + self.write_state(state) + result = run_state(self.cwd, "diff-spec", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertIn("unusable state file", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_with_a_phase_that_is_not_a_list_exits_3(self) -> None: + self.init_plan() + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state["phases"]["build"] = {} + self.write_state(state) + result = run_state(self.cwd, "attempt", "fixture-plan", "build") + self.assertEqual(result.returncode, 3) + self.assertIn("not a list of attempts", result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_show_with_a_non_object_attempt_entry_exits_3(self) -> None: + self.init_plan() + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + state["phases"]["build"] = ["done"] + self.write_state(state) + result = run_state(self.cwd, "show", "fixture-plan") + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_check_result_failed_status_with_a_stop_kind_still_exits_2(self) -> None: + path = self.write_result({"status": "failed", "stop": {"kind": "secret.found"}}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 2) + self.assertIn("secret.found", result.stdout) + + def test_check_result_done_status_with_a_stop_kind_still_exits_2(self) -> None: + path = self.write_result({"status": "done", "stop": {"kind": "action.destructive"}}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 2) + self.assertIn("action.destructive", result.stdout) + + def test_check_result_done_with_an_empty_stop_object_exits_0(self) -> None: + path = self.write_result({"status": "done", "stop": {}}) + result = run_state(self.cwd, "check-result", str(path)) + self.assertEqual(result.returncode, 0) + + def test_attempt_records_a_second_attempt_on_the_same_phase(self) -> None: + self.init_plan() + run_state(self.cwd, "attempt", "fixture-plan", "build") + first = self.write_result({"status": "failed", "reason": "flaky"}) + run_state( + self.cwd, "attempt", "fixture-plan", "build", "--status", "failed", "--result", str(first) + ) + relaunched = run_state(self.cwd, "attempt", "fixture-plan", "build") + self.assertEqual(relaunched.returncode, 0, relaunched.stderr) + self.assertIn("build attempt 2: launched", relaunched.stdout) + second = self.cwd / "result-2.json" + second.write_text(json.dumps({"status": "done"}), encoding="utf-8") + closed = run_state( + self.cwd, "attempt", "fixture-plan", "build", "--status", "done", "--result", str(second) + ) + self.assertEqual(closed.returncode, 0, closed.stderr) + self.assertIn("build attempt 2: done", closed.stdout) + attempts = json.loads((self.plan_dir / "factory-run.json").read_text())["phases"]["build"] + self.assertEqual([a["status"] for a in attempts], ["failed", "done"]) + self.assertEqual(attempts[0]["result"]["reason"], "flaky") + self.assertEqual(attempts[1]["result"]["status"], "done") + + def test_attempt_refuses_to_reclose_a_closed_attempt(self) -> None: + self.init_plan() + run_state(self.cwd, "attempt", "fixture-plan", "build") + path = self.write_result({"status": "failed", "reason": "flaky"}) + run_state( + self.cwd, "attempt", "fixture-plan", "build", "--status", "failed", "--result", str(path) + ) + before = (self.plan_dir / "factory-run.json").read_text() + again = run_state(self.cwd, "attempt", "fixture-plan", "build", "--status", "done") + self.assertEqual(again.returncode, 3) + self.assertIn("already closed", again.stdout) + self.assertEqual(before, (self.plan_dir / "factory-run.json").read_text()) + + def test_attempt_with_a_missing_result_file_records_a_failed_attempt(self) -> None: + self.init_plan() + run_state(self.cwd, "attempt", "fixture-plan", "build") + closed = run_state( + self.cwd, + "attempt", + "fixture-plan", + "build", + "--status", + "done", + "--result", + str(self.cwd / "nope.json"), + ) + self.assertEqual(closed.returncode, 0, closed.stderr) + self.assertIn("no result file", closed.stdout) + attempt = json.loads((self.plan_dir / "factory-run.json").read_text())["phases"]["build"][0] + self.assertEqual(attempt["status"], "failed") + self.assertEqual(attempt["result"]["status"], "failed") + self.assertIn("no result file", attempt["result"]["reason"]) + + def test_attempt_with_a_json_array_result_file_records_a_failed_attempt(self) -> None: + self.init_plan() + run_state(self.cwd, "attempt", "fixture-plan", "build") + path = self.cwd / "array.json" + path.write_text(json.dumps([{"status": "done"}]), encoding="utf-8") + closed = run_state( + self.cwd, "attempt", "fixture-plan", "build", "--status", "done", "--result", str(path) + ) + self.assertEqual(closed.returncode, 0, closed.stderr) + attempt = json.loads((self.plan_dir / "factory-run.json").read_text())["phases"]["build"][0] + self.assertEqual(attempt["status"], "failed") + self.assertIn("not a JSON object", attempt["result"]["reason"]) + def test_show_exits_0(self) -> None: self.init_plan() result = run_state(self.cwd, "show", "fixture-plan") diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index b9b607e..fc6f954 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -49,7 +49,7 @@ For each phase in order (`scope-review`, `build`, `ship`): 1. Take the attempt number first: `run-state.py attempt {plan} {phase}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. 2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. 3. Wait for the host's completion signal - a batch in flight is not over. -4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. +4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. 5. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). 6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. 7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. diff --git a/factory/skills/run/scripts/run-state.py b/factory/skills/run/scripts/run-state.py index f6c2fc0..9468d5d 100755 --- a/factory/skills/run/scripts/run-state.py +++ b/factory/skills/run/scripts/run-state.py @@ -14,15 +14,21 @@ init 0 written; 1 a state file already exists handoff 0 recorded; 1 no spec.md at the path; 3 unusable state file diff-spec 0 no drift; 1 drift found; 3 unusable state file or no spec.md - attempt 0 recorded; 3 unusable state file, or no launched attempt to close + attempt 0 recorded; 3 unusable state file, no launched attempt to close, + or an attempt already closed record 0 recorded; 1 action outside the four; 3 unreadable stdin JSON or unusable state file - check-result 0 done; 1 failed; 2 stopped; 3 missing or unparseable result file + check-result 0 done; 1 failed, a missing or unparseable result file included; + 2 stopped show 0 printed; 3 unusable state file -3 always means the call itself could not be carried out, never a phase outcome, -so a malformed state file or a bad call never reaches a judgment. Every function -here stays at or under cyclomatic complexity 10 (D-complexity-threshold). +3 always means the call itself could not be carried out - an unusable state +file, a usage error, or an attempt that cannot be closed - never a phase +outcome, so a bad call never reaches a judgment. A missing or unparseable +*result* file is a phase outcome, not a bad call: a subagent that dies before +writing its result is the likeliest real failure, so it is a failed attempt the +orchestrator can relaunch. Every function here stays at or under cyclomatic +complexity 10 (D-complexity-threshold). """ from __future__ import annotations @@ -34,6 +40,7 @@ import subprocess import sys from pathlib import Path +from typing import NoReturn HEADING = re.compile(r"^##\s+(.*?)\s*$") CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") @@ -45,6 +52,13 @@ ATTEMPT_STATUSES = ("launched", "done", "failed", "stopped") STATE_NAME = "factory-run.json" BAD_CALL = 3 +STATE_SHAPE = { + "phases": dict, + "scenario_texts": dict, + "not_doing_lines": list, + "decisions": list, + "repairs": list, +} def state_path(plan: str) -> Path: @@ -55,6 +69,25 @@ def spec_path(plan: str, override: str | None) -> Path: return Path(override) if override else Path(".dev") / plan / "spec.md" +def shape_problem(state: dict) -> str: + """The first field whose shape the commands rely on is wrong, or an empty string. + + Only the fields this script reads back are checked, and only when present: + a state file written by `init` carries them all, and a null or wrongly typed + one would otherwise crash a command rather than report a bad call. + """ + for field, kind in STATE_SHAPE.items(): + if field in state and not isinstance(state[field], kind): + found = type(state[field]).__name__ + return f"{field} is {found}, expected {kind.__name__}" + for phase, attempts in state.get("phases", {}).items(): + if not isinstance(attempts, list): + return f"phases.{phase} is not a list of attempts" + if any(not isinstance(attempt, dict) for attempt in attempts): + return f"phases.{phase} holds an attempt that is not an object" + return "" + + def load_state(plan: str) -> dict | None: """The parsed state file, or None (with the reason printed) when it is unusable.""" path = state_path(plan) @@ -66,6 +99,10 @@ def load_state(plan: str) -> dict | None: if not isinstance(state, dict): print(f"unusable state file {path}: not a JSON object") return None + problem = shape_problem(state) + if problem: + print(f"unusable state file {path}: {problem}") + return None return state @@ -252,13 +289,21 @@ def cmd_record(args: argparse.Namespace) -> int: return 0 -def read_result(path: Path) -> dict | None: - """The parsed result file, or None when it is missing, unreadable, or not an object.""" +def read_result(path: Path) -> tuple[dict | None, str]: + """The parsed result object, or None and why the attempt counts as failed. + + A result file that is missing, unreadable, or not a JSON object is a phase + outcome, not a bad call: the phase died before writing a usable result. + """ + if not path.is_file(): + return None, f"no result file: {path}" try: data = json.loads(path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError): - return None - return data if isinstance(data, dict) else None + except (OSError, json.JSONDecodeError) as exc: + return None, f"unparseable result file {path}: {exc}" + if not isinstance(data, dict): + return None, f"unparseable result file {path}: not a JSON object" + return data, "" def stop_kind(data: dict) -> str: @@ -275,33 +320,50 @@ def stop_kind(data: dict) -> str: return "" -def cmd_check_result(args: argparse.Namespace) -> int: - path = Path(args.path) - if not path.is_file(): - print(f"missing result file: {path}") - return BAD_CALL - try: - data = json.loads(path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError) as exc: - print(f"unparseable result file {path}: {exc}") - return BAD_CALL - if not isinstance(data, dict): - print(f"unparseable result file {path}: not a JSON object") - return BAD_CALL +def report_result(data: dict) -> int: + """Print the result's one-line verdict and return its exit code. + + A present stop kind wins over `status`: a stop is unappealable and ending a + run a phase did not mean to stop is recoverable, while continuing past a + found secret is not. + """ + kind = stop_kind(data) status = data.get("status") + if kind or status == "stopped": + print(f"stopped: {kind}") + return 2 if status == "done": print("done") return 0 if status == "failed": print(f"failed: {data.get('reason', '')}") return 1 - if status == "stopped": - print(f"stopped: {stop_kind(data)}") - return 2 print(f"failed: unrecognized status {status!r}") return 1 +def cmd_check_result(args: argparse.Namespace) -> int: + data, reason = read_result(Path(args.path)) + if data is None: + print(f"failed: {reason}") + return 1 + return report_result(data) + + +def close_attempt(attempt: dict, status: str, result: str | None) -> None: + """Close a launched attempt, recording an unusable result file as a failure.""" + attempt["status"] = status + if not result: + return + data, reason = read_result(Path(result)) + if data is None: + attempt["status"] = "failed" + attempt["result"] = {"status": "failed", "reason": reason} + print(f"recorded as a failed attempt: {reason}") + return + attempt["result"] = data + + def cmd_attempt(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: @@ -312,12 +374,17 @@ def cmd_attempt(args: argparse.Namespace) -> int: elif not attempts: print(f"no launched attempt of {args.phase} to close") return BAD_CALL + elif attempts[-1].get("status") != "launched": + closed = attempts[-1].get("status") + print( + f"{args.phase} attempt {len(attempts)} is already closed as {closed!r}; " + f"record a new launch instead of reclosing it" + ) + return BAD_CALL else: - attempts[-1]["status"] = args.status - if args.result: - attempts[-1]["result"] = read_result(Path(args.result)) + close_attempt(attempts[-1], args.status, args.result) save_state(args.plan, state) - print(f"{args.phase} attempt {len(attempts)}: {args.status}") + print(f"{args.phase} attempt {len(attempts)}: {attempts[-1]['status']}") return 0 @@ -336,8 +403,21 @@ def cmd_show(args: argparse.Namespace) -> int: return 0 +class Parser(argparse.ArgumentParser): + """An ArgumentParser whose usage errors report BAD_CALL, not argparse's 2. + + 2 is the orchestrator's unappealable hard stop, so a mistyped phase or + status must not end an unattended run as if a secret had been found. + """ + + def error(self, message: str) -> NoReturn: + self.print_usage(sys.stderr) + print(f"{self.prog}: bad call: {message}", file=sys.stderr) + raise SystemExit(BAD_CALL) + + def build_parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser(description=__doc__) + parser = Parser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) init_p = sub.add_parser("init") diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index 6ec02e0..575758c 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -48,7 +48,7 @@ For each phase in order (`scope-review`, `build`, `ship`): 1. Take the attempt number first: `run-state.py attempt {plan} {phase}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. 2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. 3. Wait for the host's completion signal - a batch in flight is not over. -4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (its own exit code, never a crash). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. +4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. 5. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). 6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. 7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. diff --git a/plugins/factory/skills/run/scripts/run-state.py b/plugins/factory/skills/run/scripts/run-state.py index f6c2fc0..9468d5d 100755 --- a/plugins/factory/skills/run/scripts/run-state.py +++ b/plugins/factory/skills/run/scripts/run-state.py @@ -14,15 +14,21 @@ init 0 written; 1 a state file already exists handoff 0 recorded; 1 no spec.md at the path; 3 unusable state file diff-spec 0 no drift; 1 drift found; 3 unusable state file or no spec.md - attempt 0 recorded; 3 unusable state file, or no launched attempt to close + attempt 0 recorded; 3 unusable state file, no launched attempt to close, + or an attempt already closed record 0 recorded; 1 action outside the four; 3 unreadable stdin JSON or unusable state file - check-result 0 done; 1 failed; 2 stopped; 3 missing or unparseable result file + check-result 0 done; 1 failed, a missing or unparseable result file included; + 2 stopped show 0 printed; 3 unusable state file -3 always means the call itself could not be carried out, never a phase outcome, -so a malformed state file or a bad call never reaches a judgment. Every function -here stays at or under cyclomatic complexity 10 (D-complexity-threshold). +3 always means the call itself could not be carried out - an unusable state +file, a usage error, or an attempt that cannot be closed - never a phase +outcome, so a bad call never reaches a judgment. A missing or unparseable +*result* file is a phase outcome, not a bad call: a subagent that dies before +writing its result is the likeliest real failure, so it is a failed attempt the +orchestrator can relaunch. Every function here stays at or under cyclomatic +complexity 10 (D-complexity-threshold). """ from __future__ import annotations @@ -34,6 +40,7 @@ import subprocess import sys from pathlib import Path +from typing import NoReturn HEADING = re.compile(r"^##\s+(.*?)\s*$") CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") @@ -45,6 +52,13 @@ ATTEMPT_STATUSES = ("launched", "done", "failed", "stopped") STATE_NAME = "factory-run.json" BAD_CALL = 3 +STATE_SHAPE = { + "phases": dict, + "scenario_texts": dict, + "not_doing_lines": list, + "decisions": list, + "repairs": list, +} def state_path(plan: str) -> Path: @@ -55,6 +69,25 @@ def spec_path(plan: str, override: str | None) -> Path: return Path(override) if override else Path(".dev") / plan / "spec.md" +def shape_problem(state: dict) -> str: + """The first field whose shape the commands rely on is wrong, or an empty string. + + Only the fields this script reads back are checked, and only when present: + a state file written by `init` carries them all, and a null or wrongly typed + one would otherwise crash a command rather than report a bad call. + """ + for field, kind in STATE_SHAPE.items(): + if field in state and not isinstance(state[field], kind): + found = type(state[field]).__name__ + return f"{field} is {found}, expected {kind.__name__}" + for phase, attempts in state.get("phases", {}).items(): + if not isinstance(attempts, list): + return f"phases.{phase} is not a list of attempts" + if any(not isinstance(attempt, dict) for attempt in attempts): + return f"phases.{phase} holds an attempt that is not an object" + return "" + + def load_state(plan: str) -> dict | None: """The parsed state file, or None (with the reason printed) when it is unusable.""" path = state_path(plan) @@ -66,6 +99,10 @@ def load_state(plan: str) -> dict | None: if not isinstance(state, dict): print(f"unusable state file {path}: not a JSON object") return None + problem = shape_problem(state) + if problem: + print(f"unusable state file {path}: {problem}") + return None return state @@ -252,13 +289,21 @@ def cmd_record(args: argparse.Namespace) -> int: return 0 -def read_result(path: Path) -> dict | None: - """The parsed result file, or None when it is missing, unreadable, or not an object.""" +def read_result(path: Path) -> tuple[dict | None, str]: + """The parsed result object, or None and why the attempt counts as failed. + + A result file that is missing, unreadable, or not a JSON object is a phase + outcome, not a bad call: the phase died before writing a usable result. + """ + if not path.is_file(): + return None, f"no result file: {path}" try: data = json.loads(path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError): - return None - return data if isinstance(data, dict) else None + except (OSError, json.JSONDecodeError) as exc: + return None, f"unparseable result file {path}: {exc}" + if not isinstance(data, dict): + return None, f"unparseable result file {path}: not a JSON object" + return data, "" def stop_kind(data: dict) -> str: @@ -275,33 +320,50 @@ def stop_kind(data: dict) -> str: return "" -def cmd_check_result(args: argparse.Namespace) -> int: - path = Path(args.path) - if not path.is_file(): - print(f"missing result file: {path}") - return BAD_CALL - try: - data = json.loads(path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError) as exc: - print(f"unparseable result file {path}: {exc}") - return BAD_CALL - if not isinstance(data, dict): - print(f"unparseable result file {path}: not a JSON object") - return BAD_CALL +def report_result(data: dict) -> int: + """Print the result's one-line verdict and return its exit code. + + A present stop kind wins over `status`: a stop is unappealable and ending a + run a phase did not mean to stop is recoverable, while continuing past a + found secret is not. + """ + kind = stop_kind(data) status = data.get("status") + if kind or status == "stopped": + print(f"stopped: {kind}") + return 2 if status == "done": print("done") return 0 if status == "failed": print(f"failed: {data.get('reason', '')}") return 1 - if status == "stopped": - print(f"stopped: {stop_kind(data)}") - return 2 print(f"failed: unrecognized status {status!r}") return 1 +def cmd_check_result(args: argparse.Namespace) -> int: + data, reason = read_result(Path(args.path)) + if data is None: + print(f"failed: {reason}") + return 1 + return report_result(data) + + +def close_attempt(attempt: dict, status: str, result: str | None) -> None: + """Close a launched attempt, recording an unusable result file as a failure.""" + attempt["status"] = status + if not result: + return + data, reason = read_result(Path(result)) + if data is None: + attempt["status"] = "failed" + attempt["result"] = {"status": "failed", "reason": reason} + print(f"recorded as a failed attempt: {reason}") + return + attempt["result"] = data + + def cmd_attempt(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: @@ -312,12 +374,17 @@ def cmd_attempt(args: argparse.Namespace) -> int: elif not attempts: print(f"no launched attempt of {args.phase} to close") return BAD_CALL + elif attempts[-1].get("status") != "launched": + closed = attempts[-1].get("status") + print( + f"{args.phase} attempt {len(attempts)} is already closed as {closed!r}; " + f"record a new launch instead of reclosing it" + ) + return BAD_CALL else: - attempts[-1]["status"] = args.status - if args.result: - attempts[-1]["result"] = read_result(Path(args.result)) + close_attempt(attempts[-1], args.status, args.result) save_state(args.plan, state) - print(f"{args.phase} attempt {len(attempts)}: {args.status}") + print(f"{args.phase} attempt {len(attempts)}: {attempts[-1]['status']}") return 0 @@ -336,8 +403,21 @@ def cmd_show(args: argparse.Namespace) -> int: return 0 +class Parser(argparse.ArgumentParser): + """An ArgumentParser whose usage errors report BAD_CALL, not argparse's 2. + + 2 is the orchestrator's unappealable hard stop, so a mistyped phase or + status must not end an unattended run as if a secret had been found. + """ + + def error(self, message: str) -> NoReturn: + self.print_usage(sys.stderr) + print(f"{self.prog}: bad call: {message}", file=sys.stderr) + raise SystemExit(BAD_CALL) + + def build_parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser(description=__doc__) + parser = Parser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) init_p = sub.add_parser("init") From 00d52264b56a85685783ae24f44c3b5acfd27413 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 16:44:27 +0200 Subject: [PATCH 15/30] fix(harden): clear the static analysis findings on the branch The ship gauntlet's ruff pass over the diff found 23 findings on added lines. Fixed at the source rather than suppressed: - re module aliases spelled out (re.S, re.M, re.I -> re.DOTALL, re.MULTILINE, re.IGNORECASE) in architecture-check.py, pr-evidence.py and aggregate-findings.py. - explicit check= on every subprocess.run that reads returncode itself. - skill-metrics.py numstat() leaked a file handle counting untracked lines; it now reads under a context manager. - two adjacent f-strings inside a collection literal made explicit as one value in skill-metrics.py and pr-evidence.py. Neither was a bug: both sites build a one-element row and a three-element tuple would have crashed the consumer. - the unused REJECTED constant in lint-spec.py deleted; the glyph is still matched through the ALTERNATIVE character class. The dev and factory copies of each shared script stay byte-identical, and plugins/ is regenerated rather than hand-edited. --- dev/scripts/architecture-check.py | 6 +++--- dev/scripts/skill-metrics.py | 9 ++++++--- dev/skills/scope/scripts/lint-spec.py | 2 +- dev/skills/ship/scripts/aggregate-findings.py | 2 +- dev/skills/ship/scripts/pr-evidence.py | 16 ++++++++-------- factory/evals/tests/test_fixture_setup.py | 2 +- factory/scripts/architecture-check.py | 6 +++--- factory/scripts/skill-metrics.py | 9 ++++++--- factory/skills/scope/scripts/lint-spec.py | 2 +- .../skills/ship/scripts/aggregate-findings.py | 2 +- factory/skills/ship/scripts/pr-evidence.py | 16 ++++++++-------- plugins/dev/scripts/architecture-check.py | 6 +++--- plugins/dev/scripts/skill-metrics.py | 9 ++++++--- plugins/dev/skills/scope/scripts/lint-spec.py | 2 +- .../skills/ship/scripts/aggregate-findings.py | 2 +- plugins/dev/skills/ship/scripts/pr-evidence.py | 16 ++++++++-------- plugins/factory/scripts/architecture-check.py | 6 +++--- plugins/factory/scripts/skill-metrics.py | 9 ++++++--- .../factory/skills/scope/scripts/lint-spec.py | 2 +- .../skills/ship/scripts/aggregate-findings.py | 2 +- .../factory/skills/ship/scripts/pr-evidence.py | 16 ++++++++-------- 21 files changed, 77 insertions(+), 65 deletions(-) diff --git a/dev/scripts/architecture-check.py b/dev/scripts/architecture-check.py index 3f2417a..495cf2d 100755 --- a/dev/scripts/architecture-check.py +++ b/dev/scripts/architecture-check.py @@ -20,7 +20,7 @@ from pathlib import Path SECTIONS = ["Components", "Flows", "Boundaries", "Cross-cutting", "Entry points"] -HEADER = re.compile(r"^##\s+(.+?)\s*$", re.M) +HEADER = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE) BACKTICK = re.compile(r"`([^`\n]+)`") PATH_LIKE = re.compile(r"^[^\s]+(/[^\s]*|\.[A-Za-z0-9]{1,6})$") TABLE_ROW = re.compile(r"^\|(.+)\|\s*$") @@ -90,9 +90,9 @@ def main() -> int: text = map_path.read_text(encoding="utf-8", errors="replace") violations: list[str] = [] - if not re.search(r"^Purpose:", text, re.M): + if not re.search(r"^Purpose:", text, re.MULTILINE): violations.append("section: no 'Purpose:' line") - if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.M): + if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.MULTILINE): violations.append("section: no dated 'Captured:' line") found = sections(text) order = [name for name in found if name in SECTIONS] diff --git a/dev/scripts/skill-metrics.py b/dev/scripts/skill-metrics.py index d9d2b7b..ac78ef0 100755 --- a/dev/scripts/skill-metrics.py +++ b/dev/scripts/skill-metrics.py @@ -148,7 +148,8 @@ def numstat(ref: str | None) -> tuple[int, int, set[str]]: for path in untracked: files.add(path) try: - added += sum(1 for _ in Path(path).open(encoding="utf-8", errors="ignore")) + with Path(path).open(encoding="utf-8", errors="ignore") as handle: + added += sum(1 for _ in handle) except OSError: pass return added, removed, files @@ -295,8 +296,10 @@ def cmd_end(skill: str, counts: dict[str, str]) -> int: change = (total(all_tokens) - med_tokens) / med_tokens * 100 if med_tokens else 0 rows.append(( f"vs previous {skill} runs", - f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " - f"this run {change:+.0f}% tokens", + ( + f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " + f"this run {change:+.0f}% tokens" + ), )) width = max(len(name) for name, _ in rows) diff --git a/dev/skills/scope/scripts/lint-spec.py b/dev/skills/scope/scripts/lint-spec.py index 2056f78..c45fdde 100755 --- a/dev/skills/scope/scripts/lint-spec.py +++ b/dev/skills/scope/scripts/lint-spec.py @@ -21,7 +21,7 @@ LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") -CHOSEN, REJECTED, OPEN, NOT_DOING = "✓", "✗", "?", "⊘" +CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: diff --git a/dev/skills/ship/scripts/aggregate-findings.py b/dev/skills/ship/scripts/aggregate-findings.py index 81d86d9..32238d2 100755 --- a/dev/skills/ship/scripts/aggregate-findings.py +++ b/dev/skills/ship/scripts/aggregate-findings.py @@ -18,7 +18,7 @@ import sys from pathlib import Path -FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S) +FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.DOTALL) RESULT_SUFFIXES = (".json", ".out") diff --git a/dev/skills/ship/scripts/pr-evidence.py b/dev/skills/ship/scripts/pr-evidence.py index 63a448f..a913e8d 100755 --- a/dev/skills/ship/scripts/pr-evidence.py +++ b/dev/skills/ship/scripts/pr-evidence.py @@ -36,16 +36,16 @@ from datetime import datetime, timezone from pathlib import Path -DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.S) +DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.DOTALL) IDENT = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*") -PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.I) +PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.IGNORECASE) IMAGE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)") PAIR_LABELS = { "before": "before", "after": "after", "on merge base": "base", "on the merge base": "base", "merge base": "base", "on this branch": "branch", "on the branch": "branch", "this branch": "branch", } -LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.I) +LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.IGNORECASE) FENCE_OPEN = re.compile(r"^\s*```") IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".webp"} @@ -138,7 +138,7 @@ def node_json(literal: str) -> str | None: handle.write(literal) script = f"process.stdout.write(JSON.stringify(eval('(' + require('fs').readFileSync({handle.name!r}, 'utf8') + ')')))" try: - result = subprocess.run(["node", "-e", script], capture_output=True, text=True) + result = subprocess.run(["node", "-e", script], capture_output=True, text=True, check=False) return result.stdout if result.returncode == 0 else None except OSError: return None @@ -147,7 +147,7 @@ def node_json(literal: str) -> str | None: def git(*args: str, env: dict | None = None, check: bool = True) -> str: - result = subprocess.run(["git", *args], capture_output=True, text=True, env=env) + result = subprocess.run(["git", *args], capture_output=True, text=True, env=env, check=False) if check and result.returncode != 0: fail(f"git {' '.join(args)} failed: {result.stderr.strip()}") return result.stdout.strip() @@ -188,7 +188,7 @@ def extract(args: argparse.Namespace) -> None: lines = ["## Evidence", "", f"Captured from the e2e run of `{plan}` at {data.get('generatedAt', 'unknown time')}: " - f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] + + f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] written: list[str] = [] for index, scenario in enumerate(scenarios, 1): scenario_id = slug(str(scenario.get("id") or scenario.get("title") or index), f"scenario-{index}") @@ -226,7 +226,7 @@ def publish(args: argparse.Namespace) -> None: if not files: fail(f"nothing to publish: no image files under {root}") ref = f"refs/remotes/{args.remote}/{args.branch}" - subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True) + subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True, check=False) parent = git("rev-parse", "--verify", "--quiet", ref, check=False) or None with tempfile.TemporaryDirectory() as tmp: env = {**os.environ, "GIT_INDEX_FILE": str(Path(tmp) / "index")} @@ -243,7 +243,7 @@ def publish(args: argparse.Namespace) -> None: def evidence_section(body: str) -> str | None: - match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.M | re.S) + match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.MULTILINE | re.DOTALL) return match.group(1) if match else None diff --git a/factory/evals/tests/test_fixture_setup.py b/factory/evals/tests/test_fixture_setup.py index 719ed5f..5efc367 100644 --- a/factory/evals/tests/test_fixture_setup.py +++ b/factory/evals/tests/test_fixture_setup.py @@ -33,7 +33,7 @@ def setUp(self) -> None: def git(self, *args: str) -> subprocess.CompletedProcess: return subprocess.run( - ["git", *args], cwd=self.dest, capture_output=True, text=True + ["git", *args], cwd=self.dest, check=False, capture_output=True, text=True ) def test_setup_produces_a_main_repo_with_one_commit_and_a_bare_origin(self) -> None: diff --git a/factory/scripts/architecture-check.py b/factory/scripts/architecture-check.py index 3f2417a..495cf2d 100755 --- a/factory/scripts/architecture-check.py +++ b/factory/scripts/architecture-check.py @@ -20,7 +20,7 @@ from pathlib import Path SECTIONS = ["Components", "Flows", "Boundaries", "Cross-cutting", "Entry points"] -HEADER = re.compile(r"^##\s+(.+?)\s*$", re.M) +HEADER = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE) BACKTICK = re.compile(r"`([^`\n]+)`") PATH_LIKE = re.compile(r"^[^\s]+(/[^\s]*|\.[A-Za-z0-9]{1,6})$") TABLE_ROW = re.compile(r"^\|(.+)\|\s*$") @@ -90,9 +90,9 @@ def main() -> int: text = map_path.read_text(encoding="utf-8", errors="replace") violations: list[str] = [] - if not re.search(r"^Purpose:", text, re.M): + if not re.search(r"^Purpose:", text, re.MULTILINE): violations.append("section: no 'Purpose:' line") - if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.M): + if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.MULTILINE): violations.append("section: no dated 'Captured:' line") found = sections(text) order = [name for name in found if name in SECTIONS] diff --git a/factory/scripts/skill-metrics.py b/factory/scripts/skill-metrics.py index d9d2b7b..ac78ef0 100755 --- a/factory/scripts/skill-metrics.py +++ b/factory/scripts/skill-metrics.py @@ -148,7 +148,8 @@ def numstat(ref: str | None) -> tuple[int, int, set[str]]: for path in untracked: files.add(path) try: - added += sum(1 for _ in Path(path).open(encoding="utf-8", errors="ignore")) + with Path(path).open(encoding="utf-8", errors="ignore") as handle: + added += sum(1 for _ in handle) except OSError: pass return added, removed, files @@ -295,8 +296,10 @@ def cmd_end(skill: str, counts: dict[str, str]) -> int: change = (total(all_tokens) - med_tokens) / med_tokens * 100 if med_tokens else 0 rows.append(( f"vs previous {skill} runs", - f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " - f"this run {change:+.0f}% tokens", + ( + f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " + f"this run {change:+.0f}% tokens" + ), )) width = max(len(name) for name, _ in rows) diff --git a/factory/skills/scope/scripts/lint-spec.py b/factory/skills/scope/scripts/lint-spec.py index 2056f78..c45fdde 100755 --- a/factory/skills/scope/scripts/lint-spec.py +++ b/factory/skills/scope/scripts/lint-spec.py @@ -21,7 +21,7 @@ LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") -CHOSEN, REJECTED, OPEN, NOT_DOING = "✓", "✗", "?", "⊘" +CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: diff --git a/factory/skills/ship/scripts/aggregate-findings.py b/factory/skills/ship/scripts/aggregate-findings.py index 81d86d9..32238d2 100755 --- a/factory/skills/ship/scripts/aggregate-findings.py +++ b/factory/skills/ship/scripts/aggregate-findings.py @@ -18,7 +18,7 @@ import sys from pathlib import Path -FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S) +FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.DOTALL) RESULT_SUFFIXES = (".json", ".out") diff --git a/factory/skills/ship/scripts/pr-evidence.py b/factory/skills/ship/scripts/pr-evidence.py index 63a448f..a913e8d 100755 --- a/factory/skills/ship/scripts/pr-evidence.py +++ b/factory/skills/ship/scripts/pr-evidence.py @@ -36,16 +36,16 @@ from datetime import datetime, timezone from pathlib import Path -DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.S) +DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.DOTALL) IDENT = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*") -PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.I) +PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.IGNORECASE) IMAGE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)") PAIR_LABELS = { "before": "before", "after": "after", "on merge base": "base", "on the merge base": "base", "merge base": "base", "on this branch": "branch", "on the branch": "branch", "this branch": "branch", } -LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.I) +LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.IGNORECASE) FENCE_OPEN = re.compile(r"^\s*```") IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".webp"} @@ -138,7 +138,7 @@ def node_json(literal: str) -> str | None: handle.write(literal) script = f"process.stdout.write(JSON.stringify(eval('(' + require('fs').readFileSync({handle.name!r}, 'utf8') + ')')))" try: - result = subprocess.run(["node", "-e", script], capture_output=True, text=True) + result = subprocess.run(["node", "-e", script], capture_output=True, text=True, check=False) return result.stdout if result.returncode == 0 else None except OSError: return None @@ -147,7 +147,7 @@ def node_json(literal: str) -> str | None: def git(*args: str, env: dict | None = None, check: bool = True) -> str: - result = subprocess.run(["git", *args], capture_output=True, text=True, env=env) + result = subprocess.run(["git", *args], capture_output=True, text=True, env=env, check=False) if check and result.returncode != 0: fail(f"git {' '.join(args)} failed: {result.stderr.strip()}") return result.stdout.strip() @@ -188,7 +188,7 @@ def extract(args: argparse.Namespace) -> None: lines = ["## Evidence", "", f"Captured from the e2e run of `{plan}` at {data.get('generatedAt', 'unknown time')}: " - f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] + + f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] written: list[str] = [] for index, scenario in enumerate(scenarios, 1): scenario_id = slug(str(scenario.get("id") or scenario.get("title") or index), f"scenario-{index}") @@ -226,7 +226,7 @@ def publish(args: argparse.Namespace) -> None: if not files: fail(f"nothing to publish: no image files under {root}") ref = f"refs/remotes/{args.remote}/{args.branch}" - subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True) + subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True, check=False) parent = git("rev-parse", "--verify", "--quiet", ref, check=False) or None with tempfile.TemporaryDirectory() as tmp: env = {**os.environ, "GIT_INDEX_FILE": str(Path(tmp) / "index")} @@ -243,7 +243,7 @@ def publish(args: argparse.Namespace) -> None: def evidence_section(body: str) -> str | None: - match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.M | re.S) + match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.MULTILINE | re.DOTALL) return match.group(1) if match else None diff --git a/plugins/dev/scripts/architecture-check.py b/plugins/dev/scripts/architecture-check.py index 3f2417a..495cf2d 100755 --- a/plugins/dev/scripts/architecture-check.py +++ b/plugins/dev/scripts/architecture-check.py @@ -20,7 +20,7 @@ from pathlib import Path SECTIONS = ["Components", "Flows", "Boundaries", "Cross-cutting", "Entry points"] -HEADER = re.compile(r"^##\s+(.+?)\s*$", re.M) +HEADER = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE) BACKTICK = re.compile(r"`([^`\n]+)`") PATH_LIKE = re.compile(r"^[^\s]+(/[^\s]*|\.[A-Za-z0-9]{1,6})$") TABLE_ROW = re.compile(r"^\|(.+)\|\s*$") @@ -90,9 +90,9 @@ def main() -> int: text = map_path.read_text(encoding="utf-8", errors="replace") violations: list[str] = [] - if not re.search(r"^Purpose:", text, re.M): + if not re.search(r"^Purpose:", text, re.MULTILINE): violations.append("section: no 'Purpose:' line") - if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.M): + if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.MULTILINE): violations.append("section: no dated 'Captured:' line") found = sections(text) order = [name for name in found if name in SECTIONS] diff --git a/plugins/dev/scripts/skill-metrics.py b/plugins/dev/scripts/skill-metrics.py index d9d2b7b..ac78ef0 100755 --- a/plugins/dev/scripts/skill-metrics.py +++ b/plugins/dev/scripts/skill-metrics.py @@ -148,7 +148,8 @@ def numstat(ref: str | None) -> tuple[int, int, set[str]]: for path in untracked: files.add(path) try: - added += sum(1 for _ in Path(path).open(encoding="utf-8", errors="ignore")) + with Path(path).open(encoding="utf-8", errors="ignore") as handle: + added += sum(1 for _ in handle) except OSError: pass return added, removed, files @@ -295,8 +296,10 @@ def cmd_end(skill: str, counts: dict[str, str]) -> int: change = (total(all_tokens) - med_tokens) / med_tokens * 100 if med_tokens else 0 rows.append(( f"vs previous {skill} runs", - f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " - f"this run {change:+.0f}% tokens", + ( + f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " + f"this run {change:+.0f}% tokens" + ), )) width = max(len(name) for name, _ in rows) diff --git a/plugins/dev/skills/scope/scripts/lint-spec.py b/plugins/dev/skills/scope/scripts/lint-spec.py index 2056f78..c45fdde 100755 --- a/plugins/dev/skills/scope/scripts/lint-spec.py +++ b/plugins/dev/skills/scope/scripts/lint-spec.py @@ -21,7 +21,7 @@ LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") -CHOSEN, REJECTED, OPEN, NOT_DOING = "✓", "✗", "?", "⊘" +CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: diff --git a/plugins/dev/skills/ship/scripts/aggregate-findings.py b/plugins/dev/skills/ship/scripts/aggregate-findings.py index 81d86d9..32238d2 100755 --- a/plugins/dev/skills/ship/scripts/aggregate-findings.py +++ b/plugins/dev/skills/ship/scripts/aggregate-findings.py @@ -18,7 +18,7 @@ import sys from pathlib import Path -FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S) +FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.DOTALL) RESULT_SUFFIXES = (".json", ".out") diff --git a/plugins/dev/skills/ship/scripts/pr-evidence.py b/plugins/dev/skills/ship/scripts/pr-evidence.py index 63a448f..a913e8d 100755 --- a/plugins/dev/skills/ship/scripts/pr-evidence.py +++ b/plugins/dev/skills/ship/scripts/pr-evidence.py @@ -36,16 +36,16 @@ from datetime import datetime, timezone from pathlib import Path -DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.S) +DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.DOTALL) IDENT = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*") -PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.I) +PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.IGNORECASE) IMAGE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)") PAIR_LABELS = { "before": "before", "after": "after", "on merge base": "base", "on the merge base": "base", "merge base": "base", "on this branch": "branch", "on the branch": "branch", "this branch": "branch", } -LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.I) +LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.IGNORECASE) FENCE_OPEN = re.compile(r"^\s*```") IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".webp"} @@ -138,7 +138,7 @@ def node_json(literal: str) -> str | None: handle.write(literal) script = f"process.stdout.write(JSON.stringify(eval('(' + require('fs').readFileSync({handle.name!r}, 'utf8') + ')')))" try: - result = subprocess.run(["node", "-e", script], capture_output=True, text=True) + result = subprocess.run(["node", "-e", script], capture_output=True, text=True, check=False) return result.stdout if result.returncode == 0 else None except OSError: return None @@ -147,7 +147,7 @@ def node_json(literal: str) -> str | None: def git(*args: str, env: dict | None = None, check: bool = True) -> str: - result = subprocess.run(["git", *args], capture_output=True, text=True, env=env) + result = subprocess.run(["git", *args], capture_output=True, text=True, env=env, check=False) if check and result.returncode != 0: fail(f"git {' '.join(args)} failed: {result.stderr.strip()}") return result.stdout.strip() @@ -188,7 +188,7 @@ def extract(args: argparse.Namespace) -> None: lines = ["## Evidence", "", f"Captured from the e2e run of `{plan}` at {data.get('generatedAt', 'unknown time')}: " - f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] + + f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] written: list[str] = [] for index, scenario in enumerate(scenarios, 1): scenario_id = slug(str(scenario.get("id") or scenario.get("title") or index), f"scenario-{index}") @@ -226,7 +226,7 @@ def publish(args: argparse.Namespace) -> None: if not files: fail(f"nothing to publish: no image files under {root}") ref = f"refs/remotes/{args.remote}/{args.branch}" - subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True) + subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True, check=False) parent = git("rev-parse", "--verify", "--quiet", ref, check=False) or None with tempfile.TemporaryDirectory() as tmp: env = {**os.environ, "GIT_INDEX_FILE": str(Path(tmp) / "index")} @@ -243,7 +243,7 @@ def publish(args: argparse.Namespace) -> None: def evidence_section(body: str) -> str | None: - match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.M | re.S) + match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.MULTILINE | re.DOTALL) return match.group(1) if match else None diff --git a/plugins/factory/scripts/architecture-check.py b/plugins/factory/scripts/architecture-check.py index 3f2417a..495cf2d 100755 --- a/plugins/factory/scripts/architecture-check.py +++ b/plugins/factory/scripts/architecture-check.py @@ -20,7 +20,7 @@ from pathlib import Path SECTIONS = ["Components", "Flows", "Boundaries", "Cross-cutting", "Entry points"] -HEADER = re.compile(r"^##\s+(.+?)\s*$", re.M) +HEADER = re.compile(r"^##\s+(.+?)\s*$", re.MULTILINE) BACKTICK = re.compile(r"`([^`\n]+)`") PATH_LIKE = re.compile(r"^[^\s]+(/[^\s]*|\.[A-Za-z0-9]{1,6})$") TABLE_ROW = re.compile(r"^\|(.+)\|\s*$") @@ -90,9 +90,9 @@ def main() -> int: text = map_path.read_text(encoding="utf-8", errors="replace") violations: list[str] = [] - if not re.search(r"^Purpose:", text, re.M): + if not re.search(r"^Purpose:", text, re.MULTILINE): violations.append("section: no 'Purpose:' line") - if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.M): + if not re.search(r"^Captured:\s*\d{4}-\d{2}-\d{2}", text, re.MULTILINE): violations.append("section: no dated 'Captured:' line") found = sections(text) order = [name for name in found if name in SECTIONS] diff --git a/plugins/factory/scripts/skill-metrics.py b/plugins/factory/scripts/skill-metrics.py index d9d2b7b..ac78ef0 100755 --- a/plugins/factory/scripts/skill-metrics.py +++ b/plugins/factory/scripts/skill-metrics.py @@ -148,7 +148,8 @@ def numstat(ref: str | None) -> tuple[int, int, set[str]]: for path in untracked: files.add(path) try: - added += sum(1 for _ in Path(path).open(encoding="utf-8", errors="ignore")) + with Path(path).open(encoding="utf-8", errors="ignore") as handle: + added += sum(1 for _ in handle) except OSError: pass return added, removed, files @@ -295,8 +296,10 @@ def cmd_end(skill: str, counts: dict[str, str]) -> int: change = (total(all_tokens) - med_tokens) / med_tokens * 100 if med_tokens else 0 rows.append(( f"vs previous {skill} runs", - f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " - f"this run {change:+.0f}% tokens", + ( + f"{len(previous)} on record, median {human(med_tokens)} tokens in {duration(med_secs)}; " + f"this run {change:+.0f}% tokens" + ), )) width = max(len(name) for name, _ in rows) diff --git a/plugins/factory/skills/scope/scripts/lint-spec.py b/plugins/factory/skills/scope/scripts/lint-spec.py index 2056f78..c45fdde 100755 --- a/plugins/factory/skills/scope/scripts/lint-spec.py +++ b/plugins/factory/skills/scope/scripts/lint-spec.py @@ -21,7 +21,7 @@ LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") -CHOSEN, REJECTED, OPEN, NOT_DOING = "✓", "✗", "?", "⊘" +CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: diff --git a/plugins/factory/skills/ship/scripts/aggregate-findings.py b/plugins/factory/skills/ship/scripts/aggregate-findings.py index 81d86d9..32238d2 100755 --- a/plugins/factory/skills/ship/scripts/aggregate-findings.py +++ b/plugins/factory/skills/ship/scripts/aggregate-findings.py @@ -18,7 +18,7 @@ import sys from pathlib import Path -FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S) +FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.DOTALL) RESULT_SUFFIXES = (".json", ".out") diff --git a/plugins/factory/skills/ship/scripts/pr-evidence.py b/plugins/factory/skills/ship/scripts/pr-evidence.py index 63a448f..a913e8d 100755 --- a/plugins/factory/skills/ship/scripts/pr-evidence.py +++ b/plugins/factory/skills/ship/scripts/pr-evidence.py @@ -36,16 +36,16 @@ from datetime import datetime, timezone from pathlib import Path -DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.S) +DATA_BLOCK = re.compile(r"E2E_DATA_START.*?const\s+E2E_DATA\s*=\s*(\{.*?\})\s*;\s*/\*\s*E2E_DATA_END", re.DOTALL) IDENT = re.compile(r"[A-Za-z_$][A-Za-z0-9_$]*") -PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.I) +PLACEHOLDERS = re.compile(r"\bTODO\b|\bTBD\b|\{[a-z][a-z0-9-]*\}|<[^>]*placeholder[^>]*>|screenshot here|data:image/", re.IGNORECASE) IMAGE = re.compile(r"!\[[^\]]*\]\((https?://[^)\s]+)\)") PAIR_LABELS = { "before": "before", "after": "after", "on merge base": "base", "on the merge base": "base", "merge base": "base", "on this branch": "branch", "on the branch": "branch", "this branch": "branch", } -LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.I) +LABEL_LINE = re.compile(r"^\s*\*\*([^*]+?)\*\*", re.IGNORECASE) FENCE_OPEN = re.compile(r"^\s*```") IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".webp"} @@ -138,7 +138,7 @@ def node_json(literal: str) -> str | None: handle.write(literal) script = f"process.stdout.write(JSON.stringify(eval('(' + require('fs').readFileSync({handle.name!r}, 'utf8') + ')')))" try: - result = subprocess.run(["node", "-e", script], capture_output=True, text=True) + result = subprocess.run(["node", "-e", script], capture_output=True, text=True, check=False) return result.stdout if result.returncode == 0 else None except OSError: return None @@ -147,7 +147,7 @@ def node_json(literal: str) -> str | None: def git(*args: str, env: dict | None = None, check: bool = True) -> str: - result = subprocess.run(["git", *args], capture_output=True, text=True, env=env) + result = subprocess.run(["git", *args], capture_output=True, text=True, env=env, check=False) if check and result.returncode != 0: fail(f"git {' '.join(args)} failed: {result.stderr.strip()}") return result.stdout.strip() @@ -188,7 +188,7 @@ def extract(args: argparse.Namespace) -> None: lines = ["## Evidence", "", f"Captured from the e2e run of `{plan}` at {data.get('generatedAt', 'unknown time')}: " - f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] + + f"{summary.get('passed', 0)}/{summary.get('total', len(scenarios))} scenarios passed.", ""] written: list[str] = [] for index, scenario in enumerate(scenarios, 1): scenario_id = slug(str(scenario.get("id") or scenario.get("title") or index), f"scenario-{index}") @@ -226,7 +226,7 @@ def publish(args: argparse.Namespace) -> None: if not files: fail(f"nothing to publish: no image files under {root}") ref = f"refs/remotes/{args.remote}/{args.branch}" - subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True) + subprocess.run(["git", "fetch", args.remote, args.branch], capture_output=True, text=True, check=False) parent = git("rev-parse", "--verify", "--quiet", ref, check=False) or None with tempfile.TemporaryDirectory() as tmp: env = {**os.environ, "GIT_INDEX_FILE": str(Path(tmp) / "index")} @@ -243,7 +243,7 @@ def publish(args: argparse.Namespace) -> None: def evidence_section(body: str) -> str | None: - match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.M | re.S) + match = re.search(r"^##\s+Evidence\s*$(.*?)(?=^##\s|\Z)", body, re.MULTILINE | re.DOTALL) return match.group(1) if match else None From a960f1899bafa606a57c6b2271f0844a1a6bffb8 Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 16:44:36 +0200 Subject: [PATCH 16/30] feat(harden): measure coverage of subprocess-invoked scripts The gauntlet's coverage-weighted complexity check was scoring every script at 0% coverage, which collapses each score to complexity squared and fails functions that are in fact well covered. The cause was the measurement, not the code: this repository's tests drive their subject scripts through subprocess.run([sys.executable, script, ...]), and a plain `coverage run -m unittest` traces only the parent process. coverage_run.sh wires coverage's documented subprocess support: a sitecustomize.py on PYTHONPATH calls coverage.process_startup() in every child, COVERAGE_PROCESS_START names the config, and both data_file and source are absolute so a child running in a temp cwd still writes to one place and still matches the sources. run-state.py measures 99.5% with it, against 0% before, and the README's gauntlet recipe now uses it. --- tools/harden/README.md | 14 ++++++++----- tools/harden/coverage_run.sh | 38 ++++++++++++++++++++++++++++++++++++ 2 files changed, 47 insertions(+), 5 deletions(-) create mode 100755 tools/harden/coverage_run.sh diff --git a/tools/harden/README.md b/tools/harden/README.md index 20b7d46..8a3832c 100644 --- a/tools/harden/README.md +++ b/tools/harden/README.md @@ -5,6 +5,7 @@ Repo-fitted pieces the `ship` gauntlet reuses on every run; the analyzers themse - `scope.sh [--python]` - the files the branch added or modified against the default branch, minus the standard exclusions and the generated `plugins/` tree. - `added_lines.py` - filters `path:line: ...` findings on stdin to the lines the branch added, so pre-existing findings are counted, not fixed. - `complexity_coverage.py RADON_JSON COVERAGE_JSON [--prefix pkg/]` - coverage-weighted complexity (complexity squared scaled down by line coverage) for functions the branch added; the line is 10 per D-complexity-threshold in `docs/decisions.md`. +- `coverage_run.sh OUT_JSON -- ` - coverage for a suite whose subject scripts run as subprocesses; a plain `coverage run` traces only the parent, so a script the tests invoke through `subprocess.run` reads 0% and its complexity score collapses to complexity squared. - `flaky.py --runs 5 MODULE ...` - runs unittest modules repeatedly in shuffled order and names any test whose outcome disagrees between runs. - `mutate.py FILE FUNC ... --tests MODULE ...` - function-scoped mutation testing: mutates named functions one change at a time and runs their fast tests. - `vulture_whitelist.py` - the committed list of names a framework calls rather than repository code, such as `BaseHTTPRequestHandler` overrides; vulture reads it as an ordinary input file, so pass it alongside the scoped files and record deliberate API surface there instead of adding per-line ignores. @@ -21,14 +22,17 @@ uvx vulture --min-confidence 60 $files tools/harden/vulture_whitelist.py | npx --yes jscpd@4 --min-tokens 50 --format python scripts # clones ``` -Coverage-weighted complexity needs a package with a test suite, so run it from -that package's directory and pass its repository-relative prefix: +Coverage-weighted complexity joins a radon report to a coverage report, so it +needs the suite run first: ```bash -cd pkg && uvx coverage run --source=. -m unittest discover -s tests -t . \ - && uvx coverage json -o /tmp/coverage.json && uvx radon cc -j . > /tmp/radon.json && cd .. \ - && python3 tools/harden/complexity_coverage.py /tmp/radon.json /tmp/coverage.json --prefix pkg/ +tools/harden/coverage_run.sh /tmp/coverage.json -- discover -s factory/evals/tests -t . +uvx radon cc -j factory scripts > /tmp/radon.json +python3 tools/harden/complexity_coverage.py /tmp/radon.json /tmp/coverage.json --threshold 10 ``` +Use `coverage_run.sh` rather than a bare `coverage run`: this repository's tests +drive their subject scripts as subprocesses, which the parent process never sees. + Dependency rules are skipped until the repository has a `docs/dependencies.md`. The repository pins no Python dependencies, so there is no manifest to audit. diff --git a/tools/harden/coverage_run.sh b/tools/harden/coverage_run.sh new file mode 100755 index 0000000..2f3c857 --- /dev/null +++ b/tools/harden/coverage_run.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# Coverage for a suite whose subject scripts run as subprocesses. +# +# A plain `coverage run -m unittest` traces only the parent process, so a script the tests invoke +# through `subprocess.run([sys.executable, script, ...])` reads 0% and its coverage-weighted +# complexity score collapses to complexity squared - a false failure, not a finding. +# +# This wires coverage's documented subprocess support: a sitecustomize.py on PYTHONPATH calls +# coverage.process_startup() in every child, COVERAGE_PROCESS_START names the config, and both +# data_file and source are absolute so a child running in a temp cwd still writes to one place +# and still matches the sources. +# +# Usage: tools/harden/coverage_run.sh OUT_JSON -- +# tools/harden/coverage_run.sh /tmp/coverage.json -- discover -s factory/evals/tests -t . +set -euo pipefail +cd "$(git rev-parse --show-toplevel)" +root=$(pwd) +out=$1; shift +[ "${1:-}" = "--" ] && shift + +work=$(mktemp -d) +trap 'rm -rf "$work"' EXIT +printf 'import coverage\ncoverage.process_startup()\n' >"$work/sitecustomize.py" +cat >"$work/coveragerc" </dev/null +uvx --from coverage coverage json --rcfile="$work/coveragerc" -o "$out" >/dev/null +echo "coverage json written to $out" From 072544950690103e7966664a5168455a0f046def Mon Sep 17 00:00:00 2001 From: tobrun Date: Sun, 20 Sep 2026 16:44:46 +0200 Subject: [PATCH 17/30] perf(factory): run the gate scenarios once each, concurrently test_factory_checks took 210.7s; it now takes 30.8s, and the full suite drops from about 215s to 35.3s. The cost was never the repo copy or the Codex rebuild, which measure 0.05s and 0.07s. It was that every test ran the whole gate, and 4.6s of each 7.4s run was the copy's own check_factory_script re-running the other 60 unit tests, which say nothing about F02 or F03. Twenty-seven gate runs, three of them the same clean run, one test running validate twice over identical state, and ten more run one after another inside a single subtest. Each distinct scenario is now named once in SCENARIOS and they run concurrently, capped at the core count. Every assertion is preserved: disabling check_factory_unattended in a copy still fails 19 of 20 tests. The scenarios that only assert about a clean gate share one run, and no scenario executes twice. The futures live at module scope rather than in a class fixture because tools/harden/flaky.py shuffles tests into one flat suite, and unittest re-runs setUpClass every time execution crosses a class boundary. A class fixture re-launched the whole fan-out many times per pass, which made a 5-run sweep take 20m51s; at module scope it is once per process. The two tests that never needed a gate run moved to their own classes. --- factory/evals/tests/test_factory_checks.py | 393 ++++++++++++++------- 1 file changed, 259 insertions(+), 134 deletions(-) diff --git a/factory/evals/tests/test_factory_checks.py b/factory/evals/tests/test_factory_checks.py index 2037c67..c47cfe0 100644 --- a/factory/evals/tests/test_factory_checks.py +++ b/factory/evals/tests/test_factory_checks.py @@ -6,6 +6,16 @@ check under test without a rebuild), then runs the copy's own scripts/validate.sh - never the real repo's. +Running that gate costs about 7 seconds, and most of it is the copy's own +check_factory_script re-running the other 60 unit tests, which say nothing +about F02 or F03. The scenarios are independent processes in separate temp +directories, so SCENARIOS below names each distinct one once and setUpClass +runs them all concurrently; every test then asserts against its own finished +run. Two savings come for free: the scenarios that differ only in what they +assert about one clean run now share that single run, and no scenario is +executed twice. Ordering is not a factor - nothing is shared between runs but +the read-only source tree. + This file is excluded by name from check_factory_script's own discovery (D-nested-validate): it runs validate.sh itself, so without the exclusion the gate would re-enter itself once per nesting level and never terminate. @@ -19,13 +29,20 @@ import subprocess import sys import tempfile +import threading import unittest +from collections.abc import Callable +from concurrent.futures import Future, ThreadPoolExecutor +from dataclasses import dataclass from pathlib import Path REPO_ROOT = Path(__file__).resolve().parents[3] IGNORE = shutil.ignore_patterns( ".git", "__pycache__", ".pytest_cache", ".ruff_cache", "node_modules", ".DS_Store" ) +# The gate is subprocess-bound, so threads are enough; the cap keeps a laptop +# from running more full validate.sh runs at once than it has cores to spend. +WORKERS = min(8, os.cpu_count() or 4) def copy_repo() -> Path: @@ -58,187 +75,291 @@ def run_validate(copy_root: Path) -> subprocess.CompletedProcess: ) -class FactoryChecksTest(unittest.TestCase): - def setUp(self) -> None: - self.copy_root = copy_repo() +@dataclass(frozen=True) +class Outcome: + """One finished validate.sh run, detached from its (deleted) temp copy.""" + + returncode: int + stdout: str + stderr: str + target: str | None + script_log: str | None + + @property + def output(self) -> str: + return self.stdout + self.stderr - def tearDown(self) -> None: - shutil.rmtree(self.copy_root.parent, ignore_errors=True) - def append_to(self, relative: str, text: str) -> Path: - target = self.copy_root / relative - with target.open("a", encoding="utf-8") as handle: +def append(relative: str, text: str) -> Callable[[Path], str]: + """A mutation that appends `text` to `relative` and names it as the target.""" + + def mutate(copy_root: Path) -> str: + with (copy_root / relative).open("a", encoding="utf-8") as handle: handle.write(text) - return target + return relative - def assert_rebuilt_copy_fails(self, tag: str, target: Path) -> None: - """Rebuild the mutated copy, run its validate.sh, expect `tag` at `target`.""" - rebuild(self.copy_root) - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 1) - self.assertIn(tag, result.stdout) - self.assertIn(str(target.relative_to(self.copy_root)), result.stdout) - self.assertNotIn("[C01]", result.stdout) + return mutate + + +def drop_one_close_marker(copy_root: Path) -> str: + relative = "factory/skills/run/SKILL.md" + target = copy_root / relative + text = target.read_text(encoding="utf-8") + if "" not in text: + raise AssertionError("expected a close marker in run/SKILL.md") + target.write_text( + text.replace("", "", 1) + + "\n\nAsk the user which one to pick.\n", + encoding="utf-8", + ) + return relative + + +def rename_result_literal(copy_root: Path) -> str: + relative = "factory/skills/build/SKILL.md" + target = copy_root / relative + text = target.read_text(encoding="utf-8") + target.write_text( + text.replace("results/{phase}-{attempt}.json", "some-other-path.json"), + encoding="utf-8", + ) + return relative + + +def break_f03_self_test(copy_root: Path) -> str: + relative = "scripts/validate.sh" + target = copy_root / relative + text = target.read_text(encoding="utf-8") + broken = text.replace( + r'ASK = (r"ask|prompt|poll', + r'ASK = (r"askXXX|promptXXX|pollXXX', + ) + if text == broken: + raise AssertionError("expected the F03 pattern text to be present") + target.write_text(broken, encoding="utf-8") + return relative + + +PERSON_ROUTING_SHAPES = ( + "Surface it to the person and wait for the answer.", + "Escalate the threshold to the requester.", + "Hand it to the owner for a decision.", + "Ask someone on the team which one to pick.", + "Raising the gate is a human decision.", + "Get approval before opening the PR.", + "Check with them before proceeding.", + "In the end it is their call.", + "Seek sign-off from the maintainer.", + "This seam is worth agreeing on with the user.", +) + +BUILD_SKILL = "factory/skills/build/SKILL.md" +SHIP_SKILL = "factory/skills/ship/SKILL.md" +RUN_SKILL = "factory/skills/run/SKILL.md" + +# name -> (mutation or None, rebuild the Codex copies before validating) +SCENARIOS: dict[str, tuple[Callable[[Path], str] | None, bool]] = { + # One clean run, shared by every scenario that only asserts about a clean gate. + "clean": (None, False), + "person_routed": (append(BUILD_SKILL, "\n\nAsk the user which one to pick.\n"), True), + "human_calls": ( + append(SHIP_SKILL, "\n\nRotation and history rewriting are both human calls.\n"), + True, + ), + "question_left": ( + append(BUILD_SKILL, "\n\nThen any question left for the user - a blocked gate.\n"), + True, + ), + "prose_about_people": ( + append( + BUILD_SKILL, + "\n\nA user-facing change in the user's repository still needs an e2e scenario.\n", + ), + True, + ), + "marker_in_run_skill": ( + append( + RUN_SKILL, + "\n\n\nAsk the user which one to pick.\n" + "\n", + ), + True, + ), + "marker_in_phase_copy": ( + append( + BUILD_SKILL, + "\n\n\nAsk the user which one to pick.\n" + "\n", + ), + True, + ), + "marker_unclosed": ( + append(RUN_SKILL, "\n\n\nAsk the user which one to pick.\n"), + True, + ), + "marker_close_dropped": (drop_one_close_marker, True), + "marker_stray_close": (append(RUN_SKILL, "\n\n\n"), True), + "marker_shares_line": ( + append(BUILD_SKILL, "\n\n Ask the user which one to pick.\n"), + True, + ), + "broken_self_test": (break_f03_self_test, False), + "result_literal_renamed": (rename_result_literal, True), + "phase_invocation": (append(BUILD_SKILL, "\n\nOn success, invoke $factory:ship.\n"), True), + **{ + f"shape_{index}": (append(BUILD_SKILL, f"\n\n{shape}\n"), True) + for index, shape in enumerate(PERSON_ROUTING_SHAPES) + }, +} + + +def run_scenario(name: str) -> Outcome: + mutate, needs_rebuild = SCENARIOS[name] + copy_root = copy_repo() + try: + target = mutate(copy_root) if mutate else None + if needs_rebuild: + rebuild(copy_root) + result = run_validate(copy_root) + log = copy_root / "validate-logs" / "factory-script-tests.log" + return Outcome( + returncode=result.returncode, + stdout=result.stdout, + stderr=result.stderr, + target=target, + script_log=log.read_text(encoding="utf-8") if log.is_file() else None, + ) + finally: + shutil.rmtree(copy_root.parent, ignore_errors=True) + + +_LOCK = threading.Lock() +_FUTURES: dict[str, Future] = {} + + +def scenario_outcome(name: str) -> Outcome: + """The finished run for `name`, starting every scenario on first call. + + The futures live at module scope rather than in a class fixture on purpose. + A shuffling runner (tools/harden/flaky.py) interleaves these tests with + other classes, and unittest re-runs setUpClass every time execution crosses + a class boundary - so a class fixture would re-launch the whole fan-out + many times per pass. Module scope makes it once per process, whatever order + the tests run in. + """ + with _LOCK: + if not _FUTURES: + pool = ThreadPoolExecutor(max_workers=WORKERS, thread_name_prefix="gate") + _FUTURES.update({name: pool.submit(run_scenario, name) for name in SCENARIOS}) + # Threads are released once every scenario is done; the pool is not + # shut down here because the futures outlive this call. + pool.shutdown(wait=False) + return _FUTURES[name].result() + + +class FactoryChecksTest(unittest.TestCase): + """The scenarios that run the full gate, each asserted against its own run.""" + + def outcome(self, name: str) -> Outcome: + return scenario_outcome(name) + + def assert_gate_fails(self, name: str, tag: str) -> Outcome: + """The gate failed with `tag` at the scenario's own target, not on staleness.""" + outcome = self.outcome(name) + self.assertEqual(outcome.returncode, 1, outcome.output) + self.assertIn(tag, outcome.stdout) + self.assertIsNotNone(outcome.target) + self.assertIn(outcome.target, outcome.stdout) + self.assertNotIn("[C01]", outcome.stdout) + return outcome def test_unmutated_copy_passes_and_excludes_the_harness(self) -> None: - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) - log = self.copy_root / "validate-logs" / "factory-script-tests.log" - if log.is_file(): - contents = log.read_text(encoding="utf-8") - self.assertIn("test_run_state", contents) - self.assertNotIn("test_factory_checks", contents) + outcome = self.outcome("clean") + self.assertEqual(outcome.returncode, 0, outcome.output) + if outcome.script_log is not None: + self.assertIn("test_run_state", outcome.script_log) + self.assertNotIn("test_factory_checks", outcome.script_log) def test_decision_routed_to_a_person_fails_f03(self) -> None: - target = self.append_to( - "factory/skills/build/SKILL.md", "\n\nAsk the user which one to pick.\n" - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("person_routed", "[F03]") def test_human_call_wording_fails_f03(self) -> None: """A phrasing the first F03 pattern missed: the plural defeated `human call`.""" - target = self.append_to( - "factory/skills/ship/SKILL.md", - "\n\nRotation and history rewriting are both human calls.\n", - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("human_calls", "[F03]") def test_question_left_for_the_user_fails_f03(self) -> None: - target = self.append_to( - "factory/skills/build/SKILL.md", - "\n\nThen any question left for the user - a blocked gate.\n", - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("question_left", "[F03]") def test_prose_about_people_does_not_fire_f03(self) -> None: """The widened pattern must not fire on possessives or compounds.""" - self.append_to( - "factory/skills/build/SKILL.md", - "\n\nA user-facing change in the user's repository still needs an e2e scenario.\n", - ) - rebuild(self.copy_root) - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + outcome = self.outcome("prose_about_people") + self.assertEqual(outcome.returncode, 0, outcome.output) def test_same_line_inside_interactive_only_passes(self) -> None: """run/SKILL.md is the one file whose pre-go lines may reach a person.""" - self.append_to( - "factory/skills/run/SKILL.md", - "\n\n\nAsk the user which one to pick.\n\n", - ) - rebuild(self.copy_root) - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + outcome = self.outcome("marker_in_run_skill") + self.assertEqual(outcome.returncode, 0, outcome.output) def test_interactive_only_marker_in_a_phase_copy_fails_f03(self) -> None: """A phase copy has nothing to route, so it may not use the escape hatch.""" - target = self.append_to( - "factory/skills/build/SKILL.md", - "\n\n\nAsk the user which one to pick.\n\n", - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("marker_in_phase_copy", "[F03]") def test_unclosed_interactive_only_block_fails_f03(self) -> None: """An unclosed marker used to silence every line after it, to end of file.""" - target = self.append_to( - "factory/skills/run/SKILL.md", - "\n\n\nAsk the user which one to pick.\n", - ) - self.assert_rebuilt_copy_fails("[F03]", target) - result = run_validate(self.copy_root) - self.assertIn("never closed", result.stdout) + outcome = self.assert_gate_fails("marker_unclosed", "[F03]") + self.assertIn("never closed", outcome.stdout) def test_deleting_a_close_marker_fails_f03(self) -> None: """Dropping one close marker leaves the rest of run/SKILL.md unscanned.""" - target = self.copy_root / "factory" / "skills" / "run" / "SKILL.md" - text = target.read_text(encoding="utf-8") - self.assertIn("", text) - target.write_text( - text.replace("", "", 1) + "\n\nAsk the user which one to pick.\n", - encoding="utf-8", - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("marker_close_dropped", "[F03]") def test_stray_close_marker_fails_f03(self) -> None: - target = self.append_to( - "factory/skills/run/SKILL.md", "\n\n\n" - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("marker_stray_close", "[F03]") def test_marker_sharing_its_line_does_not_silence_the_line(self) -> None: """Only a marker alone on its line opens a block; anything else is prose.""" - target = self.append_to( - "factory/skills/build/SKILL.md", - "\n\n Ask the user which one to pick.\n", - ) - self.assert_rebuilt_copy_fails("[F03]", target) + self.assert_gate_fails("marker_shares_line", "[F03]") def test_newly_covered_person_routing_shapes_fail_f03(self) -> None: """Shapes the round-1 pattern missed: person, requester, owner, someone.""" - shapes = ( - "Surface it to the person and wait for the answer.", - "Escalate the threshold to the requester.", - "Hand it to the owner for a decision.", - "Ask someone on the team which one to pick.", - "Raising the gate is a human decision.", - "Get approval before opening the PR.", - "Check with them before proceeding.", - "In the end it is their call.", - "Seek sign-off from the maintainer.", - "This seam is worth agreeing on with the user.", - ) - for shape in shapes: + for index, shape in enumerate(PERSON_ROUTING_SHAPES): with self.subTest(shape=shape): - copy_root = copy_repo() - try: - target = copy_root / "factory" / "skills" / "build" / "SKILL.md" - with target.open("a", encoding="utf-8") as handle: - handle.write(f"\n\n{shape}\n") - rebuild(copy_root) - result = run_validate(copy_root) - self.assertEqual(result.returncode, 1, result.stdout) - self.assertIn("[F03]", result.stdout) - self.assertIn("factory/skills/build/SKILL.md", result.stdout) - finally: - shutil.rmtree(copy_root.parent, ignore_errors=True) + outcome = self.outcome(f"shape_{index}") + self.assertEqual(outcome.returncode, 1, outcome.stdout) + self.assertIn("[F03]", outcome.stdout) + self.assertIn(BUILD_SKILL, outcome.stdout) def test_broken_self_test_regex_fails_f03(self) -> None: - validate_sh = self.copy_root / "scripts" / "validate.sh" - text = validate_sh.read_text(encoding="utf-8") - # Break the F03 pattern so it no longer matches its own self-test samples. - broken = text.replace( - r'ASK = (r"ask|prompt|poll', - r'ASK = (r"askXXX|promptXXX|pollXXX', - ) - self.assertNotEqual(text, broken, "expected the F03 pattern text to be present") - validate_sh.write_text(broken, encoding="utf-8") - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 1) - self.assertIn("[F03]", result.stdout) - self.assertIn("pattern no longer matches", result.stdout) + outcome = self.outcome("broken_self_test") + self.assertEqual(outcome.returncode, 1) + self.assertIn("[F03]", outcome.stdout) + self.assertIn("pattern no longer matches", outcome.stdout) def test_missing_result_protocol_literal_fails_f02(self) -> None: - target = self.copy_root / "factory" / "skills" / "build" / "SKILL.md" - text = target.read_text(encoding="utf-8") - text = text.replace("results/{phase}-{attempt}.json", "some-other-path.json") - target.write_text(text, encoding="utf-8") - self.assert_rebuilt_copy_fails("[F02]", target) + self.assert_gate_fails("result_literal_renamed", "[F02]") def test_invoking_another_phase_fails_f02(self) -> None: - target = self.append_to( - "factory/skills/build/SKILL.md", "\n\nOn success, invoke $factory:ship.\n" - ) - self.assert_rebuilt_copy_fails("[F02]", target) + self.assert_gate_fails("phase_invocation", "[F02]") def test_unattended_wording_check_is_clean_over_the_four_phase_copies(self) -> None: - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) - self.assertNotIn("[F03]", result.stdout) + outcome = self.outcome("clean") + self.assertEqual(outcome.returncode, 0, outcome.output) + self.assertNotIn("[F03]", outcome.stdout) def test_protocol_check_is_clean_over_every_phase_skill(self) -> None: - result = run_validate(self.copy_root) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) - self.assertNotIn("[F02]", result.stdout) + outcome = self.outcome("clean") + self.assertEqual(outcome.returncode, 0, outcome.output) + self.assertNotIn("[F02]", outcome.stdout) + + +class RepoToolsTest(unittest.TestCase): + """Checks that read a copy or the repo itself without running the gate.""" + + def setUp(self) -> None: + self.copy_root = copy_repo() + + def tearDown(self) -> None: + shutil.rmtree(self.copy_root.parent, ignore_errors=True) def test_build_codex_plugin_check_reports_up_to_date(self) -> None: rebuild(self.copy_root) @@ -262,6 +383,10 @@ def test_architecture_check_is_clean_on_the_updated_overview(self) -> None: ) self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + +class PhaseCopyDriftTest(unittest.TestCase): + """Reads the real tree only; no copy, no gate.""" + def test_phase_copies_stay_within_five_lines_of_dev(self) -> None: for skill in ("scope", "scope-review", "build", "ship"): dev_lines = len( From fb0d00c632976e654152c93b023062d3978bf97f Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 07:35:30 +0200 Subject: [PATCH 18/30] docs: promote config-driven-factory decisions and contracts Why: scope-review's phase-5 promotion step wrote these during the spec-review run but did not commit them; committing now, before build, so the ledger state build implements against is on disk and reviewable on its own. Adds 37 decisions under "## Config-driven factory" to docs/decisions.md, supersession notes on five factory-plugin entries, and three drafted contracts to docs/contracts.md with "? verify" marks that change set 7 of .dev/config-driven-factory/spec.md resolves. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- docs/contracts.md | 15 +++++ docs/decisions.md | 154 ++++++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 164 insertions(+), 5 deletions(-) diff --git a/docs/contracts.md b/docs/contracts.md index e66bc3f..2dce7f4 100644 --- a/docs/contracts.md +++ b/docs/contracts.md @@ -14,3 +14,18 @@ C-factory-plan-files: plan files under .dev/ are never committed by a factory ru guaranteed by: the run skill's preflight, which runs git check-ignore -q .dev/ in the consuming repository and appends the entry to .gitignore as a recorded repair when it is missing, so the rule has a carrier in any consuming repository and not only in the fixture ? verify: no free scenario reaches the missing-entry branch, because the fixture's setup.sh writes the entry itself (D-verification names this one of the six knowingly unproven behaviors) relied on by: any consuming repository that has not ignored .dev/ itself; git log --name-only on the run branch is the proof the closing report cites (2026-09-20, D-plan-files-ignored) + +## Config-driven factory + +C-factory-phase-contract: the launch prompt requires every phase, ours or foreign, to write .dev/{plan}/results/{phase}-{attempt}.json with schema factory.result/1 as its last action, and never to launch another phase. This imposes both, it does not guarantee them; a phase that writes nothing is read as a failed attempt + guaranteed by: the wrapper in run/references/launch.md, which validate.sh F02 checks for the results path literal, the factory.result/1 literal and the exact sentence "Never name or launch the next phase."; F02 over every built-in phase body for factory-run.json, the results path literal and that same sentence; and run-state.py check-result treating a missing or unparseable file as a failed attempt ? verify: not yet built; resolve when config-driven-factory/spec.md change set 7 lands + relied on by: the judgment loop, run-state.py check-result, and the one-invoker rule (D-invoker-invariant) + (2026-09-20, config-driven-factory/spec.md; supersedes C-factory-result once change set 7 lands) +C-factory-unattended: after the go, no FACTORY-OWNED phase body routes a decision to a person. A phase body the factory did not write is covered by no check: its author declares it unattended_safe in the config, and a phase that blocks on a person is neither prevented nor detected + guaranteed by: the validate.sh scan over factory/phases and factory/skills/run, for factory-owned bodies only ? verify: not yet built; resolve when config-driven-factory/spec.md change set 7 lands + relied on by: the judgment loop, which never waits on a person + (2026-09-20, config-driven-factory/spec.md; supersedes the entry above once change set 7 lands) +C-factory-plan-files: plan files under .dev/ are never committed by a factory run; the run branch carries code, tests, docs/ and, after an explicit inject, the .factory/ tree + guaranteed by: the run skill's preflight, which runs git check-ignore -q .dev/ in the consuming repository and appends the entry to .gitignore as a recorded repair when it is missing ? verify: not yet built; resolve when config-driven-factory/spec.md change set 7 lands + relied on by: any consuming repository that has not ignored .dev/ itself; git log --name-only on the run branch is the proof the closing report cites, and ship reviews .factory/ files in that log as part of the change + (2026-09-20, config-driven-factory/spec.md; supersedes the entry above once change set 7 lands) diff --git a/docs/decisions.md b/docs/decisions.md index f1a6464..ccdb6a2 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -30,7 +30,7 @@ D-completion-authority: Who decides that a phase is complete? (2026-09-20, facto ✓ the orchestrator judges, after reading the phase's result file and re-running the phase's own deterministic checks as evidence - the checks are `lint-spec.py`, the review verdict line, `check-tests.py`, the Validation block commands, `pr-evidence.py check`, `git log`, `gh pr view` (auto-applied at Confidence: 80%) ⚠ a wrong judgment advances bad work, and the orchestrator grades work it guided and repaired itself; every judgment is recorded with the check outputs it read, the copied ship's fresh-context panel is the one independent signal on the code, and `skill-metrics.py` numbers are never evidence ✗ a deterministic gate that overrules the result - the old "gate is authoritative" rule produced the oscillating build attempts and parks on codes nobody could act on (user (2026-09-20)) ✗ a separate judge subagent - fresh context per judgment costs a launch per phase and the judge would read the same files the orchestrator reads -D-phase-skills: Where do the phase instructions come from? (2026-09-20, factory-plugin/spec.md) +D-phase-skills: Where do the phase instructions come from? (2026-09-20, factory-plugin/spec.md) superseded by D-packaging (2026-09-20, config-driven-factory/spec.md), which moves the bodies to factory/phases/ ✓ factory-owned copies of `scope`, `scope-review`, `build`, and `ship` under `factory/skills/`, with their references and scripts copied under `factory/references/` and `factory/scripts/` - the copies are meant to diverge: the long-term goal is to split the monolith skills into smaller factory-only pieces (user (2026-09-20)) ⚠ a fix in `dev` has to be ported by hand; no drift check, because drift is the intent ✗ reuse the dev skills by path plus an unattended preamble - one source of truth, but it blocks the planned decoupling and depends on `dev` being installed next to `factory`, which plugin installs do not guarantee ✗ generated copies rewritten by the build script - the rewrite rules would encode the unattended policy twice, once in the script and once in prose @@ -38,7 +38,7 @@ D-phase-transport: How does a phase run? (2026-09-20, factory-plugin/spec.md) ✓ one fresh-context subagent per phase attempt through the host's plain subagent tool (the Agent tool on Claude Code, `spawn_agent` on Codex, `task` on opencode) - a fresh context per attempt keeps four phases out of one window, following the transport ladder in ship's `orchestration.md`; the subagent's result is a file, never its final message (auto-applied at Confidence: 85%) ✗ the orchestrator runs the phase inline - one context for four phases loses the middle of everything, and scope-review and ship each spawn panels of their own ✗ a subprocess per phase (`codex exec`, `claude -p`) - per-host argv again; the removed runner's `hosts.py` was exactly this -D-human-touchpoints: Where is the human? (2026-09-20, factory-plugin/spec.md) +D-human-touchpoints: Where is the human? (2026-09-20, factory-plugin/spec.md) superseded by D-go-placement (2026-09-20, config-driven-factory/spec.md), which puts the go after a contiguous interactive prefix that may be empty ✓ scope runs inline with the human in the orchestrator's session and ends with one explicit go - scope is an interview and the request names it as the only human phase; from then on no phase asks anyone anything, and a run that is out of options ends with a written report instead of a question (auto-applied at Confidence: 80%) ⚠ the scope interview lives in the orchestrator's context; the state file lets the user clear the session and invoke `/factory:run` again to continue with a fresh context ✗ scope as a subagent too - subagents have no user-input tool on Claude Code (docs, sub-agents: AskUserQuestion is removed from subagents), and scope is an interview ✗ a second human gate before the pull request - the request names scope as the only human phase @@ -50,7 +50,7 @@ D-plan-files-ignored: How does "plan files are never committed" hold in a consum ✓ the `run` skill's preflight runs `git check-ignore -q .dev/` in the consuming repository and, when it is not ignored, the orchestrator appends the entry to `.gitignore` itself and records it as a repair - the check sits before the handoff next to the transport check, while a person is still at the keyboard, so scope's first file is already ignored when it is written; the write is an ordinary repair under D-orchestrator-hands, committed on the run branch and recorded in `factory-run.json` with rationale and evidence, and bound by the two hard stops like any other (user (2026-09-20)) ⚠ the accepted cost: the factory edits a file the user never asked it to touch, as one of its first actions, before any phase has run ⚠ the edit is made at preflight but only committed once the run branch exists at handoff, so the default branch never carries it, and a run abandoned before the handoff leaves the line in the working tree ✗ a preflight that ends the run with the one-line fix instead of writing it - parking on a fault a person fixes in one command is the pattern D-orchestrator-hands exists to remove, and this one would park before any work has been done at all ✗ leaving it as a documented Input and a README prerequisite - nothing checks a prerequisite, so the first consuming repository without the entry commits the plan files silently; this repository is itself such a repository today, and only the fixture's `setup.sh` writes an entry -D-failure-ladder: What happens after a failed or stopped phase? (2026-09-20, factory-plugin/spec.md) +D-failure-ladder: What happens after a failed or stopped phase? (2026-09-20, factory-plugin/spec.md) attempt counts superseded by D-attempt-budget (2026-09-20, config-driven-factory/spec.md), which seeds and reports a budget but still never refuses ✓ cheapest sufficient step, recorded each time - fix it myself and re-run the checks; relaunch the phase in a fresh subagent with guidance naming what the previous attempt did and what is required instead; end the run with a report when three attempts of one phase have not produced a `done` the orchestrator accepts; at most two repairs per attempt, after which the next fix is a relaunch and counts as an attempt (auto-applied at Confidence: 80%) ⚠ three and two are guesses; reopen when a real run needs one more that would have succeeded ⚠ a repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches; the copied ship's remediation loop is built for that ⚠ three and two are prose the orchestrator follows, not counts it is held to: `run-state.py record` tallies every attempt and repair and `show` prints them, so the numbers are on disk, but nothing refuses a fourth attempt, and the refusal that would catch a miscount is the deterministic gate D-completion-authority rejects ✗ unlimited attempts - the old run with ten build attempts burned 18 hours on an oscillation ✗ one attempt then stop - most faults are fixable on the second try with guidance @@ -88,13 +88,13 @@ D-pi: Ship the factory to Pi? (2026-09-20, factory-plugin/spec.md) D-invoker-invariant: The repo says skills never invoke each other; does the orchestrator break it? (2026-09-20, factory-plugin/spec.md) ✓ the orchestrator is the one sanctioned invoker - it launches phase skills by path, and phase skills still never invoke each other or the next phase; `docs/architecture.md` and `CLAUDE.md` state the exception (auto-applied at Confidence: 85%) ✗ keep the invariant literal and make the human launch each phase - that is the dev workflow, not a factory -D-unattended-wording-check: How is "no human after scope" enforced in the copies? (2026-09-20, factory-plugin/spec.md) +D-unattended-wording-check: How is "no human after scope" enforced in the copies? (2026-09-20, factory-plugin/spec.md) scan root superseded by D-unattended-proof (2026-09-20, config-driven-factory/spec.md), which narrows the guarantee to factory-owned bodies ✓ a validate.sh check (revived old F03) - it fails any line under `factory/skills/{scope-review,build,ship,run}` (SKILL.md and their references) and `factory/references/factory-run.md` that routes a decision to a person, with `` blocks exempt; the regex self-tests against four sample phrases and a failed self-test fails validation, as the removed F03 already did - P-deterministic-guards-over-prose (auto-applied at Confidence: 85%) ⚠ the other copied references (`ci-parity.md`, `contracts.md`, `jira.md`) keep their dev wording about people because phases read them for notation, not for who decides ✗ prose in the preamble only - a rule in prose softens as the context grows (CLAUDE.md); the check's exit code does not D-protocol-check: How is the result protocol kept present in every phase copy? (2026-09-20, factory-plugin/spec.md) ✓ a validate.sh check (revived old F02) requires each phase SKILL.md to name `factory-run.json` and the result path pattern `results/{phase}-{attempt}.json`, and the `run` SKILL.md to name every phase - a copy that drifts off the protocol fails validation ✗ trust the copies - a dropped line would only show up in a paid run -D-handoff-seal: Is the approved scope protected after the human leaves? (2026-09-20, factory-plugin/spec.md) +D-handoff-seal: Is the approved scope protected after the human leaves? (2026-09-20, factory-plugin/spec.md) superseded by D-handoff-artifact (2026-09-20, config-driven-factory/spec.md), which seals what the types' seals axes name ✓ recorded, not sealed - at the go, the orchestrator stores the spec's sha256, every `⊘` line, and every `tests:` scenario text per change set (the plan format has no scenario ids; `check-tests.py` counts scenarios per change set) in the run state; before each judgment it diffs them against the current spec and treats a dropped or reworded approved scenario as evidence for its decision (auto-applied at Confidence: 75%) ✗ the old sealed intent with a gate - a reworded scenario parked the run even when the rewording was right ✗ nothing - later phases could quietly drop scenarios the human approved @@ -149,3 +149,147 @@ D-tests-location: Where do the run-state script's unit tests live? (2026-09-20, D-skill-length: How do the copies stay under the repo's SKILL.md length rule? (2026-09-20, factory-plugin/spec.md) ✓ within five lines of the dev source - the Factory context section is short and points at `factory-run.md`, and each rewritten human line replaces its dev line instead of adding one; `scope` is already 160 lines, so the rule of roughly 150 is met the same way dev meets it ✗ a full Factory context block per skill as before - the old copies grew past the rule and R00 only fails at 200 + +## Config-driven factory + +D-pipeline-shape: How much of the pipeline does the config control? (2026-09-20, config-driven-factory/spec.md) + ✓ an ordered list of typed phases, any length - a team with three stages or six can express it, and the orchestrator still knows what kind of thing each phase is (user, 2026-09-20) + ✗ fixed four slots with pluggable implementations - smallest change, but the engine is still our shape and a different method cannot be expressed + ✗ arbitrary phases with no type - the orchestrator could only check exit codes, which is the deterministic-gate runner that was removed for parking runs (D-remove-factory in the ledger) +D-type-home: Where does a phase's type live? (2026-09-20, config-driven-factory/spec.md) + ✓ in the config entry, not the skill - a team's existing skills are driven unmodified, which is the whole point of not imposing on them (user, 2026-09-20) + ✗ in SKILL.md frontmatter - self-describing, but a third-party skill must add our field before the factory can drive it, which is the imposition being removed + ✗ frontmatter default with config override - both audiences work, at the cost of two sources of truth for one fact and a precedence rule to test +D-type-set: Is the set of types closed? (2026-09-20, config-driven-factory/spec.md) + ✓ open - types are declared in the config with their own contracts, so no vocabulary tax and no fifth-kind-of-phase problem (user, 2026-09-20) ⚠ the orchestrator has no built-in knowledge of a type it has never seen, so for a type declaring only prose the judgment is an LLM reading prose a config author wrote, which is weaker than the evidence the built-in types carry + ✗ closed set of four - knowable and testable, but any pipeline that does not fit pushes people back to writing skills shaped like ours + ✗ interactive and unattended only - smallest vocabulary, but then type carries almost nothing and the config restates everything per phase +D-type-axes: What exactly may a type declare, given the set is open? (2026-09-20, config-driven-factory/spec.md) + ✓ five axes and no more - `interactive`, `requires` (the outcome contract in prose), `checks` (commands that verify it), `seals` (the artifact drift is watched on), and `attempts` (the budget) - each is something the orchestrator cannot infer for a type it has never seen (auto-applied at Confidence: 78%) ⚠ a type declaring no `checks` is judged only on its result envelope and its `requires` prose + ✗ four axes as the first draft had it - it omitted `seals` and `attempts`, which two later decisions then added informally, leaving the stated axis list wrong + ✗ include `transport` as an axis - that is per-phase harness routing, which D-cross-harness defers, so an axis for it would violate that non-goal in the same document +D-checks-precedence: `checks` may be declared on a type and on a phase, so which wins? (2026-09-20, config-driven-factory/spec.md) + ✓ the phase's `checks` replace the type's entirely when present, never merge - a phase can drop a type check that does not apply to it, and the type still supplies the default so most phases declare nothing (auto-applied at Confidence: 80%) ⚠ two homes for one fact, the cost D-type-home rejected an option over, accepted here because a type with no default checks forces every phase to restate them + ✗ merge type and phase checks - more expressive, and a phase could never drop a type check that does not apply to it + ✗ checks only on the phase - one home, and every phase in a pipeline repeats its type's checks verbatim +D-checks-safety: A check is a command the unattended orchestrator runs. What may it be? (2026-09-20, config-driven-factory/spec.md) + ✓ an argv list, never a shell string, whose executable resolves under the repository root or the plugin root, with `${plugin_root}`, `${repo_root}`, `${plan_dir}`, `${phase}` and `${attempt}` as the only placeholders, substituted when the check is run and left literal by `show --resolved` - a check executes rather than instructs, so it needs at least the allowlist a skill path gets; that allowlist partitions today's criteria: a criterion whose executable is a repository or plugin script (`lint-spec.py`, `check-tests.py`, `pr-evidence.py check`) becomes a `checks` entry, and one that needs an external tool (`gh pr view`, `git ls-remote`, the `grep -cx 'Verdict: APPROVED'` line) stays `requires` prose the orchestrator judges by running the tool itself, as it does today (auto-applied at Confidence: 82%) ⚠ a project whose verification needs a shell pipeline has to wrap it in a committed script first, and a criterion left in `requires` is verified by the model rather than by a command + ✗ any shell string - matches what a Makefile target looks like, and hands arbitrary shell to an agent holding commit and push authority, from a file D-config-drift lets change mid-run + ✗ no constraint, trust the config author - the config is committed and reviewed, and a run resumed after someone edits it is not re-reviewed +D-done-criteria: How does the orchestrator know a phase is done? (2026-09-20, config-driven-factory/spec.md) + ✓ the type declares the outcome contract, the phase declares how to check it against its own artifacts - judgment stays independent of the phase's self-report while artifacts stay the team's (user, 2026-09-20) + ✗ entirely from the type - our types' criteria name `check-tests.py`, a Validation block and change-plan commit numbering, so a foreign skill would have to emit our artifacts + ✗ always declared per phase, type only controls launch - nothing implicit, but the orchestrator has less to reason with and every config repeats itself + ✗ delegate to the phase's own claim - collapses the separation the orchestrator exists for, per `judgment.md` +D-repair-foreign: May the orchestrator repair a phase whose skill it did not write? (2026-09-20, config-driven-factory/spec.md) + ✓ no; a foreign phase's failure is relaunched or ends the run, never hand-repaired - the orchestrator never infers a phase's meaning, and a hand repair of an artifact it did not define is exactly that inference; repair stays available to built-in and injected phases, whose bodies are ours verbatim (D-phase-classes) (auto-applied at Confidence: 76%) ⚠ the cheapest failures, the two-command environment faults this whole design exists to absorb, stop being absorbed for custom pipelines + ✗ repair foreign phases too - keeps the failure ladder uniform, and repairing an artifact whose meaning the orchestrator cannot know is exactly the inference the never-infers invariant forbids +D-no-config: What happens in a repo with no config file? (2026-09-20, config-driven-factory/spec.md) + ✓ the built-in default pipeline, our phases run from the installed plugin, nothing written to the repo - keeps every current user working with no config and no file written (auto-applied at Confidence: 80%) ⚠ the default resolves to absolute paths inside the installed plugin, which the state file records per attempt, so a plugin upgrade or a resume on another machine can leave a state file naming paths that no longer exist + ✗ warn and stop until `init` is run - nothing implicit, but every current user hits a stop on upgrade + ✗ run the defaults and materialize the config - reproducible afterwards, and writes to the user's repo uninvited +D-missing-skill: What happens when a configured phase names a skill that cannot be resolved? (2026-09-20, config-driven-factory/spec.md) + ✓ warn and stop before the interview, naming the phases and the command that installs the defaults - silently substituting our method for theirs is the exact failure this change exists to prevent (user, 2026-09-20) ⚠ this also stops a resume, so a config edited into an unresolvable state strands an in-flight run until the path is fixed + ✗ fall back to our default skill for that phase - keeps runs alive, and silently substitutes our method for theirs +D-foreign-phases: How does the orchestrator drive a skill it did not write? (2026-09-20, config-driven-factory/spec.md) + ✓ the launch prompt wraps any skill and imposes the result contract, and the phase's declared checks verify the outcome independently - no third-party change needed, and `launch.md` already does most of it (user, 2026-09-20) ⚠ the wrapper is prose, so the contract must say it imposes the shape rather than guarantees it, and a skill with strong opinions about its own output can defeat it + ✗ require a conformance declaration and refuse undeclared skills - explicit about the risk, and every third-party skill then needs a config author to vouch for an output shape the wrapper can impose anyway; the declaration D-hung-phase does require is about a different property, one no wrapper can impose + ✗ an adapter shim per phase - handles skills that cannot be talked into a format, at the cost of another moving part per phase + ✗ accept it unverified with nothing declared - least work, and nobody has stated the skill is safe to run unattended, which is the gap D-hung-phase closes rather than this entry +D-hung-phase: A foreign phase that asks a person blocks forever. What is the backstop? (2026-09-20, config-driven-factory/spec.md) + ✓ there is none; the run waits, and the config author declares each foreign phase (D-phase-classes) `unattended_safe` before the orchestrator will launch it - nothing can detect or stop a phase blocked on a person, so the word of the one person who has read the skill is the only evidence there is, and the declaration is that word with no mechanism behind it (user, 2026-09-20) ⚠ the declaration is a promise rather than a proof, and a hung run is neither prevented nor detected, so a person notices it ⚠ this incurs the cost D-foreign-phases cited when it rejected a conformance declaration, a config author vouching per foreign skill; it is accepted here and not there because the result shape can be imposed by the wrapper and verified by `checks`, while whether a skill will block on a person can be neither imposed nor detected + ✗ a per-phase timeout - the first draft chose this and it cannot be built: the orchestrator is an LLM, the host's subagent call blocks, and the Agent tool, `spawn_agent` and `task` take no timeout, so there is no turn in which to notice and no process to hold a watchdog + ✗ rely on host-level limits - honest about where control lives, and the behavior differs per host and we control none of it + ✗ a supervising process that can kill a phase - the only real detection, and it reintroduces the runner outside the host that D-orchestrator-form rejected and that was deleted for parking runs + ✗ tell the subagent to stop itself after N minutes - free, and useless for the case that matters, since a skill blocked on a question is not reading its instructions +D-phase-classes: The entries above say "default", "factory-owned" and "foreign". Which phase is which? (2026-09-20, config-driven-factory/spec.md) + ✓ three classes by path and provenance: built-in (a skill under the plugin root), injected (a copy under `.factory/skills/` whose sha256 matches its `.inject.json` entry), foreign (anything else); `unattended_safe` is required for foreign phases only, an injected copy whose hash no longer matches its manifest is treated as foreign, and "factory-owned" in the contracts means built-in - each class follows from facts `factory-config.py check` can establish, and an edited copy stops being ours the moment its hash moves, so no edit can inherit the exemption or the repair rung (scope-review round 1, 2026-09-20) ⚠ one keystroke in an injected copy demands a declaration and forfeits repair, which is the posture the edit deserves and will still surprise a team fixing a typo + ✗ a config key naming the class - a user can relabel anything as built-in, so the exemption from `unattended_safe` and the repair rung would rest on a label rather than on what the file is + ✗ two classes, ours and not ours, with injected copies always ours - an edited copy is scanned by no check, so C-factory-unattended would claim a guarantee over a file nothing has read +D-unattended-proof: What happens to C-factory-unattended when phases are foreign? (2026-09-20, config-driven-factory/spec.md) + ✓ the static check keeps proving it for factory-owned phase bodies, and the contract is reworded to claim exactly that and no more - it claims only what the check actually delivers (auto-applied at Confidence: 80%) ⚠ this supersedes D-unattended-wording-check's scan root, and for a foreign phase the property rests on the author's declaration with no detection behind it + ✗ leave the contract as worded - it would claim a guarantee the check no longer delivers for custom pipelines + ✗ drop the contract - loses the property for the default pipeline too, which is where it is provable and valuable +D-packaging: How do the default phase skills stop enlarging the customer's skill set? (2026-09-20, config-driven-factory/spec.md) + ✓ phase bodies move to `factory/phases/`, leaving `run` as the plugin's only invocable skill, and the orchestrator reads a phase by path - the customer's skill list is the constraint the user named (user, 2026-09-20) ⚠ this supersedes D-phase-skills, and the blast radius is wider than a scan-root change: `skill_ui`, the F02 and F03 scan roots, the drift test, `factory-run.md`'s two hardcoded `factory/skills/{phase}` paths, `run/SKILL.md`'s sibling path to scope, two links in `factory/evals/README.md`, and the loss of R00's length gate and R09 to R14 over the four bodies, including the `disable-model-invocation` rule and the generator's frontmatter strip, which only reaches files under `skills/` + ✗ keep them as skills and simply not copy them - smallest change, and the customer still sees five skills + ✗ ship a second plugin holding only the phases - lets a team opt in, at the cost of a third plugin to version and install +D-phase-body-checks: What replaces R00 and R09 to R14 over the moved phase bodies? (2026-09-20, config-driven-factory/spec.md) + ✓ F02 and F03 extend to cover length, name and the invocation-policy rule for `factory/phases/`, so no rule is lost in the move - the rules reached the bodies only through their location under `skills/`, and a move is not a reason to lose them (auto-applied at Confidence: 84%) ⚠ two checks grow responsibilities that R00 and R09 to R14 held generically for anything under `skills/` + ✗ let the rules lapse for phase bodies - the move would silently drop the repo's own `disable-model-invocation` rule over four files +D-inject: What does injection physically do? (2026-09-20, config-driven-factory/spec.md) + ✓ copy the named phase bodies plus the references and scripts they link into `.factory/skills/` in the consuming repository, on explicit command only - a committed copy is editable and reviewable by the whole team (user, 2026-09-20) ⚠ the copies land in the run branch's diff, so ship's gauntlet and review panel review them as if they were the change, and build's commits-match-the-change-plan check sees an inject commit it cannot account for + ✗ symlink - no diff and always current, and breaks when the plugin moves or the repo is cloned elsewhere + ✗ copy into a git-ignored scratch directory - keeps the PR clean, and then the pipeline is not reproducible from the repo and a teammate cannot edit it +D-inject-config: `inject` in a repo with no config has nothing to point at. What does it write? (2026-09-20, config-driven-factory/spec.md) + ✓ the copies and a `.factory/config.yaml` naming them - asking to inject is consent to both, and a copy that no config names never runs, since a repo with no config runs the built-in defaults (auto-applied at Confidence: 81%) ⚠ one command writes two things, so the invariant about what may be written has to name the config file too + ✗ refuse until `init` has run - one file per command, and it makes the first useful thing a two-step ritual + ✗ copies only, leaving the config absent - the injected copies would never be used, since a repo with no config runs the built-in defaults +D-inject-provenance: How does an injected copy survive a plugin upgrade? (2026-09-20, config-driven-factory/spec.md) + ✓ a provenance header as a frontmatter key in each copied SKILL.md, plus a manifest `.factory/.inject.json` (at the `.factory/` root, since the copied references and scripts land beside `.factory/skills/`) recording the source plugin, its version and a sha256 per copied file, with `inject` refusing to overwrite a copy whose current hash differs from its manifest entry - a copy that rots against an upgrade with nothing detecting it is worse than a refusal, and only a recorded hash tells a local edit from a plugin upgrade (auto-applied at Confidence: 77%) ⚠ a user who edits a copy must pass `--force` to take an upgrade, and merging is theirs to do; an unedited copy takes the upgrade without it ⚠ no header is written into `.py`, `.html` or any other non-Markdown file, so for those the manifest is the only provenance, and a copy moved out from under it is unrecognisable + ✗ overwrite silently on every inject - always current, and destroys local edits without telling anyone + ✗ no provenance at all - the copies rot against plugin updates with nothing able to detect it +D-config-location: Where does the config live? (2026-09-20, config-driven-factory/spec.md) + ✓ `.factory/config.yaml` in the consuming repository - committable and reviewable, which the team needs (user, 2026-09-20) + ✗ under `.dev/` - the run's own preflight git-ignores that tree, so the config could never be shared with a team + ✗ a key inside the existing `.dev/config.json` - one fewer file, same git-ignore problem +D-config-parsing: `factory-config.py` must read and write this file, and the repo has no YAML parser. What does it use? (2026-09-20, config-driven-factory/spec.md) + ✓ a strict reader for this schema's shallow subset - block mappings, block sequences including nested ones, plain scalars, double-quoted scalars with the usual escapes, and comments, nothing else; anchors, flow syntax and multiline scalars stay out - which doubles as validation, since anything outside the subset is rejected rather than interpreted; quoted scalars and nested sequences are not optional, because `requires` prose carries `: ` (the scope-review verdict is `Verdict: APPROVED`) and each `checks` entry is itself an argv list (auto-applied at Confidence: 73%) ⚠ the first draft claimed no parser was needed, which was wrong the moment the CLI had to `check` and `set`; this is a second YAML implementation in the repository and it will reject files a general parser accepts + ✗ add PyYAML - one correct parser, and this repository's first Python dependency + ✗ shell out to `yq` from the CLI - reuses what `validate.sh` needs anyway, and makes an external tool a hard requirement in every consuming repository + ✗ the orchestrator reads it and no CLI parses anything - true only while the CLI cannot validate or edit, which `init`, `set` and `check` all require +D-config-checking: Who validates the file, given the orchestrator also reads it? (2026-09-20, config-driven-factory/spec.md) + ✓ `factory-config.py check` is the single validator, and the preflight runs it and reads its output rather than judging the file itself - one verdict on validity, so the checker and the orchestrator can never disagree about whether the file is acceptable (auto-applied at Confidence: 84%) ⚠ the model still reads the file afterwards to compose launches, so a file the checker accepts can still be misread + ✗ the checker and the orchestrator validate independently - defence in depth, and two readers can disagree about one file with no tie-break +D-terminality: How is a finished run recognised once ship may not be last? (2026-09-20, config-driven-factory/spec.md) + ✓ an `advance` on the last declared phase, with the ordered list seeded into the state file - it is true for any pipeline, including ours (auto-applied at Confidence: 88%) + ✗ keep "an advance on `ship`" - false the moment ship is not last or not present + ✗ an explicit `complete` action - unambiguous, and it extends the closed `ACTIONS` set and every run must remember to record it +D-phase-order-in-state: Where does phase order live at runtime? (2026-09-20, config-driven-factory/spec.md) + ✓ seeded into `factory-run.json` at `init` from the declared list - resume must answer earliest-not-done from the state file alone (auto-applied at Confidence: 85%) + ✗ re-read the config on resume - always current, and a mid-run config edit silently changes which phase is earliest-not-done +D-config-drift: What if the config changes mid-run? (2026-09-20, config-driven-factory/spec.md) + ✓ seal its sha256 at handoff together with the resolved pipeline (ordered ids, types and resolved skill paths), the same moment and posture `diff-spec` already uses for the spec, and report drift as evidence rather than blocking - mid-run mutation is expected and already handled this way, and a hash alone can only say "changed", so the sealed pipeline is what lets `diff-config` name the phase that moved the way `diff-spec` names a change set (auto-applied at Confidence: 76%) ⚠ edits between preflight and the go are inside the sealed value and invisible as drift, and an edit that makes a path unresolvable still stops the run under D-missing-skill, so never-blocks holds only for resolvable edits + ✗ seal at `init` - catches the pre-go window too, and `init` runs before the user has seen the interview, so every legitimate pre-go fix reads as drift + ✗ refuse to resume under a changed config - safest, and blocks a legitimate fix to a broken phase entry + ✗ ignore drift - a phase inserted mid-run changes which phase is earliest-not-done with nothing noticing +D-go-placement: Where is the go in a pipeline that is not ours? (2026-09-20, config-driven-factory/spec.md) + ✓ after the last `interactive` phase, where the interactive phases must form a contiguous prefix of the pipeline that runs inline in the orchestrator's session as scope does today; a pipeline whose prefix is empty has no go, and branches and seals before the first launch instead - the go is the last moment a person is present, so every interactive phase sits before it and every unattended phase sits after it, on the branch and behind the seal the go creates; a subagent has no user-input tool, so the prefix cannot run any other way (auto-applied at Confidence: 74%) ⚠ a fully unattended pipeline starts work with no human confirmation at all, which is a real change in posture from today's one explicit go ⚠ this supersedes the ledger's D-human-touchpoints (scope as the one human phase ending in one explicit go) and D-handoff-seal (the seal at the go), which the ledger must mark when this lands + ✗ after the first interactive phase - simpler to state, and a second interview would then run after the go, which the unattended contract forbids + ✗ a config field naming the phase the go follows - explicit, and it can name a phase that is not interactive, so the rule needs checking anyway + ✗ always require an interactive first phase - keeps today's shape exactly, and forbids the unattended pipeline the config is meant to allow +D-interactive-after-go: May an interactive phase appear after an unattended one? (2026-09-20, config-driven-factory/spec.md) + ✓ no, rejected at preflight by `factory-config.py check` as an interactive phase after an unattended one, which is the check that the prefix D-go-placement requires is contiguous - an interactive phase after the go contradicts the unattended contract and would hang a resumed run with nobody present, and an unattended phase before the go would run with no branch and no seal (auto-applied at Confidence: 78%) + ✗ allow it and drop the contract for that config - more expressive, and the run can then block forever on an absent human +D-skill-addressing: How is a phase's skill named? (2026-09-20, config-driven-factory/spec.md) + ✓ a path, with a `${plugin_root}` placeholder the CLI resolves from its own file location and the orchestrator resolves from the host's injected base directory line - only a path can be checked before the interview is paid for (auto-applied at Confidence: 80%) ⚠ two resolvers for one placeholder, so a config that checks clean from the repository can still resolve differently inside the orchestrator's session + ✗ a host skill name like `$dev:scope` - reads nicely, and cannot be checked until mid-run, after the interview is paid for +D-skill-path-allowlist: May a config name any path on disk? (2026-09-20, config-driven-factory/spec.md) + ✓ no, only under the plugin root or the repository root - a config-named skill is instructions handed to an agent holding commit and push authority (auto-applied at Confidence: 75%) ⚠ a team keeping skills in a shared directory outside the repo has to symlink or copy them in + ✗ any absolute path - maximally flexible, and hands arbitrary instructions to an unattended agent holding commit and push authority +D-handoff-artifact: What does handoff seal when the first phase is not our scope? (2026-09-20, config-driven-factory/spec.md) + ✓ the sealed artifact is the `seals` axis of a type, and `run-state.py handoff` takes the paths it seals from the `seals` axes of the phases that ran before it rather than assuming `.dev/{plan}/spec.md`; when those axes are empty it seals nothing, records that, and `diff-spec` reports "not watched" and exits 0 instead of failing - drift detection that silently no-ops is worse than one that says it is off, and today `handoff` exits 1 and `diff-spec` exits 3 without a spec, which would stop every pipeline that has no spec-like artifact (auto-applied at Confidence: 74%) ⚠ a pipeline whose types seal nothing loses drift detection entirely ⚠ the scenario-text and `⊘` parse applies only to a sealed artifact written in the dev notation; any other sealed artifact is watched by its hash alone + ✗ keep requiring `spec.md` - drift detection keeps working for our pipeline and silently no-ops for every other + ✗ make an empty seal a hard failure - honest, and blocks any pipeline that has no spec-like artifact +D-attempt-budget: Do the retry limits stay global? (2026-09-20, config-driven-factory/spec.md) + ✓ the `attempts` axis carries a per-type default with a per-phase override, plus a total-run ceiling, which `run-state.py init` seeds into the state and `show` reports as exhausted without ever refusing an attempt - a ten-phase pipeline has no ceiling at all today, and a count that refuses would be the deterministic gate above the model's judgment that D-completion-authority and D-verification reject (auto-applied at Confidence: 72%) ⚠ a ceiling set too low turns a recoverable run into an ended one ⚠ this supersedes D-failure-ladder's fixed three attempts, which becomes the built-in types' default beside a built-in ceiling of 12 attempts per run; the two repairs per attempt have no axis and stay prose in `judgment.md` as the ledger has them; the posture that the numbers are prose the orchestrator follows rather than counts it is held to is kept + ✗ keep two repairs and three attempts globally - simple, and unbounded across a long pipeline +D-loop-driver: Is the phase loop code or prose? (2026-09-20, config-driven-factory/spec.md) + ✓ prose in `run/SKILL.md` as today, with `run-state.py` recording rather than driving - the loop is judgment, and a script that owned it would be the runner outside the model that D-orchestrator-form rejected (auto-applied at Confidence: 79%) ⚠ this is the honest limit of the stub fixture: stubs plus `run-state.py` prove the bookkeeping, and the loop itself stays proven only by the paid per-host run + ✗ a driver script that iterates phases - testable end to end without a host, and it is the runner process outside the model that D-orchestrator-form rejected +D-stub-fixture: How is as much as possible tested without paid host runs? (2026-09-20, config-driven-factory/spec.md) + ✓ a stub pipeline whose phases are trivial scripts writing conforming result files, exercising `run-state.py` and the phase bookkeeping - those branches are otherwise reachable only through a paid host run (auto-applied at Confidence: 84%) ⚠ per D-loop-driver the stubs prove the bookkeeping and not the orchestrator's prose loop, and the change plan must not claim otherwise + ✗ only the paid per-host run - closest to reality, and far too slow and expensive to cover the state machine's branches + ✗ treat this as the rejected offline benchmark - D-verification rejected stub hosts because the old bench never ran in real mode and proved nothing about parking; that rejection stands for parking behavior, and this fixture claims only bookkeeping +D-state-versioning: What about state files already on disk? (2026-09-20, config-driven-factory/spec.md) + ✓ add a `schema` field, and read a file without one as the default four-phase pipeline - in-flight runs in consuming repositories keep resuming (auto-applied at Confidence: 79%) + ✗ refuse to resume old files - clean, and strands in-flight runs in consuming repositories +D-phase-timings: Where do the closing report's per-phase timings come from? (2026-09-20, config-driven-factory/spec.md) + ✓ `run-state.py attempt` records a timestamp when it opens and closes an attempt, and at open also the phase's type and resolved skill path, so the state file carries everything the closing report names without asking anything of the skill - the closing report promised per-phase timings, types and paths that nothing recorded, and a failed attempt with no result file has no other carrier for its path (auto-applied at Confidence: 83%) ⚠ wall-clock around a launch includes the orchestrator's own judging time, so the numbers are coarse + ✗ leave the report's timings unsourced as the first draft did - the closing report promised per-phase timings that nothing recorded +D-cross-harness: Does this slice also route phases to other harnesses and models? (2026-09-20, config-driven-factory/spec.md) + ✗ include it - the design note in `.dev/factory-config/` covers it and it needs two ledger decisions reopened + ⊘ not doing - the pipeline must be declarable before it is worth asking what runs each phase; reopen once a custom pipeline has run end to end ? verify: a real run on a custom pipeline +D-foreign-metrics: Do foreign phases report skill metrics? (2026-09-20, config-driven-factory/spec.md) + ✗ require the `skill-metrics.py` call in the launch prompt - imposes our tooling on a foreign skill + ⊘ not doing - the orchestrator records its own per-attempt timestamps under D-phase-timings, so nothing is required of the skill; reopen if a custom pipeline needs per-skill metrics ? verify: the closing report shows per-phase timings on a custom pipeline From f5e43320ed0a33197fad209488346ff957b8cc26 Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 07:56:42 +0200 Subject: [PATCH 19/30] feat(factory): add factory-config.py, the pipeline config reader and CLI Change set 1 of config-driven-factory/spec.md: a strict shallow-YAML reader/writer for .factory/config.yaml, the resolve/check/init/show/set/unset CLI, and factory/references/pipeline-config.md describing the schema. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/evals/tests/test_factory_config.py | 587 +++++++++++ .../evals/tests/test_protocol_and_copies.py | 4 +- factory/references/pipeline-config.md | 88 ++ factory/scripts/factory-config.py | 922 ++++++++++++++++++ plugins/factory/references/pipeline-config.md | 88 ++ plugins/factory/scripts/factory-config.py | 922 ++++++++++++++++++ 6 files changed, 2610 insertions(+), 1 deletion(-) create mode 100644 factory/evals/tests/test_factory_config.py create mode 100644 factory/references/pipeline-config.md create mode 100755 factory/scripts/factory-config.py create mode 100644 plugins/factory/references/pipeline-config.md create mode 100755 plugins/factory/scripts/factory-config.py diff --git a/factory/evals/tests/test_factory_config.py b/factory/evals/tests/test_factory_config.py new file mode 100644 index 0000000..8fbb404 --- /dev/null +++ b/factory/evals/tests/test_factory_config.py @@ -0,0 +1,587 @@ +"""Unit tests for factory-config.py: the config reader, validator and CLI. + +Drives the script as a subprocess in a temporary repository per change set 1's +and change set 2's `tests:` lines in .dev/config-driven-factory/spec.md. +""" + +from __future__ import annotations + +import hashlib +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +SCRIPT = REPO_ROOT / "factory" / "scripts" / "factory-config.py" + + +def run(repo: Path, *args: str, plugin_root: Path | None = None) -> subprocess.CompletedProcess: + cmd = [sys.executable, str(SCRIPT), "--repo-root", str(repo)] + if plugin_root is not None: + cmd += ["--plugin-root", str(plugin_root)] + cmd += list(args) + return subprocess.run(cmd, capture_output=True, text=True) + + +def write_raw(repo: Path, text: str) -> None: + factory = repo / ".factory" + factory.mkdir(parents=True, exist_ok=True) + (factory / "config.yaml").write_text(text) + + +def write_json_doc(repo: Path, doc: dict) -> None: + """Write a config.yaml from a plain dict via the script's own dump_config, + so tests build fixtures the same way the CLI would produce them.""" + sys.path.insert(0, str(SCRIPT.parent)) + import importlib.util + + spec = importlib.util.spec_from_file_location("factory_config_under_test", SCRIPT) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + module.write_doc(repo, doc) + + +def make_skill(repo: Path, rel: str, content: str = "skill body\n") -> Path: + path = repo / rel + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content) + return path + + +class TempRepoTestCase(unittest.TestCase): + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + self.repo = Path(self._tmp.name) + + def tearDown(self) -> None: + self._tmp.cleanup() + + +class CheckValidationTest(TempRepoTestCase): + def test_duplicate_phase_id(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + {"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}, + {"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}, + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("'a'", result.stdout) + + def test_undeclared_type(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {}, + "phases": [{"id": "a", "type": "missing", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("missing", result.stdout) + + def test_empty_phase_list(self) -> None: + write_json_doc(self.repo, {"version": 1, "types": {}, "phases": []}) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("empty phase list", result.stdout) + + def test_check_executable_outside_roots(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + { + "id": "a", + "type": "t", + "skill": "${repo_root}/x/SKILL.md", + "checks": [["${repo_root}/../../etc/passwd"]], + "unattended_safe": True, + } + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("etc/passwd", result.stdout) + + def test_check_as_shell_string(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + { + "id": "a", + "type": "t", + "skill": "${repo_root}/x/SKILL.md", + "checks": ["lint-spec.py ${plan_dir}"], + "unattended_safe": True, + } + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("'a'", result.stdout) + + def test_check_naming_unsupported_placeholder(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + { + "id": "a", + "type": "t", + "skill": "${repo_root}/x/SKILL.md", + "checks": [["${home}/bin/x"]], + "unattended_safe": True, + } + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("${home}", result.stdout) + + def test_check_executable_with_plan_dir(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + { + "id": "a", + "type": "t", + "skill": "${repo_root}/x/SKILL.md", + "checks": [["${plan_dir}/x"]], + "unattended_safe": True, + } + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("${plan_dir}", result.stdout) + + def test_reserved_phase_id(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "defaults", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("reserved", result.stdout) + + def test_unsupported_constructs_never_traceback(self) -> None: + for raw in ( + "version: 1\nphases: &x [1,2]\n", + "version: 1\nphases: {a: 1}\n", + "version: 1\nphases: [1,2]\n", + "version: 1\nphases: |\n x\n", + ): + write_raw(self.repo, raw) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1, raw) + self.assertNotIn("Traceback", result.stderr) + + def test_requires_with_colon_in_double_quotes(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"requires": "Verdict: APPROVED must appear"}}, + "phases": [{"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 0) + result = run(self.repo, "show", "--resolved") + self.assertIn("Verdict: APPROVED", result.stdout) + + def test_nested_checks_both_lists_shown(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"checks": [["lint-spec.py", "${plan_dir}/spec.md"], ["check-tests.py", "${plan_dir}"]]}}, + "phases": [{"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + result = run(self.repo, "show", "--resolved") + self.assertEqual(result.returncode, 0) + self.assertIn("lint-spec.py", result.stdout) + self.assertIn("check-tests.py", result.stdout) + + def test_built_in_phase_without_unattended_safe_ok(self) -> None: + make_skill(self.repo, "plugin/skills/a/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "a", "type": "t", "skill": "${plugin_root}/skills/a/SKILL.md"}], + }, + ) + result = run(self.repo, "check", plugin_root=self.repo / "plugin") + self.assertEqual(result.returncode, 0, result.stdout) + + def test_foreign_phase_without_unattended_safe_fails(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md"}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("'a'", result.stdout) + + def test_injected_copy_hash_mismatch_without_unattended_safe(self) -> None: + path = make_skill(self.repo, ".factory/skills/scope/SKILL.md", "original\n") + digest = hashlib.sha256(path.read_bytes()).hexdigest() + manifest = self.repo / ".factory" / ".inject.json" + manifest.write_text(json.dumps({"plugin": "factory", "version": "1.0.0", "files": {"skills/scope/SKILL.md": digest}})) + path.write_text("edited\n") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "scope", "type": "t", "skill": "${repo_root}/.factory/skills/scope/SKILL.md"}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("skills/scope/SKILL.md", result.stdout) + self.assertIn("'scope'", result.stdout) + + def test_injected_copy_hash_match_without_unattended_safe_ok(self) -> None: + path = make_skill(self.repo, ".factory/skills/scope/SKILL.md", "original\n") + digest = hashlib.sha256(path.read_bytes()).hexdigest() + manifest = self.repo / ".factory" / ".inject.json" + manifest.write_text(json.dumps({"plugin": "factory", "version": "1.0.0", "files": {"skills/scope/SKILL.md": digest}})) + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "scope", "type": "t", "skill": "${repo_root}/.factory/skills/scope/SKILL.md"}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 0, result.stdout) + + def test_unparseable_manifest_without_unattended_safe(self) -> None: + make_skill(self.repo, ".factory/skills/scope/SKILL.md", "original\n") + (self.repo / ".factory" / ".inject.json").write_text("not json") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "scope", "type": "t", "skill": "${repo_root}/.factory/skills/scope/SKILL.md"}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("'scope'", result.stdout) + self.assertIn(".inject.json", result.stdout) + + def test_interactive_after_unattended(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"unatt": {"interactive": False}, "inter": {"interactive": True}}, + "phases": [ + {"id": "a", "type": "unatt", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}, + {"id": "b", "type": "inter", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}, + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("'a'", result.stdout) + self.assertIn("'b'", result.stdout) + + def test_no_interactive_phase_go_none(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"interactive": False}}, + "phases": [{"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + self.assertEqual(run(self.repo, "check").returncode, 0) + result = run(self.repo, "show", "--resolved") + self.assertIn("go: none", result.stdout) + + def test_builtin_default_go_after_scope(self) -> None: + result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") + self.assertIn("go: after scope", result.stdout) + + +class InitTest(TempRepoTestCase): + def test_init_twice_second_fails(self) -> None: + self.assertEqual(run(self.repo, "init").returncode, 0) + result = run(self.repo, "init") + self.assertEqual(result.returncode, 1) + self.assertIn("config.yaml", result.stderr) + + +class SetUnsetTest(TempRepoTestCase): + def _write_phase(self, pid: str = "build") -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": pid, "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + + def test_set_phase_skill_roundtrips(self) -> None: + self._write_phase() + result = run(self.repo, "set", "build.skill", "${repo_root}/y/SKILL.md") + self.assertEqual(result.returncode, 0, result.stderr) + result = run(self.repo, "show") + text = (self.repo / ".factory" / "config.yaml").read_text() + self.assertIn("${repo_root}/y/SKILL.md", text) + self.assertIn("id: build", text) + self.assertIn("type: t", text) + + def test_set_defaults_attempts_reflected_in_show_resolved(self) -> None: + self._write_phase() + run(self.repo, "set", "defaults.attempts", "5") + result = run(self.repo, "show", "--resolved") + self.assertIn("attempts: 5", result.stdout) + + def test_set_types_checks_with_items(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {}, + "phases": [{"id": "a", "type": "review", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + run(self.repo, "set", "types.review.checks", "--item", "lint-spec.py", "--item", "${plan_dir}/spec.md") + result = run(self.repo, "show", "--resolved") + self.assertEqual(result.returncode, 0, result.stdout) + self.assertIn("lint-spec.py", result.stdout) + self.assertIn("${plan_dir}/spec.md", result.stdout) + + def test_unset_last_key_drops_phase(self) -> None: + self._write_phase() + run(self.repo, "unset", "build.unattended_safe") + run(self.repo, "unset", "build.skill") + run(self.repo, "unset", "build.type") + run(self.repo, "unset", "build.id") + text = (self.repo / ".factory" / "config.yaml").read_text() + self.assertNotIn("build", text) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("empty phase list", result.stdout) + + def test_set_on_missing_phase_id(self) -> None: + self._write_phase(pid="build") + result = run(self.repo, "set", "nosuch.skill", "x") + self.assertEqual(result.returncode, 1) + self.assertIn("nosuch", result.stderr) + + def test_set_malformed_path(self) -> None: + result = run(self.repo, "set", "build") + self.assertEqual(result.returncode, 3) + self.assertIn("build", result.stderr) + + def test_set_twice_byte_identical(self) -> None: + self._write_phase() + run(self.repo, "set", "defaults.attempts", "5") + before = (self.repo / ".factory" / "config.yaml").read_bytes() + run(self.repo, "set", "defaults.attempts", "5") + after = (self.repo / ".factory" / "config.yaml").read_bytes() + self.assertEqual(before, after) + + def test_set_on_hand_written_file_reparses_to_same_pipeline(self) -> None: + write_raw( + self.repo, + "version: 1\n" + "types:\n" + " t:\n" + " interactive: false\n" + "phases:\n" + " - id: build\n" + " type: t\n" + " skill: ${repo_root}/x/SKILL.md\n" + " unattended_safe: true\n", + ) + make_skill(self.repo, "x/SKILL.md") + before = run(self.repo, "show", "--resolved", "--json") + run(self.repo, "set", "defaults.ceiling", "9") + after_text = (self.repo / ".factory" / "config.yaml").read_text() + self.assertTrue(after_text.startswith("# .factory/config.yaml\n")) + after = run(self.repo, "show", "--resolved", "--json") + self.assertEqual( + [p["id"] for p in json.loads(before.stdout)["phases"]], + [p["id"] for p in json.loads(after.stdout)["phases"]], + ) + + +class ResolutionTest(TempRepoTestCase): + def test_no_config_file_shows_four_default_phases(self) -> None: + result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") + self.assertEqual(result.returncode, 0, result.stdout) + self.assertIn("built-in default", result.stdout) + for pid in ("scope", "scope-review", "build", "ship"): + self.assertIn(pid, result.stdout) + + def test_config_with_two_phases_shows_exactly_those_in_order(self) -> None: + make_skill(self.repo, "x/SKILL.md") + make_skill(self.repo, "y/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + {"id": "first", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}, + {"id": "second", "type": "t", "skill": "${repo_root}/y/SKILL.md", "unattended_safe": True}, + ], + }, + ) + result = run(self.repo, "show", "--resolved", "--json") + ids = [p["id"] for p in json.loads(result.stdout)["phases"]] + self.assertEqual(ids, ["first", "second"]) + + def test_plugin_root_placeholder_resolves_under_installed_plugin(self) -> None: + plugin = self.repo / "plugin" + make_skill(plugin, "skills/a/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [{"id": "a", "type": "t", "skill": "${plugin_root}/skills/a/SKILL.md"}], + }, + ) + result = run(self.repo, "show", "--resolved", "--json", plugin_root=plugin) + self.assertEqual(result.returncode, 0, result.stdout) + skill = json.loads(result.stdout)["phases"][0]["skill"] + self.assertTrue(skill.startswith(str(plugin))) + + def test_relative_path_escaping_repo_root(self) -> None: + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + {"id": "a", "type": "t", "skill": "${repo_root}/../outside/SKILL.md", "unattended_safe": True} + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("outside", result.stdout) + + def test_phase_omitting_checks_inherits_type_checks(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"checks": [["lint-spec.py", "${plan_dir}/spec.md"]]}}, + "phases": [{"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + result = run(self.repo, "show", "--resolved") + self.assertIn("lint-spec.py", result.stdout) + + def test_phase_checks_replace_type_checks(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"checks": [["lint-spec.py", "${plan_dir}/spec.md"]]}}, + "phases": [ + { + "id": "a", + "type": "t", + "skill": "${repo_root}/x/SKILL.md", + "checks": [["check-tests.py", "${plan_dir}"]], + "unattended_safe": True, + } + ], + }, + ) + result = run(self.repo, "show", "--resolved") + self.assertNotIn("lint-spec.py", result.stdout) + self.assertIn("check-tests.py", result.stdout) + + def test_builtin_pipeline_exactly_three_checks_and_requires_prose(self) -> None: + result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") + self.assertEqual(result.returncode, 0, result.stdout) + self.assertEqual(result.stdout.count("check:"), 3) + self.assertIn("gh pr view", result.stdout) + self.assertIn("git ls-remote", result.stdout) + self.assertIn("Verdict: APPROVED", result.stdout) + + def test_check_argv_with_plan_dir_left_literal(self) -> None: + result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") + self.assertIn("${plan_dir}", result.stdout) + + def test_builtin_scope_seals_plan_dir_spec_and_ceiling_twelve(self) -> None: + result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") + self.assertIn("seal: ${plan_dir}/spec.md", result.stdout) + self.assertIn("ceiling: 12", result.stdout) + + def test_show_resolved_json_shape(self) -> None: + result = run(self.repo, "show", "--resolved", "--json", plugin_root=REPO_ROOT / "factory") + data = json.loads(result.stdout) + self.assertIn("phases", data) + for phase in data["phases"]: + self.assertEqual(set(phase.keys()), {"id", "type", "skill", "interactive"}) + self.assertEqual([p["id"] for p in data["phases"]], ["scope", "scope-review", "build", "ship"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/factory/evals/tests/test_protocol_and_copies.py b/factory/evals/tests/test_protocol_and_copies.py index f9d89a1..6ae61e5 100644 --- a/factory/evals/tests/test_protocol_and_copies.py +++ b/factory/evals/tests/test_protocol_and_copies.py @@ -43,7 +43,9 @@ def test_copied_tree_holds_dev_references_and_factory_run(self) -> None: self.assertIn("factory-run.md", factory_reference_names) factory_script_names = {p.name for p in (FACTORY / "scripts").glob("*.py")} - self.assertEqual(factory_script_names, {"skill-metrics.py", "architecture-check.py"}) + self.assertEqual( + factory_script_names, {"skill-metrics.py", "architecture-check.py", "factory-config.py"} + ) def test_backticked_reference_and_script_paths_resolve_under_factory(self) -> None: problems = [] diff --git a/factory/references/pipeline-config.md b/factory/references/pipeline-config.md new file mode 100644 index 0000000..d72662b --- /dev/null +++ b/factory/references/pipeline-config.md @@ -0,0 +1,88 @@ +# Pipeline config + +`.factory/config.yaml` in the consuming repository declares the pipeline the orchestrator drives. +Its absence is not an error: the built-in default pipeline applies, our four phases run from the installed plugin, and nothing is written to the repo. + +## Schema + +```yaml +version: 1 +defaults: + ceiling: 12 + attempts: 3 +types: + : + interactive: true + requires: "prose the orchestrator judges by running a tool itself" + checks: + - - lint-spec.py + - ${plan_dir}/spec.md + seals: + - ${plan_dir}/spec.md + attempts: 3 +phases: + - id: + type: + skill: ${plugin_root}/phases//SKILL.md + checks: + - - check-tests.py + - ${plan_dir} + unattended_safe: true +``` + +`version` is always `1`. `defaults`, `types` and `phases` are the schema's only other top-level keys; an unknown key is a `check` finding. + +### Types: exactly five axes, no more + +- `interactive` (bool) - runs inline in the orchestrator's own session rather than launched. +- `requires` (prose) - the outcome contract the orchestrator judges when the type's `checks` don't cover it, the same way `judgment.md` reads a phase today. +- `checks` (a list of argv lists) - commands that verify the outcome. Each entry is an argv list, never a shell string. +- `seals` (a list of paths) - artifacts whose drift `run-state.py handoff`/`diff-spec` watches. May use `${plan_dir}`. +- `attempts` (int) - the per-type default retry budget. + +The type set is open: any name declared under `types` is usable by a phase. A type the orchestrator has never seen is judged by its declared axes alone. + +### Phases: an ordered list + +Each entry is `{id, type, skill, checks, unattended_safe}`. `id`, `type` and `skill` are required; `checks` and `unattended_safe` are optional per-phase overrides. + +- `checks`, when present on a phase, **replaces** the type's `checks` entirely - it never merges. +- `unattended_safe: true` is required for a foreign phase (see Phase classes below); built-in and injected phases don't need it. + +`defaults` and `types` are reserved words: no phase may take either as its `id`, since the key-path grammar below roots at both. + +### Placeholders + +A `skill` path and a check's argv elements may use `${plugin_root}`, `${repo_root}`, `${plan_dir}`, `${phase}` and `${attempt}`. A check's **executable** element (argv[0]) may only use `${plugin_root}` or `${repo_root}` - `${plan_dir}` and the others can't be allowlisted before a run exists. `show --resolved` leaves every placeholder literal; the orchestrator substitutes them when it launches or runs a check. + +### Phase classes + +Three classes, by path and provenance, established by `factory-config.py check`: + +- **built-in** - the skill path starts with `${plugin_root}`. +- **injected** - the skill sits under `.factory/skills/` and its current sha256 matches the entry `.factory/.inject.json` recorded for it. +- **foreign** - anything else, including an injected copy whose hash has drifted from its manifest entry. + +Only a foreign phase needs `unattended_safe: true`; the orchestrator never hand-repairs a foreign phase's failure, only relaunches it or ends the run. + +## The key-path grammar + +`set` and `unset` address one value at a time by a dotted path: + +- `.` - a key on an existing phase entry. The phase must already exist; `set` never creates a phase implicitly. +- `defaults.` - a key under `defaults`, created if absent. +- `types..` - an axis of a type, created if absent. + +A list-valued key takes repeated `--item` flags in place of a positional value. `unset` removes the named key; a phase left with no keys at all is dropped from `phases` entirely. + +## The manifest `check` reads + +`.factory/.inject.json`, written by `inject`, at the shape: + +```json +{"plugin": "", "version": "", "files": {"": ""}} +``` + +## The built-in default pipeline + +Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check` (ship) become `checks` entries; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. diff --git a/factory/scripts/factory-config.py b/factory/scripts/factory-config.py new file mode 100755 index 0000000..98bdc78 --- /dev/null +++ b/factory/scripts/factory-config.py @@ -0,0 +1,922 @@ +#!/usr/bin/env python3 +"""Read, write, and validate .factory/config.yaml: the declared pipeline. + +Usage: + factory-config.py init + factory-config.py show [--resolved] [--json] + factory-config.py check + factory-config.py set [--item ...] + factory-config.py unset + +Exit codes: + init 0 written; 1 the file already exists + show 0 always (there is nothing to fail on a read) + check 0 valid; 1 one or more findings, one per line; 3 a bad call + set 0 recorded; 1 the phase id in the path does not exist; + 3 a malformed path or an unsupported document construct + unset 0 recorded (a no-op if the key was already absent); + 3 a malformed path + +3 always means the call itself could not be carried out - a malformed key +path, an unparseable document, or a usage error - never a finding about the +pipeline's own content. `check` is the single validator: every rule below is +enforced there and nowhere else, so the orchestrator's preflight only ever +relays `check`'s output rather than judging the file itself. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +from pathlib import Path +from typing import NoReturn + +BAD_CALL = 3 +CONFIG_DIR = ".factory" +CONFIG_NAME = "config.yaml" +MANIFEST_NAME = ".inject.json" +HEADER = ( + "# .factory/config.yaml\n" + "# Managed by factory-config.py. Edit with `factory-config.py set`.\n" +) +PLACEHOLDERS = ("${plugin_root}", "${repo_root}", "${plan_dir}", "${phase}", "${attempt}") +CHECK_PLACEHOLDERS = ("${plugin_root}", "${repo_root}", "${plan_dir}", "${phase}", "${attempt}") +EXECUTABLE_PLACEHOLDERS = ("${plugin_root}", "${repo_root}") +RESERVED_PHASE_IDS = {"defaults", "types"} + +# --------------------------------------------------------------------------- +# The built-in default pipeline: today's four phases, equal to what +# judgment.md and the copied phase skills already enforce. Every path uses +# ${plugin_root} so a plugin move only ever touches this one constant. +# --------------------------------------------------------------------------- +BUILTIN_TYPES = { + "interview": { + "interactive": True, + "requires": "the go was recorded (the interview happened inline in this session, " + "so the orchestrator already knows)", + "checks": [["lint-spec.py", "${plan_dir}/spec.md"]], + "seals": ["${plan_dir}/spec.md"], + "attempts": 3, + }, + "review": { + "interactive": False, + "requires": 'spec-review_N.md contains a line whose whole text is exactly ' + '"Verdict: APPROVED" - nothing after it - with a "Rounds:" line', + "attempts": 3, + }, + "implement": { + "interactive": False, + "requires": "the spec's Validation block is green, and commits since handoff " + "match the change plan's numbering", + "checks": [["check-tests.py", "${plan_dir}"]], + "attempts": 3, + }, + "ship": { + "interactive": False, + "requires": "review_N.md carries a verdict for HEAD and gh pr view shows the PR " + "open with head equal to HEAD, or (without a GitHub remote) git ls-remote origin " + "shows the branch at HEAD and pr.md is written", + "checks": [["pr-evidence.py", "check"]], + "attempts": 3, + }, +} +BUILTIN_PHASES = [ + {"id": "scope", "type": "interview", "skill": "${plugin_root}/skills/scope/SKILL.md"}, + {"id": "scope-review", "type": "review", "skill": "${plugin_root}/skills/scope-review/SKILL.md"}, + {"id": "build", "type": "implement", "skill": "${plugin_root}/skills/build/SKILL.md"}, + {"id": "ship", "type": "ship", "skill": "${plugin_root}/skills/ship/SKILL.md"}, +] +BUILTIN_DEFAULTS = {"ceiling": 12} +DEFAULT_ATTEMPTS = 3 + +# --------------------------------------------------------------------------- +# A strict subset of YAML: block mappings, block sequences (including nested +# argv-list-of-lists for `checks`), plain and double-quoted scalars, and +# comments. Flow syntax, anchors, multiline block scalars, and single-quoted +# strings are rejected with ConfigError rather than silently mishandled. +# --------------------------------------------------------------------------- + + +class ConfigError(Exception): + """An unsupported construct or a malformed document. Never a traceback.""" + + +def _strip_comment(text: str) -> str: + in_quotes = False + escaped = False + for i, ch in enumerate(text): + if escaped: + escaped = False + continue + if ch == "\\" and in_quotes: + escaped = True + continue + if ch == '"': + in_quotes = not in_quotes + continue + if ch == "#" and not in_quotes and (i == 0 or text[i - 1].isspace()): + return text[:i] + return text + + +def _raw_lines(text: str) -> list: + out = [] + for raw in text.splitlines(): + leading = raw[: len(raw) - len(raw.lstrip(" \t"))] + if "\t" in leading: + raise ConfigError("a tab in leading indentation") + stripped = _strip_comment(raw).rstrip() + if not stripped.strip(): + continue + indent = len(stripped) - len(stripped.lstrip(" ")) + if indent % 2 != 0: + raise ConfigError(f"odd indentation ({indent} spaces): {stripped.strip()!r}") + out.append((indent, stripped.strip())) + return out + + +def _unfold(indent: int, content: str) -> list: + if not (content == "-" or content.startswith("- ")): + return [(indent, content, False)] + out = [] + while True: + rest = "" if content == "-" else content[2:] + if rest == "": + out.append((indent, None, True)) + return out + if rest == "-" or rest.startswith("- "): + out.append((indent, None, True)) + indent += 2 + content = rest + continue + out.append((indent, rest, True)) + return out + + +def _tokenize(text: str) -> list: + entries = [] + for indent, content in _raw_lines(text): + entries.extend(_unfold(indent, content)) + return entries + + +def _parse_scalar(text: str): + text = text.strip() + if text == "": + return "" + if text == "[]": + return [] + if text == "{}": + return {} + if text.startswith("["): + raise ConfigError(f"flow sequence: {text!r}") + if text.startswith("{"): + raise ConfigError(f"flow mapping: {text!r}") + if text in ("|", ">") or (len(text) == 2 and text[0] in "|>" and text[1] in "+-"): + raise ConfigError(f"multiline scalar block: {text!r}") + if text[0] in "&*": + raise ConfigError(f"anchor or alias: {text!r}") + if text.startswith('"'): + if not text.endswith('"') or len(text) < 2: + raise ConfigError(f"unterminated double-quoted scalar: {text!r}") + body = text[1:-1] + out = [] + i = 0 + while i < len(body): + ch = body[i] + if ch == "\\" and i + 1 < len(body): + nxt = body[i + 1] + out.append({"n": "\n", "t": "\t", '"': '"', "\\": "\\"}.get(nxt, nxt)) + i += 2 + continue + out.append(ch) + i += 1 + return "".join(out) + if text.startswith("'"): + raise ConfigError(f"single-quoted scalar: {text!r}") + if text == "true": + return True + if text == "false": + return False + if text in ("null", "~"): + return None + try: + return int(text) + except ValueError: + pass + return text + + +def _parse_mapping(entries, i, indent): + result = {} + while i < len(entries) and entries[i][0] == indent and not entries[i][2]: + _, content, _ = entries[i] + key, sep, rest = content.partition(":") + if not sep: + raise ConfigError(f"expected 'key: value' or 'key:', found {content!r}") + key = key.strip() + rest = rest.strip() + i += 1 + if rest == "": + if i < len(entries) and entries[i][0] > indent: + value, i = _parse_node(entries, i, entries[i][0]) + else: + value = None + else: + value = _parse_scalar(rest) + result[key] = value + return result, i + + +def _parse_sequence(entries, i, indent): + items = [] + while i < len(entries) and entries[i][0] == indent and entries[i][2]: + _, content, _ = entries[i] + i += 1 + if content is None: + if i < len(entries) and entries[i][0] > indent: + value, i = _parse_node(entries, i, entries[i][0]) + else: + raise ConfigError("sequence item has no value") + elif ":" in content and not content.startswith('"'): + item_indent = indent + 2 + synthetic = [(item_indent, content, False)] + entries[i:] + value, consumed = _parse_mapping(synthetic, 0, item_indent) + i += consumed - 1 + else: + value = _parse_scalar(content) + items.append(value) + return items, i + + +def _parse_node(entries, i, indent): + if i >= len(entries) or entries[i][0] != indent: + raise ConfigError(f"expected content at indent {indent}") + if entries[i][2]: + return _parse_sequence(entries, i, indent) + return _parse_mapping(entries, i, indent) + + +def parse_config(text: str) -> dict: + entries = _tokenize(text) + if not entries: + return {} + value, next_i = _parse_node(entries, 0, entries[0][0]) + if next_i != len(entries): + raise ConfigError(f"unexpected indentation back to {entries[next_i][0]} after the document") + if not isinstance(value, dict): + raise ConfigError("document root is not a mapping") + return value + + +_RESERVED_SCALARS = {"true", "false", "null", "~", ""} + + +def _needs_quotes(text: str) -> bool: + if text in _RESERVED_SCALARS: + return True + if text != text.strip(): + return True + if ": " in text or text.endswith(":"): + return True + if text[0] in "\"'[]{}&*|>#-?:@`": + return True + try: + int(text) + return True + except ValueError: + pass + return False + + +def _dump_scalar(value) -> str: + if isinstance(value, bool): + return "true" if value else "false" + if isinstance(value, int): + return str(value) + if value is None: + return "null" + text = str(value) + if _needs_quotes(text): + escaped = text.replace("\\", "\\\\").replace('"', '\\"') + return f'"{escaped}"' + return text + + +def _dump_node(value, indent: int, lines: list) -> None: + pad = " " * indent + if isinstance(value, dict): + for key, item in value.items(): + if isinstance(item, (dict, list)) and item: + lines.append(f"{pad}{key}:") + _dump_node(item, indent + 2, lines) + elif isinstance(item, dict): + lines.append(f"{pad}{key}: {{}}") + elif isinstance(item, list): + lines.append(f"{pad}{key}: []") + else: + lines.append(f"{pad}{key}: {_dump_scalar(item)}") + elif isinstance(value, list): + for item in value: + if isinstance(item, dict): + keys = list(item.items()) + first_key, first_val = keys[0] + if isinstance(first_val, (dict, list)) and first_val: + lines.append(f"{pad}- {first_key}:") + _dump_node(first_val, indent + 4, lines) + else: + lines.append(f"{pad}- {first_key}: {_dump_scalar(first_val)}") + for key, val in keys[1:]: + if isinstance(val, (dict, list)) and val: + lines.append(f"{pad} {key}:") + _dump_node(val, indent + 4, lines) + else: + lines.append(f"{pad} {key}: {_dump_scalar(val)}") + elif isinstance(item, list): + if not item: + lines.append(f"{pad}- []") + continue + first = item[0] + if isinstance(first, list): + raise ConfigError("triply-nested list is outside the supported subset") + lines.append(f"{pad}- - {_dump_scalar(first)}") + for element in item[1:]: + lines.append(f"{pad} - {_dump_scalar(element)}") + else: + lines.append(f"{pad}- {_dump_scalar(item)}") + + +def dump_config(doc: dict) -> str: + lines: list = [] + _dump_node(doc, 0, lines) + return "\n".join(lines) + "\n" if lines else "" + + +# --------------------------------------------------------------------------- +# File location and loading. +# --------------------------------------------------------------------------- + + +def plugin_root() -> Path: + return Path(__file__).resolve().parent.parent + + +def config_path(repo_root: Path) -> Path: + return repo_root / CONFIG_DIR / CONFIG_NAME + + +def manifest_path(repo_root: Path) -> Path: + return repo_root / CONFIG_DIR / MANIFEST_NAME + + +def load_raw_doc(repo_root: Path): + """Return (doc, error) - doc is None with a config file present but unparseable.""" + path = config_path(repo_root) + if not path.exists(): + return None, None + try: + return parse_config(path.read_text()), None + except ConfigError as exc: + return None, str(exc) + + +def builtin_doc() -> dict: + return { + "version": 1, + "defaults": dict(BUILTIN_DEFAULTS), + "types": {name: dict(t) for name, t in BUILTIN_TYPES.items()}, + "phases": [dict(p) for p in BUILTIN_PHASES], + } + + +def effective_doc(repo_root: Path): + """Return (doc, using_builtin, error).""" + raw, error = load_raw_doc(repo_root) + if error is not None: + return None, False, error + if raw is None: + return builtin_doc(), True, None + return raw, False, None + + +# --------------------------------------------------------------------------- +# Placeholder substitution. +# --------------------------------------------------------------------------- + + +def substitute(text: str, context: dict) -> str: + if not isinstance(text, str): + return text + for name in PLACEHOLDERS: + key = name[2:-1] + if key in context and name in text: + text = text.replace(name, str(context[key])) + return text + + +def placeholders_in(text: str) -> list: + if not isinstance(text, str): + return [] + found = [] + i = 0 + while True: + i = text.find("${", i) + if i == -1: + break + j = text.find("}", i) + if j == -1: + break + found.append(text[i : j + 1]) + i = j + 1 + return found + + +# --------------------------------------------------------------------------- +# Resolution: phase list with type axes merged in, checks/attempts precedence +# applied, and paths substituted. +# --------------------------------------------------------------------------- + + +class ResolutionError(Exception): + def __init__(self, findings): + super().__init__("; ".join(findings)) + self.findings = list(findings) + + +def _normalize_under_root(raw: str, roots: dict) -> str: + """Substitute ${plugin_root}/${repo_root} and normalize .. segments.""" + text = raw + for name in ("${plugin_root}", "${repo_root}"): + key = name[2:-1] + if key in roots: + text = text.replace(name, str(roots[key])) + return text + + +def _resolves_under_allowlist(path_text: str, roots: dict) -> bool: + resolved = Path(_normalize_under_root(path_text, roots)) + try: + resolved = resolved.resolve() if resolved.is_absolute() else (roots["repo_root"] / resolved).resolve() + except (OSError, RuntimeError): + return False + for root_key in ("plugin_root", "repo_root"): + root = Path(roots[root_key]).resolve() + try: + resolved.relative_to(root) + return True + except ValueError: + continue + return False + + +def sha256_of(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def classify_phase(phase: dict, roots: dict, repo_root: Path): + """Return (class_name, findings) where class_name in built-in/injected/foreign.""" + skill = phase.get("skill", "") + findings = [] + if isinstance(skill, str) and skill.startswith("${plugin_root}"): + return "built-in", findings + + factory_dir = repo_root / CONFIG_DIR + resolved_skill = Path(_normalize_under_root(skill, roots)) if isinstance(skill, str) else None + if resolved_skill is not None and not resolved_skill.is_absolute(): + resolved_skill = repo_root / resolved_skill + + under_factory_skills = False + if resolved_skill is not None: + try: + resolved_skill.resolve().relative_to((factory_dir / "skills").resolve()) + under_factory_skills = True + except (ValueError, OSError): + under_factory_skills = False + + if not under_factory_skills: + return "foreign", findings + + manifest_file = manifest_path(repo_root) + if not manifest_file.exists(): + findings.append( + f"phase {phase.get('id', '?')}: skill under .factory/skills/ but .factory/.inject.json is missing" + ) + return "foreign", findings + try: + manifest = json.loads(manifest_file.read_text()) + except (json.JSONDecodeError, OSError): + findings.append( + f"phase {phase.get('id', '?')}: .factory/.inject.json is unparseable" + ) + return "foreign", findings + + files = manifest.get("files", {}) if isinstance(manifest, dict) else {} + try: + rel = resolved_skill.resolve().relative_to(factory_dir.resolve()) + except (ValueError, OSError): + findings.append(f"phase {phase.get('id', '?')}: skill path could not be related to .factory/") + return "foreign", findings + rel_str = str(rel) + entry_hash = files.get(rel_str) + if entry_hash is None: + findings.append( + f"phase {phase.get('id', '?')}: {rel_str} has no entry in .factory/.inject.json" + ) + return "foreign", findings + if not resolved_skill.exists(): + findings.append(f"phase {phase.get('id', '?')}: {rel_str} does not exist") + return "foreign", findings + if sha256_of(resolved_skill) != entry_hash: + findings.append( + f"phase {phase.get('id', '?')}: {rel_str} does not match its .factory/.inject.json entry" + ) + return "foreign", findings + return "injected", findings + + +def resolve(repo_root: Path, plugin_root_path: Path = None): + """Return (resolved, findings). resolved is None when findings is non-empty.""" + doc, using_builtin, parse_error = effective_doc(repo_root) + findings = [] + if parse_error is not None: + findings.append(f"malformed document: {parse_error}") + return None, findings + + proot = plugin_root_path or plugin_root() + roots = { + "plugin_root": str(proot), + "repo_root": str(repo_root), + } + + version = doc.get("version") + if version != 1: + findings.append(f"unsupported version: {version!r}") + + defaults = doc.get("defaults") or {} + types = doc.get("types") or {} + phases = doc.get("phases") or [] + + if not isinstance(phases, list) or not phases: + findings.append("empty phase list") + return None, findings + + seen_ids = set() + for phase in phases: + if not isinstance(phase, dict): + findings.append(f"malformed phase entry: {phase!r}") + continue + pid = phase.get("id") + if pid in RESERVED_PHASE_IDS: + findings.append(f"phase id {pid!r} is reserved") + if pid in seen_ids: + findings.append(f"duplicate phase id: {pid!r}") + seen_ids.add(pid) + ptype = phase.get("type") + if ptype not in types and not (using_builtin and ptype in BUILTIN_TYPES): + findings.append(f"phase {pid!r} declares undeclared type {ptype!r}") + + if findings: + return None, findings + + resolved_phases = [] + ceiling = defaults.get("ceiling", BUILTIN_DEFAULTS["ceiling"]) + + for phase in phases: + pid = phase["id"] + ptype = phase["type"] + type_def = types.get(ptype, BUILTIN_TYPES.get(ptype, {})) + + skill_raw = phase.get("skill", "") + if not _resolves_under_allowlist(skill_raw, roots): + findings.append(f"phase {pid!r}: skill path {skill_raw!r} does not resolve under the plugin or repo root") + skill_resolved = substitute(skill_raw, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) + + checks_raw = phase["checks"] if "checks" in phase else type_def.get("checks", []) + if not isinstance(checks_raw, list): + findings.append(f"phase {pid!r}: checks must be a list of argv lists") + checks_raw = [] + resolved_checks = [] + for entry in checks_raw: + if not isinstance(entry, list) or not entry or not all(isinstance(e, str) for e in entry): + findings.append(f"phase {pid!r}: a checks entry must be an argv list of strings, not a shell string") + continue + executable = entry[0] + for ph in placeholders_in(executable): + if ph not in EXECUTABLE_PLACEHOLDERS: + findings.append( + f"phase {pid!r}: check executable {executable!r} uses placeholder {ph}, " + f"only {EXECUTABLE_PLACEHOLDERS} are allowed there" + ) + for element in entry: + for ph in placeholders_in(element): + if ph not in CHECK_PLACEHOLDERS: + findings.append(f"phase {pid!r}: check {entry!r} uses unsupported placeholder {ph}") + if not _resolves_under_allowlist(executable, roots): + findings.append(f"phase {pid!r}: check executable {executable!r} does not resolve under the plugin or repo root") + resolved_entry = [substitute(e, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) for e in entry] + resolved_checks.append(resolved_entry) + + seals_raw = type_def.get("seals", []) + resolved_seals = list(seals_raw) + + attempts = phase.get("attempts", type_def.get("attempts", defaults.get("attempts", DEFAULT_ATTEMPTS))) + + phase_class, class_findings = classify_phase(phase, roots, repo_root) + findings.extend(class_findings) + unattended_safe = phase.get("unattended_safe", False) + if phase_class == "foreign" and not unattended_safe: + findings.append(f"phase {pid!r} is foreign and lacks unattended_safe") + + resolved_phases.append( + { + "id": pid, + "type": ptype, + "class": phase_class, + "interactive": bool(type_def.get("interactive", False)), + "skill": skill_resolved, + "checks": resolved_checks, + "requires": type_def.get("requires", ""), + "seals": resolved_seals, + "attempts": attempts, + "unattended_safe": unattended_safe, + } + ) + + seen_unattended = False + for i, rp in enumerate(resolved_phases): + if rp["interactive"]: + if seen_unattended: + prev = next(p for p in resolved_phases[:i] if not p["interactive"]) + findings.append( + f"interactive phase {rp['id']!r} declared after unattended phase {prev['id']!r}" + ) + else: + seen_unattended = True + + go_after = None + for rp in resolved_phases: + if rp["interactive"]: + go_after = rp["id"] + else: + break + + if findings: + return None, findings + + return {"phases": resolved_phases, "ceiling": ceiling, "go_after": go_after, "using_builtin": using_builtin}, [] + + +# --------------------------------------------------------------------------- +# Writing: schema key order, two-space indent (dump_config's default), +# block style only, one trailing newline, a regenerated header comment. +# --------------------------------------------------------------------------- + +SCHEMA_KEY_ORDER = ("version", "defaults", "types", "phases") + + +def write_doc(repo_root: Path, doc: dict) -> None: + top = {key: doc[key] for key in SCHEMA_KEY_ORDER if key in doc} + text = HEADER + dump_config(top) + path = config_path(repo_root) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text) + + +def read_doc_for_write(repo_root: Path): + """Load the raw file for a set/unset call, defaulting to an empty shell.""" + raw, error = load_raw_doc(repo_root) + if error is not None: + raise ConfigError(error) + if raw is None: + raw = {"version": 1, "defaults": {}, "types": {}, "phases": []} + raw.setdefault("version", 1) + raw.setdefault("defaults", {}) + raw.setdefault("types", {}) + raw.setdefault("phases", []) + return raw + + +# --------------------------------------------------------------------------- +# show --resolved rendering. +# --------------------------------------------------------------------------- + + +def render_human(resolved: dict) -> str: + lines = [] + source = "built-in default" if resolved["using_builtin"] else ".factory/config.yaml" + lines.append(f"pipeline: {source}") + lines.append(f"ceiling: {resolved['ceiling']}") + lines.append(f"go: {'after ' + resolved['go_after'] if resolved['go_after'] else 'none'}") + for phase in resolved["phases"]: + lines.append(f"- {phase['id']} ({phase['type']}, {phase['class']})") + lines.append(f" interactive: {phase['interactive']}") + lines.append(f" skill: {phase['skill']}") + lines.append(f" attempts: {phase['attempts']}") + if phase["requires"]: + lines.append(f" requires: {phase['requires']}") + for entry in phase["checks"]: + lines.append(f" check: {entry}") + for seal in phase["seals"]: + lines.append(f" seal: {seal}") + if phase["unattended_safe"]: + lines.append(" unattended_safe: true") + return "\n".join(lines) + "\n" + + +def render_json(resolved: dict) -> str: + phases = [ + {"id": p["id"], "type": p["type"], "skill": p["skill"], "interactive": p["interactive"]} + for p in resolved["phases"] + ] + return json.dumps({"phases": phases}, indent=2) + "\n" + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + + +class Parser(argparse.ArgumentParser): + def error(self, message: str) -> NoReturn: + sys.stderr.write(f"{self.prog}: {message}\n") + raise SystemExit(BAD_CALL) + + +def build_parser() -> Parser: + parser = Parser(prog="factory-config.py", description=__doc__) + parser.add_argument("--repo-root", type=Path, default=Path.cwd()) + parser.add_argument("--plugin-root", type=Path, default=None) + sub = parser.add_subparsers(dest="command", required=True) + + sub.add_parser("init") + + p_show = sub.add_parser("show") + p_show.add_argument("--resolved", action="store_true") + p_show.add_argument("--json", action="store_true") + + sub.add_parser("check") + + p_set = sub.add_parser("set") + p_set.add_argument("path") + p_set.add_argument("value", nargs="?") + p_set.add_argument("--item", action="append", default=None) + + p_unset = sub.add_parser("unset") + p_unset.add_argument("path") + + p_inject = sub.add_parser("inject") + p_inject.add_argument("phases", nargs="*") + p_inject.add_argument("--force", action="store_true") + + return parser + + +def cmd_init(args) -> int: + path = config_path(args.repo_root) + if path.exists(): + print(f"{path} already exists", file=sys.stderr) + return 1 + write_doc(args.repo_root, {"version": 1, "defaults": {}, "types": {}, "phases": []}) + return 0 + + +def cmd_show(args) -> int: + resolved, findings = resolve(args.repo_root, args.plugin_root) + if findings: + for f in findings: + print(f, file=sys.stderr) + return 1 + if args.json: + print(render_json(resolved), end="") + else: + print(render_human(resolved), end="") + return 0 + + +def cmd_check(args) -> int: + resolved, findings = resolve(args.repo_root, args.plugin_root) + if findings: + for f in findings: + print(f) + return 1 + return 0 + + +def _parse_path(path: str): + """Return (head, key) for ., defaults. or types...""" + if "." not in path: + return None + head, rest = path.split(".", 1) + if head == "types": + if "." not in rest: + return None + name, axis = rest.split(".", 1) + return ("types", name, axis) + return (head, rest) + + +def _coerce_cli_value(text): + if text is None: + return None + if text == "true": + return True + if text == "false": + return False + try: + return int(text) + except ValueError: + return text + + +def cmd_set(args) -> int: + parsed = _parse_path(args.path) + if parsed is None: + print(f"malformed path: {args.path!r}", file=sys.stderr) + return BAD_CALL + + try: + doc = read_doc_for_write(args.repo_root) + except ConfigError as exc: + print(f"malformed document: {exc}", file=sys.stderr) + return BAD_CALL + + if args.item: + key_name = parsed[-1] + if key_name == "checks": + value = [list(args.item)] + else: + value = list(args.item) + else: + value = _coerce_cli_value(args.value) + + if len(parsed) == 3: + _, name, axis = parsed + doc["types"].setdefault(name, {})[axis] = value + elif parsed[0] == "defaults": + _, key = parsed + doc["defaults"][key] = value + else: + head, key = parsed + phase = next((p for p in doc["phases"] if p.get("id") == head), None) + if phase is None: + print(f"no phase with id {head!r}", file=sys.stderr) + return 1 + phase[key] = value + + write_doc(args.repo_root, doc) + return 0 + + +def cmd_unset(args) -> int: + parsed = _parse_path(args.path) + if parsed is None: + print(f"malformed path: {args.path!r}", file=sys.stderr) + return BAD_CALL + + try: + doc = read_doc_for_write(args.repo_root) + except ConfigError as exc: + print(f"malformed document: {exc}", file=sys.stderr) + return BAD_CALL + + if len(parsed) == 3: + _, name, axis = parsed + doc["types"].get(name, {}).pop(axis, None) + elif parsed[0] == "defaults": + _, key = parsed + doc["defaults"].pop(key, None) + else: + head, key = parsed + phase = next((p for p in doc["phases"] if p.get("id") == head), None) + if phase is not None: + phase.pop(key, None) + if not phase: + doc["phases"].remove(phase) + + write_doc(args.repo_root, doc) + return 0 + + +def main(argv=None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + if args.plugin_root is None: + args.plugin_root = plugin_root() + handlers = { + "init": cmd_init, + "show": cmd_show, + "check": cmd_check, + "set": cmd_set, + "unset": cmd_unset, + } + handler = handlers.get(args.command) + if handler is None: + print(f"'{args.command}' is not yet implemented", file=sys.stderr) + return BAD_CALL + return handler(args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugins/factory/references/pipeline-config.md b/plugins/factory/references/pipeline-config.md new file mode 100644 index 0000000..d72662b --- /dev/null +++ b/plugins/factory/references/pipeline-config.md @@ -0,0 +1,88 @@ +# Pipeline config + +`.factory/config.yaml` in the consuming repository declares the pipeline the orchestrator drives. +Its absence is not an error: the built-in default pipeline applies, our four phases run from the installed plugin, and nothing is written to the repo. + +## Schema + +```yaml +version: 1 +defaults: + ceiling: 12 + attempts: 3 +types: + : + interactive: true + requires: "prose the orchestrator judges by running a tool itself" + checks: + - - lint-spec.py + - ${plan_dir}/spec.md + seals: + - ${plan_dir}/spec.md + attempts: 3 +phases: + - id: + type: + skill: ${plugin_root}/phases//SKILL.md + checks: + - - check-tests.py + - ${plan_dir} + unattended_safe: true +``` + +`version` is always `1`. `defaults`, `types` and `phases` are the schema's only other top-level keys; an unknown key is a `check` finding. + +### Types: exactly five axes, no more + +- `interactive` (bool) - runs inline in the orchestrator's own session rather than launched. +- `requires` (prose) - the outcome contract the orchestrator judges when the type's `checks` don't cover it, the same way `judgment.md` reads a phase today. +- `checks` (a list of argv lists) - commands that verify the outcome. Each entry is an argv list, never a shell string. +- `seals` (a list of paths) - artifacts whose drift `run-state.py handoff`/`diff-spec` watches. May use `${plan_dir}`. +- `attempts` (int) - the per-type default retry budget. + +The type set is open: any name declared under `types` is usable by a phase. A type the orchestrator has never seen is judged by its declared axes alone. + +### Phases: an ordered list + +Each entry is `{id, type, skill, checks, unattended_safe}`. `id`, `type` and `skill` are required; `checks` and `unattended_safe` are optional per-phase overrides. + +- `checks`, when present on a phase, **replaces** the type's `checks` entirely - it never merges. +- `unattended_safe: true` is required for a foreign phase (see Phase classes below); built-in and injected phases don't need it. + +`defaults` and `types` are reserved words: no phase may take either as its `id`, since the key-path grammar below roots at both. + +### Placeholders + +A `skill` path and a check's argv elements may use `${plugin_root}`, `${repo_root}`, `${plan_dir}`, `${phase}` and `${attempt}`. A check's **executable** element (argv[0]) may only use `${plugin_root}` or `${repo_root}` - `${plan_dir}` and the others can't be allowlisted before a run exists. `show --resolved` leaves every placeholder literal; the orchestrator substitutes them when it launches or runs a check. + +### Phase classes + +Three classes, by path and provenance, established by `factory-config.py check`: + +- **built-in** - the skill path starts with `${plugin_root}`. +- **injected** - the skill sits under `.factory/skills/` and its current sha256 matches the entry `.factory/.inject.json` recorded for it. +- **foreign** - anything else, including an injected copy whose hash has drifted from its manifest entry. + +Only a foreign phase needs `unattended_safe: true`; the orchestrator never hand-repairs a foreign phase's failure, only relaunches it or ends the run. + +## The key-path grammar + +`set` and `unset` address one value at a time by a dotted path: + +- `.` - a key on an existing phase entry. The phase must already exist; `set` never creates a phase implicitly. +- `defaults.` - a key under `defaults`, created if absent. +- `types..` - an axis of a type, created if absent. + +A list-valued key takes repeated `--item` flags in place of a positional value. `unset` removes the named key; a phase left with no keys at all is dropped from `phases` entirely. + +## The manifest `check` reads + +`.factory/.inject.json`, written by `inject`, at the shape: + +```json +{"plugin": "", "version": "", "files": {"": ""}} +``` + +## The built-in default pipeline + +Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check` (ship) become `checks` entries; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. diff --git a/plugins/factory/scripts/factory-config.py b/plugins/factory/scripts/factory-config.py new file mode 100755 index 0000000..98bdc78 --- /dev/null +++ b/plugins/factory/scripts/factory-config.py @@ -0,0 +1,922 @@ +#!/usr/bin/env python3 +"""Read, write, and validate .factory/config.yaml: the declared pipeline. + +Usage: + factory-config.py init + factory-config.py show [--resolved] [--json] + factory-config.py check + factory-config.py set [--item ...] + factory-config.py unset + +Exit codes: + init 0 written; 1 the file already exists + show 0 always (there is nothing to fail on a read) + check 0 valid; 1 one or more findings, one per line; 3 a bad call + set 0 recorded; 1 the phase id in the path does not exist; + 3 a malformed path or an unsupported document construct + unset 0 recorded (a no-op if the key was already absent); + 3 a malformed path + +3 always means the call itself could not be carried out - a malformed key +path, an unparseable document, or a usage error - never a finding about the +pipeline's own content. `check` is the single validator: every rule below is +enforced there and nowhere else, so the orchestrator's preflight only ever +relays `check`'s output rather than judging the file itself. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +from pathlib import Path +from typing import NoReturn + +BAD_CALL = 3 +CONFIG_DIR = ".factory" +CONFIG_NAME = "config.yaml" +MANIFEST_NAME = ".inject.json" +HEADER = ( + "# .factory/config.yaml\n" + "# Managed by factory-config.py. Edit with `factory-config.py set`.\n" +) +PLACEHOLDERS = ("${plugin_root}", "${repo_root}", "${plan_dir}", "${phase}", "${attempt}") +CHECK_PLACEHOLDERS = ("${plugin_root}", "${repo_root}", "${plan_dir}", "${phase}", "${attempt}") +EXECUTABLE_PLACEHOLDERS = ("${plugin_root}", "${repo_root}") +RESERVED_PHASE_IDS = {"defaults", "types"} + +# --------------------------------------------------------------------------- +# The built-in default pipeline: today's four phases, equal to what +# judgment.md and the copied phase skills already enforce. Every path uses +# ${plugin_root} so a plugin move only ever touches this one constant. +# --------------------------------------------------------------------------- +BUILTIN_TYPES = { + "interview": { + "interactive": True, + "requires": "the go was recorded (the interview happened inline in this session, " + "so the orchestrator already knows)", + "checks": [["lint-spec.py", "${plan_dir}/spec.md"]], + "seals": ["${plan_dir}/spec.md"], + "attempts": 3, + }, + "review": { + "interactive": False, + "requires": 'spec-review_N.md contains a line whose whole text is exactly ' + '"Verdict: APPROVED" - nothing after it - with a "Rounds:" line', + "attempts": 3, + }, + "implement": { + "interactive": False, + "requires": "the spec's Validation block is green, and commits since handoff " + "match the change plan's numbering", + "checks": [["check-tests.py", "${plan_dir}"]], + "attempts": 3, + }, + "ship": { + "interactive": False, + "requires": "review_N.md carries a verdict for HEAD and gh pr view shows the PR " + "open with head equal to HEAD, or (without a GitHub remote) git ls-remote origin " + "shows the branch at HEAD and pr.md is written", + "checks": [["pr-evidence.py", "check"]], + "attempts": 3, + }, +} +BUILTIN_PHASES = [ + {"id": "scope", "type": "interview", "skill": "${plugin_root}/skills/scope/SKILL.md"}, + {"id": "scope-review", "type": "review", "skill": "${plugin_root}/skills/scope-review/SKILL.md"}, + {"id": "build", "type": "implement", "skill": "${plugin_root}/skills/build/SKILL.md"}, + {"id": "ship", "type": "ship", "skill": "${plugin_root}/skills/ship/SKILL.md"}, +] +BUILTIN_DEFAULTS = {"ceiling": 12} +DEFAULT_ATTEMPTS = 3 + +# --------------------------------------------------------------------------- +# A strict subset of YAML: block mappings, block sequences (including nested +# argv-list-of-lists for `checks`), plain and double-quoted scalars, and +# comments. Flow syntax, anchors, multiline block scalars, and single-quoted +# strings are rejected with ConfigError rather than silently mishandled. +# --------------------------------------------------------------------------- + + +class ConfigError(Exception): + """An unsupported construct or a malformed document. Never a traceback.""" + + +def _strip_comment(text: str) -> str: + in_quotes = False + escaped = False + for i, ch in enumerate(text): + if escaped: + escaped = False + continue + if ch == "\\" and in_quotes: + escaped = True + continue + if ch == '"': + in_quotes = not in_quotes + continue + if ch == "#" and not in_quotes and (i == 0 or text[i - 1].isspace()): + return text[:i] + return text + + +def _raw_lines(text: str) -> list: + out = [] + for raw in text.splitlines(): + leading = raw[: len(raw) - len(raw.lstrip(" \t"))] + if "\t" in leading: + raise ConfigError("a tab in leading indentation") + stripped = _strip_comment(raw).rstrip() + if not stripped.strip(): + continue + indent = len(stripped) - len(stripped.lstrip(" ")) + if indent % 2 != 0: + raise ConfigError(f"odd indentation ({indent} spaces): {stripped.strip()!r}") + out.append((indent, stripped.strip())) + return out + + +def _unfold(indent: int, content: str) -> list: + if not (content == "-" or content.startswith("- ")): + return [(indent, content, False)] + out = [] + while True: + rest = "" if content == "-" else content[2:] + if rest == "": + out.append((indent, None, True)) + return out + if rest == "-" or rest.startswith("- "): + out.append((indent, None, True)) + indent += 2 + content = rest + continue + out.append((indent, rest, True)) + return out + + +def _tokenize(text: str) -> list: + entries = [] + for indent, content in _raw_lines(text): + entries.extend(_unfold(indent, content)) + return entries + + +def _parse_scalar(text: str): + text = text.strip() + if text == "": + return "" + if text == "[]": + return [] + if text == "{}": + return {} + if text.startswith("["): + raise ConfigError(f"flow sequence: {text!r}") + if text.startswith("{"): + raise ConfigError(f"flow mapping: {text!r}") + if text in ("|", ">") or (len(text) == 2 and text[0] in "|>" and text[1] in "+-"): + raise ConfigError(f"multiline scalar block: {text!r}") + if text[0] in "&*": + raise ConfigError(f"anchor or alias: {text!r}") + if text.startswith('"'): + if not text.endswith('"') or len(text) < 2: + raise ConfigError(f"unterminated double-quoted scalar: {text!r}") + body = text[1:-1] + out = [] + i = 0 + while i < len(body): + ch = body[i] + if ch == "\\" and i + 1 < len(body): + nxt = body[i + 1] + out.append({"n": "\n", "t": "\t", '"': '"', "\\": "\\"}.get(nxt, nxt)) + i += 2 + continue + out.append(ch) + i += 1 + return "".join(out) + if text.startswith("'"): + raise ConfigError(f"single-quoted scalar: {text!r}") + if text == "true": + return True + if text == "false": + return False + if text in ("null", "~"): + return None + try: + return int(text) + except ValueError: + pass + return text + + +def _parse_mapping(entries, i, indent): + result = {} + while i < len(entries) and entries[i][0] == indent and not entries[i][2]: + _, content, _ = entries[i] + key, sep, rest = content.partition(":") + if not sep: + raise ConfigError(f"expected 'key: value' or 'key:', found {content!r}") + key = key.strip() + rest = rest.strip() + i += 1 + if rest == "": + if i < len(entries) and entries[i][0] > indent: + value, i = _parse_node(entries, i, entries[i][0]) + else: + value = None + else: + value = _parse_scalar(rest) + result[key] = value + return result, i + + +def _parse_sequence(entries, i, indent): + items = [] + while i < len(entries) and entries[i][0] == indent and entries[i][2]: + _, content, _ = entries[i] + i += 1 + if content is None: + if i < len(entries) and entries[i][0] > indent: + value, i = _parse_node(entries, i, entries[i][0]) + else: + raise ConfigError("sequence item has no value") + elif ":" in content and not content.startswith('"'): + item_indent = indent + 2 + synthetic = [(item_indent, content, False)] + entries[i:] + value, consumed = _parse_mapping(synthetic, 0, item_indent) + i += consumed - 1 + else: + value = _parse_scalar(content) + items.append(value) + return items, i + + +def _parse_node(entries, i, indent): + if i >= len(entries) or entries[i][0] != indent: + raise ConfigError(f"expected content at indent {indent}") + if entries[i][2]: + return _parse_sequence(entries, i, indent) + return _parse_mapping(entries, i, indent) + + +def parse_config(text: str) -> dict: + entries = _tokenize(text) + if not entries: + return {} + value, next_i = _parse_node(entries, 0, entries[0][0]) + if next_i != len(entries): + raise ConfigError(f"unexpected indentation back to {entries[next_i][0]} after the document") + if not isinstance(value, dict): + raise ConfigError("document root is not a mapping") + return value + + +_RESERVED_SCALARS = {"true", "false", "null", "~", ""} + + +def _needs_quotes(text: str) -> bool: + if text in _RESERVED_SCALARS: + return True + if text != text.strip(): + return True + if ": " in text or text.endswith(":"): + return True + if text[0] in "\"'[]{}&*|>#-?:@`": + return True + try: + int(text) + return True + except ValueError: + pass + return False + + +def _dump_scalar(value) -> str: + if isinstance(value, bool): + return "true" if value else "false" + if isinstance(value, int): + return str(value) + if value is None: + return "null" + text = str(value) + if _needs_quotes(text): + escaped = text.replace("\\", "\\\\").replace('"', '\\"') + return f'"{escaped}"' + return text + + +def _dump_node(value, indent: int, lines: list) -> None: + pad = " " * indent + if isinstance(value, dict): + for key, item in value.items(): + if isinstance(item, (dict, list)) and item: + lines.append(f"{pad}{key}:") + _dump_node(item, indent + 2, lines) + elif isinstance(item, dict): + lines.append(f"{pad}{key}: {{}}") + elif isinstance(item, list): + lines.append(f"{pad}{key}: []") + else: + lines.append(f"{pad}{key}: {_dump_scalar(item)}") + elif isinstance(value, list): + for item in value: + if isinstance(item, dict): + keys = list(item.items()) + first_key, first_val = keys[0] + if isinstance(first_val, (dict, list)) and first_val: + lines.append(f"{pad}- {first_key}:") + _dump_node(first_val, indent + 4, lines) + else: + lines.append(f"{pad}- {first_key}: {_dump_scalar(first_val)}") + for key, val in keys[1:]: + if isinstance(val, (dict, list)) and val: + lines.append(f"{pad} {key}:") + _dump_node(val, indent + 4, lines) + else: + lines.append(f"{pad} {key}: {_dump_scalar(val)}") + elif isinstance(item, list): + if not item: + lines.append(f"{pad}- []") + continue + first = item[0] + if isinstance(first, list): + raise ConfigError("triply-nested list is outside the supported subset") + lines.append(f"{pad}- - {_dump_scalar(first)}") + for element in item[1:]: + lines.append(f"{pad} - {_dump_scalar(element)}") + else: + lines.append(f"{pad}- {_dump_scalar(item)}") + + +def dump_config(doc: dict) -> str: + lines: list = [] + _dump_node(doc, 0, lines) + return "\n".join(lines) + "\n" if lines else "" + + +# --------------------------------------------------------------------------- +# File location and loading. +# --------------------------------------------------------------------------- + + +def plugin_root() -> Path: + return Path(__file__).resolve().parent.parent + + +def config_path(repo_root: Path) -> Path: + return repo_root / CONFIG_DIR / CONFIG_NAME + + +def manifest_path(repo_root: Path) -> Path: + return repo_root / CONFIG_DIR / MANIFEST_NAME + + +def load_raw_doc(repo_root: Path): + """Return (doc, error) - doc is None with a config file present but unparseable.""" + path = config_path(repo_root) + if not path.exists(): + return None, None + try: + return parse_config(path.read_text()), None + except ConfigError as exc: + return None, str(exc) + + +def builtin_doc() -> dict: + return { + "version": 1, + "defaults": dict(BUILTIN_DEFAULTS), + "types": {name: dict(t) for name, t in BUILTIN_TYPES.items()}, + "phases": [dict(p) for p in BUILTIN_PHASES], + } + + +def effective_doc(repo_root: Path): + """Return (doc, using_builtin, error).""" + raw, error = load_raw_doc(repo_root) + if error is not None: + return None, False, error + if raw is None: + return builtin_doc(), True, None + return raw, False, None + + +# --------------------------------------------------------------------------- +# Placeholder substitution. +# --------------------------------------------------------------------------- + + +def substitute(text: str, context: dict) -> str: + if not isinstance(text, str): + return text + for name in PLACEHOLDERS: + key = name[2:-1] + if key in context and name in text: + text = text.replace(name, str(context[key])) + return text + + +def placeholders_in(text: str) -> list: + if not isinstance(text, str): + return [] + found = [] + i = 0 + while True: + i = text.find("${", i) + if i == -1: + break + j = text.find("}", i) + if j == -1: + break + found.append(text[i : j + 1]) + i = j + 1 + return found + + +# --------------------------------------------------------------------------- +# Resolution: phase list with type axes merged in, checks/attempts precedence +# applied, and paths substituted. +# --------------------------------------------------------------------------- + + +class ResolutionError(Exception): + def __init__(self, findings): + super().__init__("; ".join(findings)) + self.findings = list(findings) + + +def _normalize_under_root(raw: str, roots: dict) -> str: + """Substitute ${plugin_root}/${repo_root} and normalize .. segments.""" + text = raw + for name in ("${plugin_root}", "${repo_root}"): + key = name[2:-1] + if key in roots: + text = text.replace(name, str(roots[key])) + return text + + +def _resolves_under_allowlist(path_text: str, roots: dict) -> bool: + resolved = Path(_normalize_under_root(path_text, roots)) + try: + resolved = resolved.resolve() if resolved.is_absolute() else (roots["repo_root"] / resolved).resolve() + except (OSError, RuntimeError): + return False + for root_key in ("plugin_root", "repo_root"): + root = Path(roots[root_key]).resolve() + try: + resolved.relative_to(root) + return True + except ValueError: + continue + return False + + +def sha256_of(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def classify_phase(phase: dict, roots: dict, repo_root: Path): + """Return (class_name, findings) where class_name in built-in/injected/foreign.""" + skill = phase.get("skill", "") + findings = [] + if isinstance(skill, str) and skill.startswith("${plugin_root}"): + return "built-in", findings + + factory_dir = repo_root / CONFIG_DIR + resolved_skill = Path(_normalize_under_root(skill, roots)) if isinstance(skill, str) else None + if resolved_skill is not None and not resolved_skill.is_absolute(): + resolved_skill = repo_root / resolved_skill + + under_factory_skills = False + if resolved_skill is not None: + try: + resolved_skill.resolve().relative_to((factory_dir / "skills").resolve()) + under_factory_skills = True + except (ValueError, OSError): + under_factory_skills = False + + if not under_factory_skills: + return "foreign", findings + + manifest_file = manifest_path(repo_root) + if not manifest_file.exists(): + findings.append( + f"phase {phase.get('id', '?')}: skill under .factory/skills/ but .factory/.inject.json is missing" + ) + return "foreign", findings + try: + manifest = json.loads(manifest_file.read_text()) + except (json.JSONDecodeError, OSError): + findings.append( + f"phase {phase.get('id', '?')}: .factory/.inject.json is unparseable" + ) + return "foreign", findings + + files = manifest.get("files", {}) if isinstance(manifest, dict) else {} + try: + rel = resolved_skill.resolve().relative_to(factory_dir.resolve()) + except (ValueError, OSError): + findings.append(f"phase {phase.get('id', '?')}: skill path could not be related to .factory/") + return "foreign", findings + rel_str = str(rel) + entry_hash = files.get(rel_str) + if entry_hash is None: + findings.append( + f"phase {phase.get('id', '?')}: {rel_str} has no entry in .factory/.inject.json" + ) + return "foreign", findings + if not resolved_skill.exists(): + findings.append(f"phase {phase.get('id', '?')}: {rel_str} does not exist") + return "foreign", findings + if sha256_of(resolved_skill) != entry_hash: + findings.append( + f"phase {phase.get('id', '?')}: {rel_str} does not match its .factory/.inject.json entry" + ) + return "foreign", findings + return "injected", findings + + +def resolve(repo_root: Path, plugin_root_path: Path = None): + """Return (resolved, findings). resolved is None when findings is non-empty.""" + doc, using_builtin, parse_error = effective_doc(repo_root) + findings = [] + if parse_error is not None: + findings.append(f"malformed document: {parse_error}") + return None, findings + + proot = plugin_root_path or plugin_root() + roots = { + "plugin_root": str(proot), + "repo_root": str(repo_root), + } + + version = doc.get("version") + if version != 1: + findings.append(f"unsupported version: {version!r}") + + defaults = doc.get("defaults") or {} + types = doc.get("types") or {} + phases = doc.get("phases") or [] + + if not isinstance(phases, list) or not phases: + findings.append("empty phase list") + return None, findings + + seen_ids = set() + for phase in phases: + if not isinstance(phase, dict): + findings.append(f"malformed phase entry: {phase!r}") + continue + pid = phase.get("id") + if pid in RESERVED_PHASE_IDS: + findings.append(f"phase id {pid!r} is reserved") + if pid in seen_ids: + findings.append(f"duplicate phase id: {pid!r}") + seen_ids.add(pid) + ptype = phase.get("type") + if ptype not in types and not (using_builtin and ptype in BUILTIN_TYPES): + findings.append(f"phase {pid!r} declares undeclared type {ptype!r}") + + if findings: + return None, findings + + resolved_phases = [] + ceiling = defaults.get("ceiling", BUILTIN_DEFAULTS["ceiling"]) + + for phase in phases: + pid = phase["id"] + ptype = phase["type"] + type_def = types.get(ptype, BUILTIN_TYPES.get(ptype, {})) + + skill_raw = phase.get("skill", "") + if not _resolves_under_allowlist(skill_raw, roots): + findings.append(f"phase {pid!r}: skill path {skill_raw!r} does not resolve under the plugin or repo root") + skill_resolved = substitute(skill_raw, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) + + checks_raw = phase["checks"] if "checks" in phase else type_def.get("checks", []) + if not isinstance(checks_raw, list): + findings.append(f"phase {pid!r}: checks must be a list of argv lists") + checks_raw = [] + resolved_checks = [] + for entry in checks_raw: + if not isinstance(entry, list) or not entry or not all(isinstance(e, str) for e in entry): + findings.append(f"phase {pid!r}: a checks entry must be an argv list of strings, not a shell string") + continue + executable = entry[0] + for ph in placeholders_in(executable): + if ph not in EXECUTABLE_PLACEHOLDERS: + findings.append( + f"phase {pid!r}: check executable {executable!r} uses placeholder {ph}, " + f"only {EXECUTABLE_PLACEHOLDERS} are allowed there" + ) + for element in entry: + for ph in placeholders_in(element): + if ph not in CHECK_PLACEHOLDERS: + findings.append(f"phase {pid!r}: check {entry!r} uses unsupported placeholder {ph}") + if not _resolves_under_allowlist(executable, roots): + findings.append(f"phase {pid!r}: check executable {executable!r} does not resolve under the plugin or repo root") + resolved_entry = [substitute(e, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) for e in entry] + resolved_checks.append(resolved_entry) + + seals_raw = type_def.get("seals", []) + resolved_seals = list(seals_raw) + + attempts = phase.get("attempts", type_def.get("attempts", defaults.get("attempts", DEFAULT_ATTEMPTS))) + + phase_class, class_findings = classify_phase(phase, roots, repo_root) + findings.extend(class_findings) + unattended_safe = phase.get("unattended_safe", False) + if phase_class == "foreign" and not unattended_safe: + findings.append(f"phase {pid!r} is foreign and lacks unattended_safe") + + resolved_phases.append( + { + "id": pid, + "type": ptype, + "class": phase_class, + "interactive": bool(type_def.get("interactive", False)), + "skill": skill_resolved, + "checks": resolved_checks, + "requires": type_def.get("requires", ""), + "seals": resolved_seals, + "attempts": attempts, + "unattended_safe": unattended_safe, + } + ) + + seen_unattended = False + for i, rp in enumerate(resolved_phases): + if rp["interactive"]: + if seen_unattended: + prev = next(p for p in resolved_phases[:i] if not p["interactive"]) + findings.append( + f"interactive phase {rp['id']!r} declared after unattended phase {prev['id']!r}" + ) + else: + seen_unattended = True + + go_after = None + for rp in resolved_phases: + if rp["interactive"]: + go_after = rp["id"] + else: + break + + if findings: + return None, findings + + return {"phases": resolved_phases, "ceiling": ceiling, "go_after": go_after, "using_builtin": using_builtin}, [] + + +# --------------------------------------------------------------------------- +# Writing: schema key order, two-space indent (dump_config's default), +# block style only, one trailing newline, a regenerated header comment. +# --------------------------------------------------------------------------- + +SCHEMA_KEY_ORDER = ("version", "defaults", "types", "phases") + + +def write_doc(repo_root: Path, doc: dict) -> None: + top = {key: doc[key] for key in SCHEMA_KEY_ORDER if key in doc} + text = HEADER + dump_config(top) + path = config_path(repo_root) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text) + + +def read_doc_for_write(repo_root: Path): + """Load the raw file for a set/unset call, defaulting to an empty shell.""" + raw, error = load_raw_doc(repo_root) + if error is not None: + raise ConfigError(error) + if raw is None: + raw = {"version": 1, "defaults": {}, "types": {}, "phases": []} + raw.setdefault("version", 1) + raw.setdefault("defaults", {}) + raw.setdefault("types", {}) + raw.setdefault("phases", []) + return raw + + +# --------------------------------------------------------------------------- +# show --resolved rendering. +# --------------------------------------------------------------------------- + + +def render_human(resolved: dict) -> str: + lines = [] + source = "built-in default" if resolved["using_builtin"] else ".factory/config.yaml" + lines.append(f"pipeline: {source}") + lines.append(f"ceiling: {resolved['ceiling']}") + lines.append(f"go: {'after ' + resolved['go_after'] if resolved['go_after'] else 'none'}") + for phase in resolved["phases"]: + lines.append(f"- {phase['id']} ({phase['type']}, {phase['class']})") + lines.append(f" interactive: {phase['interactive']}") + lines.append(f" skill: {phase['skill']}") + lines.append(f" attempts: {phase['attempts']}") + if phase["requires"]: + lines.append(f" requires: {phase['requires']}") + for entry in phase["checks"]: + lines.append(f" check: {entry}") + for seal in phase["seals"]: + lines.append(f" seal: {seal}") + if phase["unattended_safe"]: + lines.append(" unattended_safe: true") + return "\n".join(lines) + "\n" + + +def render_json(resolved: dict) -> str: + phases = [ + {"id": p["id"], "type": p["type"], "skill": p["skill"], "interactive": p["interactive"]} + for p in resolved["phases"] + ] + return json.dumps({"phases": phases}, indent=2) + "\n" + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + + +class Parser(argparse.ArgumentParser): + def error(self, message: str) -> NoReturn: + sys.stderr.write(f"{self.prog}: {message}\n") + raise SystemExit(BAD_CALL) + + +def build_parser() -> Parser: + parser = Parser(prog="factory-config.py", description=__doc__) + parser.add_argument("--repo-root", type=Path, default=Path.cwd()) + parser.add_argument("--plugin-root", type=Path, default=None) + sub = parser.add_subparsers(dest="command", required=True) + + sub.add_parser("init") + + p_show = sub.add_parser("show") + p_show.add_argument("--resolved", action="store_true") + p_show.add_argument("--json", action="store_true") + + sub.add_parser("check") + + p_set = sub.add_parser("set") + p_set.add_argument("path") + p_set.add_argument("value", nargs="?") + p_set.add_argument("--item", action="append", default=None) + + p_unset = sub.add_parser("unset") + p_unset.add_argument("path") + + p_inject = sub.add_parser("inject") + p_inject.add_argument("phases", nargs="*") + p_inject.add_argument("--force", action="store_true") + + return parser + + +def cmd_init(args) -> int: + path = config_path(args.repo_root) + if path.exists(): + print(f"{path} already exists", file=sys.stderr) + return 1 + write_doc(args.repo_root, {"version": 1, "defaults": {}, "types": {}, "phases": []}) + return 0 + + +def cmd_show(args) -> int: + resolved, findings = resolve(args.repo_root, args.plugin_root) + if findings: + for f in findings: + print(f, file=sys.stderr) + return 1 + if args.json: + print(render_json(resolved), end="") + else: + print(render_human(resolved), end="") + return 0 + + +def cmd_check(args) -> int: + resolved, findings = resolve(args.repo_root, args.plugin_root) + if findings: + for f in findings: + print(f) + return 1 + return 0 + + +def _parse_path(path: str): + """Return (head, key) for ., defaults. or types...""" + if "." not in path: + return None + head, rest = path.split(".", 1) + if head == "types": + if "." not in rest: + return None + name, axis = rest.split(".", 1) + return ("types", name, axis) + return (head, rest) + + +def _coerce_cli_value(text): + if text is None: + return None + if text == "true": + return True + if text == "false": + return False + try: + return int(text) + except ValueError: + return text + + +def cmd_set(args) -> int: + parsed = _parse_path(args.path) + if parsed is None: + print(f"malformed path: {args.path!r}", file=sys.stderr) + return BAD_CALL + + try: + doc = read_doc_for_write(args.repo_root) + except ConfigError as exc: + print(f"malformed document: {exc}", file=sys.stderr) + return BAD_CALL + + if args.item: + key_name = parsed[-1] + if key_name == "checks": + value = [list(args.item)] + else: + value = list(args.item) + else: + value = _coerce_cli_value(args.value) + + if len(parsed) == 3: + _, name, axis = parsed + doc["types"].setdefault(name, {})[axis] = value + elif parsed[0] == "defaults": + _, key = parsed + doc["defaults"][key] = value + else: + head, key = parsed + phase = next((p for p in doc["phases"] if p.get("id") == head), None) + if phase is None: + print(f"no phase with id {head!r}", file=sys.stderr) + return 1 + phase[key] = value + + write_doc(args.repo_root, doc) + return 0 + + +def cmd_unset(args) -> int: + parsed = _parse_path(args.path) + if parsed is None: + print(f"malformed path: {args.path!r}", file=sys.stderr) + return BAD_CALL + + try: + doc = read_doc_for_write(args.repo_root) + except ConfigError as exc: + print(f"malformed document: {exc}", file=sys.stderr) + return BAD_CALL + + if len(parsed) == 3: + _, name, axis = parsed + doc["types"].get(name, {}).pop(axis, None) + elif parsed[0] == "defaults": + _, key = parsed + doc["defaults"].pop(key, None) + else: + head, key = parsed + phase = next((p for p in doc["phases"] if p.get("id") == head), None) + if phase is not None: + phase.pop(key, None) + if not phase: + doc["phases"].remove(phase) + + write_doc(args.repo_root, doc) + return 0 + + +def main(argv=None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + if args.plugin_root is None: + args.plugin_root = plugin_root() + handlers = { + "init": cmd_init, + "show": cmd_show, + "check": cmd_check, + "set": cmd_set, + "unset": cmd_unset, + } + handler = handlers.get(args.command) + if handler is None: + print(f"'{args.command}' is not yet implemented", file=sys.stderr) + return BAD_CALL + return handler(args) + + +if __name__ == "__main__": + raise SystemExit(main()) From eaa4e7a6dd00b7f3c589673bfed428eec2cfa3ca Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 09:08:52 +0200 Subject: [PATCH 20/30] feat(factory): run state carries the declared pipeline Change set 3 of config-driven-factory/spec.md: run-state.py seeds an ordered phase list, budgets and a ceiling, records attempt types, skill paths and timestamps, seals arbitrary files at handoff, adds diff-config and reports terminality. Adds the stub pipeline fixture and its tests. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/evals/stubs/stub-done.py | 13 + factory/evals/stubs/stub-failed.py | 15 + factory/evals/stubs/stub-silent.py | 2 + factory/evals/tests/test_factory_checks.py | 2 + factory/evals/tests/test_run_state.py | 272 ++++++++++++- factory/evals/tests/test_stub_pipeline.py | 156 +++++++ factory/references/factory-run.md | 9 +- factory/skills/run/SKILL.md | 2 +- factory/skills/run/scripts/run-state.py | 379 ++++++++++++++++-- plugins/factory/references/factory-run.md | 9 +- plugins/factory/skills/run/SKILL.md | 2 +- .../factory/skills/run/scripts/run-state.py | 379 ++++++++++++++++-- 12 files changed, 1151 insertions(+), 89 deletions(-) create mode 100755 factory/evals/stubs/stub-done.py create mode 100755 factory/evals/stubs/stub-failed.py create mode 100755 factory/evals/stubs/stub-silent.py create mode 100644 factory/evals/tests/test_stub_pipeline.py diff --git a/factory/evals/stubs/stub-done.py b/factory/evals/stubs/stub-done.py new file mode 100755 index 0000000..e39a054 --- /dev/null +++ b/factory/evals/stubs/stub-done.py @@ -0,0 +1,13 @@ +#!/usr/bin/env python3 +"""A stub phase: write a conforming done result to the path in argv[1].""" + +import json +import sys +from pathlib import Path + +path = Path(sys.argv[1]) +path.parent.mkdir(parents=True, exist_ok=True) +path.write_text( + json.dumps({"schema": "factory.result/1", "phase": path.stem, "status": "done", "reason": ""}), + encoding="utf-8", +) diff --git a/factory/evals/stubs/stub-failed.py b/factory/evals/stubs/stub-failed.py new file mode 100755 index 0000000..7c37d37 --- /dev/null +++ b/factory/evals/stubs/stub-failed.py @@ -0,0 +1,15 @@ +#!/usr/bin/env python3 +"""A stub phase: write a conforming failed result to the path in argv[1].""" + +import json +import sys +from pathlib import Path + +path = Path(sys.argv[1]) +path.parent.mkdir(parents=True, exist_ok=True) +path.write_text( + json.dumps( + {"schema": "factory.result/1", "phase": path.stem, "status": "failed", "reason": "stub failure"} + ), + encoding="utf-8", +) diff --git a/factory/evals/stubs/stub-silent.py b/factory/evals/stubs/stub-silent.py new file mode 100755 index 0000000..38dc73e --- /dev/null +++ b/factory/evals/stubs/stub-silent.py @@ -0,0 +1,2 @@ +#!/usr/bin/env python3 +"""A stub phase that dies quietly: exit 0 and write nothing, as a crashed subagent would.""" diff --git a/factory/evals/tests/test_factory_checks.py b/factory/evals/tests/test_factory_checks.py index c47cfe0..48f4c7d 100644 --- a/factory/evals/tests/test_factory_checks.py +++ b/factory/evals/tests/test_factory_checks.py @@ -278,6 +278,8 @@ def test_unmutated_copy_passes_and_excludes_the_harness(self) -> None: self.assertEqual(outcome.returncode, 0, outcome.output) if outcome.script_log is not None: self.assertIn("test_run_state", outcome.script_log) + self.assertIn("test_stub_pipeline", outcome.script_log) + self.assertIn("test_factory_config", outcome.script_log) self.assertNotIn("test_factory_checks", outcome.script_log) def test_decision_routed_to_a_person_fails_f03(self) -> None: diff --git a/factory/evals/tests/test_run_state.py b/factory/evals/tests/test_run_state.py index e72511b..b324bff 100644 --- a/factory/evals/tests/test_run_state.py +++ b/factory/evals/tests/test_run_state.py @@ -18,6 +18,8 @@ REPO_ROOT = Path(__file__).resolve().parents[3] SCRIPT = REPO_ROOT / "factory" / "skills" / "run" / "scripts" / "run-state.py" +SEAL_SPEC = ("--seal", ".dev/fixture-plan/spec.md") + SPEC_TWO_CHANGE_SETS = """# Fixture plan ## Research @@ -117,7 +119,7 @@ def init_plan(self) -> None: def handoff_state(self) -> dict: """Run a successful handoff on the fixture plan and return the state it wrote.""" - result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan", *SEAL_SPEC) self.assertEqual(result.returncode, 0, result.stderr) return json.loads((self.plan_dir / "factory-run.json").read_text()) @@ -131,7 +133,9 @@ def test_handoff_records_scenarios_and_not_doing_lines(self) -> None: self.init_plan() self.write_spec() state = self.handoff_state() - self.assertIsNotNone(state["spec_sha256"]) + self.assertEqual([s["path"] for s in state["seals"]], [".dev/fixture-plan/spec.md"]) + self.assertTrue(state["seals"][0]["notation"]) + self.assertIsNotNone(state["seals"][0]["sha256"]) scenario_texts = state["scenario_texts"] self.assertEqual(sorted(scenario_texts), ["1", "2"]) total = sum(len(v) for v in scenario_texts.values()) @@ -141,7 +145,7 @@ def test_handoff_records_scenarios_and_not_doing_lines(self) -> None: def test_diff_spec_after_reworded_scenario_exits_1(self) -> None: self.init_plan() self.write_spec() - run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan", *SEAL_SPEC) reworded = SPEC_TWO_CHANGE_SETS.replace( "tests: [unit] a -> b; [unit] c -> d", "tests: [unit] a -> z; [unit] c -> d", @@ -156,14 +160,14 @@ def test_diff_spec_after_reworded_scenario_exits_1(self) -> None: def test_diff_spec_unchanged_exits_0(self) -> None: self.init_plan() self.write_spec() - run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan", *SEAL_SPEC) result = run_state(self.cwd, "diff-spec", "fixture-plan") self.assertEqual(result.returncode, 0) def test_diff_spec_after_dropped_scenario_and_not_doing_line(self) -> None: self.init_plan() self.write_spec() - run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan", *SEAL_SPEC) dropped = SPEC_TWO_CHANGE_SETS.replace( "tests: [unit] a -> b; [unit] c -> d", "tests: [unit] c -> d" ).replace("- ⊘ something else - reason two\n", "") @@ -283,11 +287,19 @@ def test_handoff_ignores_numbered_lines_outside_the_change_plan(self) -> None: self.assertEqual(sorted(state["scenario_texts"]), ["1", "2"]) self.assertEqual(sum(len(v) for v in state["scenario_texts"].values()), 3) - def test_handoff_without_a_spec_exits_1(self) -> None: + def test_handoff_without_a_seal_records_sealed_nothing(self) -> None: self.init_plan() result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "factory/fixture-plan") + self.assertEqual(result.returncode, 0, result.stdout) + state = json.loads((self.plan_dir / "factory-run.json").read_text()) + self.assertEqual(state["seals"], []) + self.assertTrue(state["sealed_nothing"]) + + def test_handoff_with_a_missing_sealed_file_exits_1(self) -> None: + self.init_plan() + result = run_state(self.cwd, "handoff", "fixture-plan", "--branch", "b", *SEAL_SPEC) self.assertEqual(result.returncode, 1) - self.assertIn("no spec.md", result.stdout) + self.assertIn("no sealed file", result.stdout) def test_handoff_without_dirty_flag_records_git_status_paths(self) -> None: subprocess.run(["git", "init", "-q"], cwd=self.cwd, check=True, capture_output=True) @@ -339,7 +351,9 @@ def test_attempt_records_a_launch_then_its_result(self) -> None: launched = run_state(self.cwd, "attempt", "fixture-plan", "build") self.assertEqual(launched.returncode, 0, launched.stderr) state = json.loads((self.plan_dir / "factory-run.json").read_text()) - self.assertEqual(state["phases"]["build"], [{"status": "launched", "result": None}]) + self.assertEqual(state["phases"]["build"][0]["status"], "launched") + self.assertIsNone(state["phases"]["build"][0]["result"]) + self.assertIn("opened", state["phases"]["build"][0]) path = self.write_result({"schema": "factory.result/1", "status": "done"}) closed = run_state( self.cwd, "attempt", "fixture-plan", "build", "--status", "done", "--result", str(path) @@ -383,11 +397,11 @@ def test_diff_spec_with_a_truncated_state_file_exits_3(self) -> None: self.assertEqual(result.returncode, 3) self.assertNotIn("Traceback", result.stderr) - def test_diff_spec_without_a_spec_exits_3(self) -> None: + def test_diff_spec_when_nothing_was_sealed_reports_not_watched(self) -> None: self.init_plan() result = run_state(self.cwd, "diff-spec", "fixture-plan") - self.assertEqual(result.returncode, 3) - self.assertNotIn("Traceback", result.stderr) + self.assertEqual(result.returncode, 0) + self.assertIn("not watched", result.stdout) def test_record_with_a_corrupt_state_file_exits_3(self) -> None: self.init_plan() @@ -629,5 +643,241 @@ def test_show_exits_0(self) -> None: self.assertIn("scope: no attempts", result.stdout) +def pipeline_json(*phases: tuple[str, str, str]) -> str: + return json.dumps( + {"phases": [{"id": i, "type": ty, "skill": sk, "interactive": False} for i, ty, sk in phases]} + ) + + +class PipelineStateTest(unittest.TestCase): + """Change set 3: the run state carries the declared pipeline.""" + + def setUp(self) -> None: + self.tmp = tempfile.TemporaryDirectory() + self.cwd = Path(self.tmp.name) + self.plan_dir = self.cwd / ".dev" / "p" + self.plan_dir.mkdir(parents=True) + + def tearDown(self) -> None: + self.tmp.cleanup() + + def init(self, *extra: str) -> subprocess.CompletedProcess: + return run_state(self.cwd, "init", "p", "--request", "r", "--base", "main", *extra) + + def state(self) -> dict: + return json.loads((self.plan_dir / "factory-run.json").read_text()) + + def attempt(self, phase: str, *extra: str) -> subprocess.CompletedProcess: + return run_state(self.cwd, "attempt", "p", phase, *extra) + + def test_init_with_phases_seeds_exactly_those_in_order(self) -> None: + self.assertEqual(self.init("--phases", "a,b,c").returncode, 0) + state = self.state() + self.assertEqual(list(state["phases"]), ["a", "b", "c"]) + self.assertEqual(state["schema"], 2) + + def test_init_without_phases_seeds_the_four_built_in(self) -> None: + self.init() + self.assertEqual(list(self.state()["phases"]), ["scope", "scope-review", "build", "ship"]) + + def test_init_with_unusable_phases_or_budgets_exits_3(self) -> None: + for extra in (["--phases", "a,,b"], ["--phases", "a,a"], ["--phases", "a", "--attempts", "z=2"], + ["--phases", "a", "--attempts", "a=zero"], ["--ceiling", "0"]): + with self.subTest(extra=extra): + result = self.init(*extra) + self.assertEqual(result.returncode, 3, result.stdout) + self.assertNotIn("Traceback", result.stderr) + + def test_attempt_on_a_phase_outside_the_run_exits_3_naming_it(self) -> None: + self.init("--phases", "a,b") + result = self.attempt("c") + self.assertEqual(result.returncode, 3) + self.assertIn("'c'", result.stdout) + + def test_attempt_on_a_custom_phase_in_the_list_exits_0(self) -> None: + self.init("--phases", "design,code") + self.assertEqual(self.attempt("design").returncode, 0) + + def test_launch_records_type_skill_and_opened(self) -> None: + self.init() + self.attempt("build", "--type", "implement", "--skill", "/x/SKILL.md") + record = self.state()["phases"]["build"][0] + self.assertEqual((record["type"], record["skill"]), ("implement", "/x/SKILL.md")) + self.assertIn("opened", record) + + def test_close_records_a_closed_timestamp(self) -> None: + self.init() + self.attempt("build", "--type", "implement", "--skill", "/x") + result_file = self.cwd / "r.json" + result_file.write_text(json.dumps({"schema": "factory.result/1", "status": "done"})) + self.attempt("build", "--status", "done", "--result", str(result_file)) + record = self.state()["phases"]["build"][0] + self.assertIn("opened", record) + self.assertIn("closed", record) + + def test_type_with_a_close_status_exits_3(self) -> None: + self.init() + self.attempt("build") + result = self.attempt("build", "--status", "done", "--type", "implement") + self.assertEqual(result.returncode, 3) + + def test_failed_close_without_a_result_records_no_result_file(self) -> None: + self.init() + self.attempt("build") + self.assertEqual(self.attempt("build", "--status", "failed").returncode, 0) + record = self.state()["phases"]["build"][0] + self.assertEqual(record["result"]["reason"], "no result file") + self.attempt("build") + self.assertEqual(len(self.state()["phases"]["build"]), 2) + + def test_budgets_and_ceiling_are_reported_never_enforced(self) -> None: + self.init("--phases", "a,b", "--attempts", "b=1", "--ceiling", "2") + for _ in range(2): + self.assertEqual(self.attempt("b").returncode, 0) + shown = run_state(self.cwd, "show", "p").stdout + self.assertIn("b: 2/1 attempts (exhausted)", shown) + self.assertIn("total: 2/2 attempts (exhausted)", shown) + self.assertEqual(self.attempt("b").returncode, 0) + self.assertIn("total: 3/2 attempts (exhausted)", run_state(self.cwd, "show", "p").stdout) + + def test_a_state_file_without_schema_reads_as_the_default_pipeline(self) -> None: + legacy = { + "plan": "p", "request": "r", "base": "main", "branch": None, "spec_sha256": None, + "scenario_texts": {}, "not_doing_lines": [], "dirty_files": [], + "phases": {"scope": [], "scope-review": [], "build": [], "ship": []}, + "decisions": [], "repairs": [], + } + (self.plan_dir / "factory-run.json").write_text(json.dumps(legacy)) + shown = run_state(self.cwd, "show", "p").stdout + for phase in legacy["phases"]: + self.assertIn(f"{phase}: 0/3 attempts", shown) + self.assertIn("total: 0/12 attempts", shown) + + def test_a_legacy_state_still_watches_its_spec(self) -> None: + spec = self.plan_dir / "spec.md" + spec.write_text(SPEC_TWO_CHANGE_SETS) + self.init() + run_state(self.cwd, "handoff", "p", "--branch", "b", "--seal", str(spec)) + state = self.state() + legacy = {k: v for k, v in state.items() if k not in ("seals", "schema", "budgets", "ceiling")} + legacy["spec_sha256"] = state["seals"][0]["sha256"] + (self.plan_dir / "factory-run.json").write_text(json.dumps(legacy)) + spec.write_text(SPEC_TWO_CHANGE_SETS.replace("[unit] e -> f", "[unit] e -> g")) + result = run_state(self.cwd, "diff-spec", "p") + self.assertEqual(result.returncode, 1, result.stdout) + + def test_seal_of_a_file_without_dev_notation_is_hash_only(self) -> None: + notes = self.cwd / "brief.txt" + notes.write_text("just a brief\n") + self.init() + run_state(self.cwd, "handoff", "p", "--branch", "b", "--seal", str(notes)) + state = self.state() + self.assertEqual(len(state["seals"][0]["sha256"]), 64) + self.assertFalse(state["seals"][0]["notation"]) + self.assertEqual(state["scenario_texts"], {}) + self.assertEqual(run_state(self.cwd, "diff-spec", "p").returncode, 0) + notes.write_text("edited\n") + result = run_state(self.cwd, "diff-spec", "p") + self.assertEqual(result.returncode, 1) + self.assertIn("changed", result.stdout) + + def test_seal_of_a_spec_records_scenarios_and_not_doing_lines(self) -> None: + (self.plan_dir / "spec.md").write_text(SPEC_TWO_CHANGE_SETS) + self.init() + run_state(self.cwd, "handoff", "p", "--branch", "b", "--seal", ".dev/p/spec.md") + state = self.state() + self.assertEqual(sum(len(v) for v in state["scenario_texts"].values()), 3) + self.assertEqual(len(state["not_doing_lines"]), 2) + + def seal_pipeline(self, *phases: tuple[str, str, str]) -> Path: + config = self.cwd / ".factory" / "config.yaml" + config.parent.mkdir(exist_ok=True) + config.write_text("version: 1\n") + pipeline = self.cwd / "pipeline.json" + pipeline.write_text(pipeline_json(*phases)) + self.init() + result = run_state(self.cwd, "handoff", "p", "--branch", "b", "--config", str(config), + "--pipeline", str(pipeline)) + self.assertEqual(result.returncode, 0, result.stdout) + return config + + def diff_config(self, config: Path, *phases: tuple[str, str, str]) -> subprocess.CompletedProcess: + pipeline = self.cwd / "pipeline-now.json" + pipeline.write_text(pipeline_json(*phases)) + return run_state(self.cwd, "diff-config", "p", "--config", str(config), "--pipeline", str(pipeline)) + + ABC = (("a", "t", "/s/a"), ("b", "t", "/s/b"), ("c", "t", "/s/c")) + + def test_sealed_pipeline_records_ids_types_and_paths_in_order(self) -> None: + self.seal_pipeline(*self.ABC) + state = self.state() + self.assertEqual([p["id"] for p in state["pipeline"]], ["a", "b", "c"]) + self.assertEqual(len(state["config_sha256"]), 64) + + def test_diff_config_after_a_repoint_names_the_phase_and_both_paths(self) -> None: + config = self.seal_pipeline(*self.ABC) + result = self.diff_config(config, ("a", "t", "/s/a"), ("b", "t", "/s/NEW"), ("c", "t", "/s/c")) + self.assertEqual(result.returncode, 1) + self.assertIn("phase b repointed: /s/b -> /s/NEW", result.stdout) + + def test_diff_config_after_a_swap_names_the_phase_and_both_positions(self) -> None: + config = self.seal_pipeline(*self.ABC) + result = self.diff_config(config, ("a", "t", "/s/a"), ("c", "t", "/s/c"), ("b", "t", "/s/b")) + self.assertEqual(result.returncode, 1) + self.assertIn("phase b reordered: position 2 -> 3", result.stdout) + + def test_diff_config_names_inserted_removed_and_retyped(self) -> None: + config = self.seal_pipeline(*self.ABC) + result = self.diff_config(config, ("a", "u", "/s/a"), ("c", "t", "/s/c"), ("d", "t", "/s/d")) + self.assertEqual(result.returncode, 1) + for expected in ("phase b removed", "phase d inserted", "phase a retyped: t -> u"): + self.assertIn(expected, result.stdout) + self.assertNotIn("reordered", result.stdout) + + def test_diff_config_after_a_whitespace_only_edit_exits_0(self) -> None: + config = self.seal_pipeline(*self.ABC) + config.write_text("version: 1\n\n\n") + result = self.diff_config(config, *self.ABC) + self.assertEqual(result.returncode, 0, result.stdout) + + def test_diff_config_with_no_config_file_exits_0(self) -> None: + self.init() + result = run_state(self.cwd, "diff-config", "p") + self.assertEqual(result.returncode, 0) + self.assertIn("not watched", result.stdout) + + def test_diff_config_with_an_unusable_pipeline_file_exits_3(self) -> None: + config = self.seal_pipeline(*self.ABC) + (self.cwd / "pipeline-now.json").write_text("not json") + result = run_state(self.cwd, "diff-config", "p", "--config", str(config), + "--pipeline", str(self.cwd / "pipeline-now.json")) + self.assertEqual(result.returncode, 3) + self.assertNotIn("Traceback", result.stderr) + + def test_handoff_with_an_unusable_pipeline_file_exits_3(self) -> None: + self.init() + (self.cwd / "bad.json").write_text('{"phases": [{"id": "a"}]}') + result = run_state(self.cwd, "handoff", "p", "--branch", "b", "--pipeline", "bad.json") + self.assertEqual(result.returncode, 3) + + def test_show_reports_finished_only_after_an_advance_on_the_last_phase(self) -> None: + self.init("--phases", "a,b") + self.assertIn("finished: no", run_state(self.cwd, "show", "p").stdout) + advance = lambda phase: run_state( # noqa: E731 + self.cwd, "record", "p", stdin=json.dumps({"action": "advance", "phase": phase, "attempt": 1}) + ) + advance("a") + self.assertIn("finished: no", run_state(self.cwd, "show", "p").stdout) + advance("b") + self.assertIn("finished: yes", run_state(self.cwd, "show", "p").stdout) + + def test_state_with_a_non_numeric_budget_exits_3(self) -> None: + self.init() + state = self.state() + state["budgets"]["build"] = "three" + (self.plan_dir / "factory-run.json").write_text(json.dumps(state)) + self.assertEqual(run_state(self.cwd, "show", "p").returncode, 3) + + if __name__ == "__main__": unittest.main() diff --git a/factory/evals/tests/test_stub_pipeline.py b/factory/evals/tests/test_stub_pipeline.py new file mode 100644 index 0000000..184f736 --- /dev/null +++ b/factory/evals/tests/test_stub_pipeline.py @@ -0,0 +1,156 @@ +"""Change set 3f and 4: a stub pipeline driving run-state.py and factory-config.py. + +Each stub in factory/evals/stubs/ is a script taking the result path as argv[1]. +These tests prove the bookkeeping only - attempts, timestamps, terminality, budgets, +drift - never the orchestrator's prose loop, which only a paid per-host run proves +(D-loop-driver). +""" + +from __future__ import annotations + +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +STUBS = REPO_ROOT / "factory" / "evals" / "stubs" +RUN_STATE = REPO_ROOT / "factory" / "skills" / "run" / "scripts" / "run-state.py" +FACTORY_CONFIG = REPO_ROOT / "factory" / "scripts" / "factory-config.py" + + +def sh(cwd: Path, *argv: str, stdin: str | None = None) -> subprocess.CompletedProcess: + return subprocess.run(argv, cwd=cwd, input=stdin, capture_output=True, text=True, check=False) + + +class StubPipelineTest(unittest.TestCase): + def setUp(self) -> None: + self.tmp = tempfile.TemporaryDirectory() + self.cwd = Path(self.tmp.name) + (self.cwd / ".dev" / "p").mkdir(parents=True) + + def tearDown(self) -> None: + self.tmp.cleanup() + + def rs(self, *args: str, stdin: str | None = None) -> subprocess.CompletedProcess: + return sh(self.cwd, sys.executable, str(RUN_STATE), *args, stdin=stdin) + + def state(self) -> dict: + return json.loads((self.cwd / ".dev" / "p" / "factory-run.json").read_text()) + + def result_path(self, phase: str, attempt: int = 1) -> Path: + return self.cwd / ".dev" / "p" / "results" / f"{phase}-{attempt}.json" + + def init(self, phases: str, *extra: str) -> None: + result = self.rs("init", "p", "--request", "r", "--base", "main", "--phases", phases, *extra) + self.assertEqual(result.returncode, 0, result.stdout) + + def run_stub(self, stub: str, phase: str, attempt: int = 1) -> Path: + path = self.result_path(phase, attempt) + sh(self.cwd, sys.executable, str(STUBS / stub), str(path)) + return path + + def launch(self, phase: str, type_: str = "stub") -> None: + result = self.rs("attempt", "p", phase, "--type", type_, "--skill", f"/stubs/{phase}") + self.assertEqual(result.returncode, 0, result.stdout) + + def advance(self, phase: str) -> None: + payload = json.dumps({"action": "advance", "phase": phase, "attempt": 1}) + self.assertEqual(self.rs("record", "p", stdin=payload).returncode, 0) + + def drive(self, phase: str, stub: str = "stub-done.py") -> Path: + """One launched phase, in the orchestrator's order: launch, run, check, close.""" + self.launch(phase) + path = self.run_stub(stub, phase) + self.assertEqual(self.rs("check-result", str(path)).returncode, 0) + closed = self.rs("attempt", "p", phase, "--status", "done", "--result", str(path)) + self.assertEqual(closed.returncode, 0, closed.stdout) + return path + + def test_each_conforming_stub_writes_a_file_check_result_accepts(self) -> None: + self.assertEqual(self.rs("check-result", str(self.run_stub("stub-done.py", "a"))).returncode, 0) + failed = self.rs("check-result", str(self.run_stub("stub-failed.py", "b"))) + self.assertEqual(failed.returncode, 1) + self.assertIn("stub failure", failed.stdout) + + def test_the_silent_stub_writes_nothing(self) -> None: + path = self.run_stub("stub-silent.py", "a") + self.assertFalse(path.exists()) + result = self.rs("check-result", str(path)) + self.assertEqual(result.returncode, 1) + self.assertIn("no result file", result.stdout) + + def test_three_stubs_are_recorded_in_order_with_type_path_and_timestamps(self) -> None: + self.init("one,two,three") + for phase in ("one", "two", "three"): + self.drive(phase) + self.advance("three") + state = self.state() + self.assertEqual(list(state["phases"]), ["one", "two", "three"]) + for phase in ("one", "two", "three"): + (attempt,) = state["phases"][phase] + self.assertEqual(attempt["status"], "done") + self.assertEqual((attempt["type"], attempt["skill"]), ("stub", f"/stubs/{phase}")) + self.assertIn("opened", attempt) + self.assertIn("closed", attempt) + self.assertIn("finished: yes", self.rs("show", "p").stdout) + + def test_a_silent_stub_closes_its_attempt_failed_with_no_result_file(self) -> None: + self.init("one") + self.launch("one") + path = self.run_stub("stub-silent.py", "one") + self.assertEqual(self.rs("check-result", str(path)).returncode, 1) + self.assertEqual(self.rs("attempt", "p", "one", "--status", "failed").returncode, 0) + record = self.state()["phases"]["one"][0] + self.assertEqual(record["status"], "failed") + self.assertEqual(record["result"]["reason"], "no result file") + + def test_an_attempt_left_launched_is_closed_by_the_orchestrator_and_the_next_is_numbered_n_plus_1( + self, + ) -> None: + self.init("one") + self.launch("one") + self.assertEqual(self.state()["phases"]["one"][0]["status"], "launched") + self.assertEqual(self.rs("attempt", "p", "one", "--status", "failed").returncode, 0) + self.assertEqual(self.state()["phases"]["one"][0]["result"]["reason"], "no result file") + again = self.rs("attempt", "p", "one") + self.assertIn("one attempt 2: launched", again.stdout) + + def test_an_interactive_first_phase_is_bracketed_like_a_launched_one(self) -> None: + self.init("interview,build") + self.launch("interview", "interview") + path = self.run_stub("stub-done.py", "interview") + closed = self.rs("attempt", "p", "interview", "--status", "done", "--result", str(path)) + self.assertEqual(closed.returncode, 0) + record = self.state()["phases"]["interview"][0] + self.assertEqual((record["type"], record["skill"]), ("interview", "/stubs/interview")) + self.assertIn("opened", record) + self.assertIn("closed", record) + self.assertIn("interview attempt 1: done", self.rs("show", "p").stdout) + + def test_a_ceiling_is_reported_exhausted_with_every_phases_attempts_never_refused(self) -> None: + self.init("one,two", "--ceiling", "2") + for phase in ("one", "two", "two"): + self.assertEqual(self.rs("attempt", "p", phase).returncode, 0) + shown = self.rs("show", "p").stdout + self.assertIn("total: 3/2 attempts (exhausted)", shown) + self.assertIn("one: 1/3 attempts", shown) + self.assertIn("two: 2/3 attempts", shown) + + def test_the_resolved_pipeline_from_factory_config_is_sealed_at_handoff(self) -> None: + shown = sh(self.cwd, sys.executable, str(FACTORY_CONFIG), "--repo-root", str(self.cwd), + "show", "--resolved", "--json") + self.assertEqual(shown.returncode, 0, shown.stderr) + (self.cwd / "pipeline.json").write_text(shown.stdout) + self.init("scope,scope-review,build,ship") + handoff = self.rs("handoff", "p", "--branch", "b", "--pipeline", "pipeline.json") + self.assertEqual(handoff.returncode, 0, handoff.stdout) + sealed = [phase["id"] for phase in self.state()["pipeline"]] + expected = [phase["id"] for phase in json.loads(shown.stdout)["phases"]] + self.assertEqual(sealed, expected) + + +if __name__ == "__main__": + unittest.main() diff --git a/factory/references/factory-run.md b/factory/references/factory-run.md index 36b0103..a2ffc79 100644 --- a/factory/references/factory-run.md +++ b/factory/references/factory-run.md @@ -6,10 +6,11 @@ Owned by the `run` skill. Every phase copy (`scope`, `scope-review`, `build`, `s `.dev/{plan}/factory-run.json`, written only by the `run` skill, via `run-state.py`. It exists from `init` on, before any branch. Fields: -- `plan`, `request`, `base` (the default branch at `init`). -- `branch`, `spec_sha256`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `dirty_files` - all written at `handoff`. -- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. The list is the attempt numbering: an attempt's number is its 1-based position, it is appended as `launched` before the phase is launched, and its result file is `results/{phase}-{that number}.json`. A trailing `launched` entry on resume therefore means a session died mid-phase, which is what tells that case apart from a phase never started (an empty list). -- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. The list's end is also the run's terminal marker: a run is finished when its last entry is an `end`, or an `advance` on `ship`. There is no separate done field. +- `schema` (currently `2`), `plan`, `request`, `base` (the default branch at `init`). A state file with no `schema` is read as the default four-phase pipeline, three attempts per phase and a ceiling of 12, so runs already in flight keep resuming. +- `branch`, `seals`, `sealed_nothing`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `config_path`, `config_sha256`, `pipeline`, `dirty_files` - all written at `handoff`. `seals` lists `{path, sha256, notation}` per sealed file: the paths come from the `seals` of the phases that ran before the go, and `notation` marks a file in the dev spec notation, whose scenario texts and `⊘` lines are what `diff-spec` compares, where any other sealed file is compared by sha256 alone. With no `--seal`, `seals` is empty and `sealed_nothing` is true, and `diff-spec` reports that nothing is watched. `pipeline` is the ordered `{id, type, skill}` list from `factory-config.py show --resolved --json`, and `diff-config` names the phase that was inserted, removed, reordered, retyped or repointed. +- `phases`: one key per phase of this run, in declared order (seeded at `init` from `--phases`, the built-in four when omitted), each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: , opened, closed, type, skill}`, where `opened` and `type` and `skill` are written when the attempt is launched (`attempt {plan} {phase} --type --skill `) and `closed` when it is closed. A `failed` close with no result file records the reason `no result file`. The list is the attempt numbering: an attempt's number is its 1-based position, it is appended as `launched` before the phase is launched, and its result file is `results/{phase}-{that number}.json`. A trailing `launched` entry on resume therefore means a session died mid-phase, which is what tells that case apart from a phase never started (an empty list). +- `budgets` (attempts allowed per phase) and `ceiling` (attempts allowed across the run), seeded at `init` from the resolved config. `show` reports each as exhausted once reached and `attempt` never refuses one: the numbers are prose the orchestrator follows. +- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. The list's end is also the run's terminal marker: a run is finished when its last entry is an `end`, or an `advance` on the last declared phase, which `show` prints as `finished: yes`. There is no separate done field. - `repairs`: an ordered list of `{phase, attempt, description, files: [], evidence: []}`. ## Result envelope: `factory.result/1` diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index fc6f954..8ab315e 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -40,7 +40,7 @@ Launch the copied `scope` skill inline, in this session: read its `SKILL.md` at ## 4. Handoff -At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan}` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). +At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan} --seal .dev/{plan}/spec.md` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). ## 5. Per phase: launch, judge, act diff --git a/factory/skills/run/scripts/run-state.py b/factory/skills/run/scripts/run-state.py index 9468d5d..f1a1164 100755 --- a/factory/skills/run/scripts/run-state.py +++ b/factory/skills/run/scripts/run-state.py @@ -1,21 +1,30 @@ #!/usr/bin/env python3 -"""Own factory-run.json: init, handoff, diff-spec, attempt, record, check-result, show. +"""Own factory-run.json: init, handoff, diff-spec, diff-config, attempt, record, check-result, show. Usage: run-state.py init --request --base - run-state.py handoff --branch [--spec ] [--dirty ...] - run-state.py diff-spec [--spec ] + [--phases a,b,c] [--attempts = ...] [--ceiling ] + run-state.py handoff --branch [--seal ...] [--dirty ...] + [--config ] [--pipeline ] + run-state.py diff-spec + run-state.py diff-config [--config ] [--pipeline ] run-state.py attempt [--status launched|done|failed|stopped] [--result ] + [--type --skill ] # type and skill only on a launch run-state.py record # reads one JSON decision object on stdin run-state.py check-result run-state.py show Exit codes, which the orchestrator branches on: - init 0 written; 1 a state file already exists - handoff 0 recorded; 1 no spec.md at the path; 3 unusable state file - diff-spec 0 no drift; 1 drift found; 3 unusable state file or no spec.md - attempt 0 recorded; 3 unusable state file, no launched attempt to close, - or an attempt already closed + init 0 written; 1 a state file already exists; 3 unusable --phases, + --attempts or --ceiling + handoff 0 recorded (sealing nothing when no --seal is given); 1 a sealed + file is missing; 3 unusable state file or --pipeline file + diff-spec 0 no drift, or nothing was sealed; 1 drift found; 3 unusable state file + diff-config 0 no drift, or no config file to watch; 1 a phase moved; 3 unusable + state file or --pipeline file + attempt 0 recorded; 3 unusable state file, a phase outside this run's list, + --type or --skill on a close, no launched attempt to close, or an + attempt already closed record 0 recorded; 1 action outside the four; 3 unreadable stdin JSON or unusable state file check-result 0 done; 1 failed, a missing or unparseable result file included; @@ -39,6 +48,7 @@ import re import subprocess import sys +from datetime import datetime, timezone from pathlib import Path from typing import NoReturn @@ -48,9 +58,13 @@ NOT_DOING = re.compile(r"^\s*(?:[-*]\s+)?⊘\s") ACTIONS = {"advance", "repair", "relaunch", "end"} -PHASES = ("scope", "scope-review", "build", "ship") +DEFAULT_PHASES = ("scope", "scope-review", "build", "ship") +DEFAULT_ATTEMPTS = 3 +DEFAULT_CEILING = 12 +SCHEMA = 2 ATTEMPT_STATUSES = ("launched", "done", "failed", "stopped") STATE_NAME = "factory-run.json" +DEFAULT_CONFIG = Path(".factory") / "config.yaml" BAD_CALL = 3 STATE_SHAPE = { "phases": dict, @@ -58,6 +72,9 @@ "not_doing_lines": list, "decisions": list, "repairs": list, + "seals": list, + "pipeline": list, + "budgets": dict, } @@ -65,8 +82,12 @@ def state_path(plan: str) -> Path: return Path(".dev") / plan / STATE_NAME -def spec_path(plan: str, override: str | None) -> Path: - return Path(override) if override else Path(".dev") / plan / "spec.md" +def spec_path(plan: str) -> Path: + return Path(".dev") / plan / "spec.md" + + +def now() -> str: + return datetime.now(timezone.utc).isoformat(timespec="seconds") def shape_problem(state: dict) -> str: @@ -85,6 +106,19 @@ def shape_problem(state: dict) -> str: return f"phases.{phase} is not a list of attempts" if any(not isinstance(attempt, dict) for attempt in attempts): return f"phases.{phase} holds an attempt that is not an object" + return budget_problem(state) + + +def is_count(value: object) -> bool: + return isinstance(value, int) and not isinstance(value, bool) + + +def budget_problem(state: dict) -> str: + for phase, budget in state.get("budgets", {}).items(): + if not is_count(budget): + return f"budgets.{phase} is not a number" + if "ceiling" in state and not is_count(state["ceiling"]): + return "ceiling is not a number" return "" @@ -169,22 +203,58 @@ def git_dirty_files() -> list[str]: return [line[3:].strip() for line in out.splitlines() if line.strip()] +def phase_ids(raw: str | None) -> list[str]: + """The ordered phase ids of a run: --phases, or the built-in four.""" + if raw is None: + return list(DEFAULT_PHASES) + ids = [part.strip() for part in raw.split(",")] + if not ids or any(not part for part in ids) or len(set(ids)) != len(ids): + raise ValueError(f"--phases {raw!r}: expected distinct, non-empty ids separated by commas") + return ids + + +def phase_budgets(phases: list[str], pairs: list[str] | None) -> dict[str, int]: + """Per-phase attempt budgets: the default, overridden by --attempts =.""" + budgets = {phase: DEFAULT_ATTEMPTS for phase in phases} + for pair in pairs or []: + phase, _, count = pair.partition("=") + if phase not in budgets or not count.isdigit() or int(count) < 1: + raise ValueError(f"--attempts {pair!r}: expected = for a phase in this run") + budgets[phase] = int(count) + return budgets + + def cmd_init(args: argparse.Namespace) -> int: path = state_path(args.plan) if path.exists(): print(f"factory-run.json already exists at {path}") return 1 + try: + phases = phase_ids(args.phases) + budgets = phase_budgets(phases, args.attempts) + except ValueError as exc: + print(f"bad call: {exc}") + return BAD_CALL + if args.ceiling is not None and args.ceiling < 1: + print(f"bad call: --ceiling {args.ceiling}: expected a positive number") + return BAD_CALL path.parent.mkdir(parents=True, exist_ok=True) state = { + "schema": SCHEMA, "plan": args.plan, "request": args.request, "base": args.base, "branch": None, - "spec_sha256": None, + "seals": [], "scenario_texts": {}, "not_doing_lines": [], + "config_path": None, + "config_sha256": None, + "pipeline": [], "dirty_files": [], - "phases": {"scope": [], "scope-review": [], "build": [], "ship": []}, + "phases": {phase: [] for phase in phases}, + "budgets": budgets, + "ceiling": args.ceiling if args.ceiling is not None else DEFAULT_CEILING, "decisions": [], "repairs": [], } @@ -193,23 +263,97 @@ def cmd_init(args: argparse.Namespace) -> int: return 0 +def sha256_of(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def has_dev_notation(lines: list[str]) -> bool: + """Whether a sealed file is a spec in the dev notation: a Change plan or a ⊘ line.""" + if not_doing_lines(lines): + return True + for line in lines: + heading = HEADING.match(line) + if heading and heading.group(1).lower() == "change plan": + return True + return False + + +def read_lines(path: Path) -> list[str]: + return path.read_text(encoding="utf-8", errors="replace").splitlines() + + +def merge_notation(paths: list[Path]) -> tuple[dict[str, list[str]], list[str]]: + """Scenario texts and ⊘ lines merged across every sealed spec in the dev notation.""" + texts: dict[str, list[str]] = {} + not_doing: list[str] = [] + for path in paths: + lines = read_lines(path) + for change_set, scenarios in scenario_texts(lines).items(): + texts.setdefault(change_set, []).extend(scenarios) + not_doing.extend(not_doing_lines(lines)) + return texts, not_doing + + +def seal_records(paths: list[Path]) -> list[dict]: + return [ + {"path": str(path), "sha256": sha256_of(path), "notation": has_dev_notation(read_lines(path))} + for path in paths + ] + + +def load_pipeline(path: str) -> list[dict] | None: + """The ids, types and skill paths of a `show --resolved --json` file, or None.""" + try: + data = json.loads(Path(path).read_text(encoding="utf-8")) + return [ + {"id": phase["id"], "type": phase["type"], "skill": phase["skill"]} + for phase in data["phases"] + ] + except (OSError, json.JSONDecodeError, KeyError, TypeError) as exc: + print(f"unusable pipeline file {path}: {exc!r}") + return None + + +def record_config_seal(state: dict, config: str | None, pipeline: str | None) -> bool: + """Seal the config hash and the resolved pipeline; False when --pipeline is unusable.""" + if pipeline is not None: + sealed = load_pipeline(pipeline) + if sealed is None: + return False + state["pipeline"] = sealed + path = Path(config) if config else DEFAULT_CONFIG + state["config_path"] = str(path) + state["config_sha256"] = sha256_of(path) if path.is_file() else None + return True + + def cmd_handoff(args: argparse.Namespace) -> int: - spec = spec_path(args.plan, args.spec) - if not spec.is_file(): - print(f"no spec.md at {spec}") - return 1 + paths = [Path(seal) for seal in args.seal or []] + for path in paths: + if not path.is_file(): + print(f"no sealed file at {path}") + return 1 state = load_state(args.plan) if state is None: return BAD_CALL - digest = hashlib.sha256(spec.read_bytes()).hexdigest() - texts, not_doing = scenario_texts_and_not_doing(spec) + if not record_config_seal(state, args.config, args.pipeline): + return BAD_CALL + records = seal_records(paths) + texts, not_doing = merge_notation([Path(r["path"]) for r in records if r["notation"]]) state["branch"] = args.branch - state["spec_sha256"] = digest + state["seals"] = records + state["sealed_nothing"] = not records state["scenario_texts"] = texts state["not_doing_lines"] = not_doing state["dirty_files"] = args.dirty if args.dirty else git_dirty_files() save_state(args.plan, state) - print(f"handoff recorded: {sum(len(v) for v in texts.values())} scenario(s), {len(not_doing)} not-doing line(s)") + if not records: + print("handoff recorded: sealed nothing") + else: + print( + f"handoff recorded: {len(records)} file(s) sealed, " + f"{sum(len(v) for v in texts.values())} scenario(s), {len(not_doing)} not-doing line(s)" + ) return 0 @@ -232,19 +376,124 @@ def not_doing_drift(stored: list[str], not_doing: list[str]) -> list[str]: return [f"⊘ line dropped: {line!r}" for line in stored if line not in not_doing] +def sealed_records(state: dict, plan: str) -> list[dict]: + """The sealed files of a run; an older state file sealed `.dev/{plan}/spec.md` alone.""" + if "seals" in state: + return [seal for seal in state["seals"] if isinstance(seal, dict)] + if state.get("spec_sha256"): + return [{"path": str(spec_path(plan)), "sha256": state["spec_sha256"], "notation": True}] + return [] + + +def hash_drift(records: list[dict]) -> list[str]: + """One message per sealed file without dev notation that is missing or changed.""" + problems = [] + for record in records: + path = Path(record.get("path", "")) + if record.get("notation"): + continue + if not path.is_file(): + problems.append(f"changed: sealed file is missing: {path}") + elif sha256_of(path) != record.get("sha256"): + problems.append(f"changed: {path} no longer matches its sealed sha256") + return problems + + +def notation_drift(state: dict, records: list[dict]) -> list[str]: + paths = [Path(r.get("path", "")) for r in records if r.get("notation")] + missing = [f"changed: sealed file is missing: {p}" for p in paths if not p.is_file()] + if missing or not paths: + return missing + texts, not_doing = merge_notation(paths) + problems = scenario_drift(state.get("scenario_texts", {}), texts) + return problems + not_doing_drift(state.get("not_doing_lines", []), not_doing) + + def cmd_diff_spec(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: return BAD_CALL - spec = spec_path(args.plan, args.spec) - if not spec.is_file(): - print(f"no spec.md at {spec}") + records = sealed_records(state, args.plan) + if not records: + print("not watched: nothing was sealed at handoff") + return 0 + problems = hash_drift(records) + notation_drift(state, records) + for message in problems: + print(message) + return 1 if problems else 0 + + +def pipeline_positions(pipeline: list[dict]) -> dict[str, int]: + return {phase.get("id"): index for index, phase in enumerate(pipeline, start=1)} + + +def moved_phases(sealed: list[dict], current: list[dict]) -> list[str]: + """Phases whose place among the phases both pipelines share changed.""" + now_at, then_at = pipeline_positions(current), pipeline_positions(sealed) + common_then = [p["id"] for p in sealed if p["id"] in now_at] + common_now = [p["id"] for p in current if p["id"] in then_at] + return [ + f"phase {pid} reordered: position {then_at[pid]} -> {now_at[pid]}" + for index, pid in enumerate(common_then) + if common_now.index(pid) != index + ] + + +def edited_phases(sealed: list[dict], current: list[dict]) -> list[str]: + """Phases present in both pipelines whose type or skill path changed.""" + before = {p["id"]: p for p in sealed} + problems = [] + for phase in current: + old = before.get(phase["id"]) + if old is None: + continue + if old["type"] != phase["type"]: + problems.append(f"phase {phase['id']} retyped: {old['type']} -> {phase['type']}") + if old["skill"] != phase["skill"]: + problems.append(f"phase {phase['id']} repointed: {old['skill']} -> {phase['skill']}") + return problems + + +def pipeline_drift(sealed: list[dict], current: list[dict]) -> list[str]: + then_ids = {p["id"] for p in sealed} + now_ids = {p["id"] for p in current} + problems = [f"phase {p['id']} removed" for p in sealed if p["id"] not in now_ids] + problems += [f"phase {p['id']} inserted" for p in current if p["id"] not in then_ids] + return problems + moved_phases(sealed, current) + edited_phases(sealed, current) + + +def config_drift(state: dict, config: Path, pipeline: str | None) -> tuple[list[str], str] | None: + """The drift messages and a one-line note, or None when --pipeline is unusable.""" + changed = sha256_of(config) != state.get("config_sha256") + if pipeline is None: + return ([f"config changed: {config} no longer matches its sealed sha256"] if changed else []), "" + current = load_pipeline(pipeline) + if current is None: + return None + sealed = state.get("pipeline", []) + if not sealed: + return [], "not watched: no pipeline was sealed at handoff" + problems = pipeline_drift(sealed, current) + note = "config text changed, resolved pipeline unchanged" if changed and not problems else "" + return problems, note + + +def cmd_diff_config(args: argparse.Namespace) -> int: + state = load_state(args.plan) + if state is None: + return BAD_CALL + config = Path(args.config) if args.config else DEFAULT_CONFIG + if not config.is_file(): + print("not watched: no config file, so the run uses the built-in default pipeline") + return 0 + result = config_drift(state, config, args.pipeline) + if result is None: return BAD_CALL - texts, not_doing = scenario_texts_and_not_doing(spec) - problems = scenario_drift(state.get("scenario_texts", {}), texts) - problems += not_doing_drift(state.get("not_doing_lines", []), not_doing) + problems, note = result for message in problems: print(message) + if note: + print(note) return 1 if problems else 0 @@ -353,7 +602,10 @@ def cmd_check_result(args: argparse.Namespace) -> int: def close_attempt(attempt: dict, status: str, result: str | None) -> None: """Close a launched attempt, recording an unusable result file as a failure.""" attempt["status"] = status + attempt["closed"] = now() if not result: + if status == "failed": + attempt["result"] = {"status": "failed", "reason": "no result file"} return data, reason = read_result(Path(result)) if data is None: @@ -364,13 +616,30 @@ def close_attempt(attempt: dict, status: str, result: str | None) -> None: attempt["result"] = data +def open_attempt(attempts: list[dict], args: argparse.Namespace) -> None: + attempt = {"status": "launched", "result": None, "opened": now()} + if args.type: + attempt["type"] = args.type + if args.skill: + attempt["skill"] = args.skill + attempts.append(attempt) + + def cmd_attempt(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: return BAD_CALL - attempts = state.setdefault("phases", {}).setdefault(args.phase, []) - if args.status == "launched": - attempts.append({"status": "launched", "result": None}) + launching = args.status in (None, "launched") + if not launching and (args.type or args.skill): + print("--type and --skill are only valid when launching an attempt") + return BAD_CALL + phases = state.setdefault("phases", {}) + if args.phase not in phases: + print(f"{args.phase!r} is not a phase of this run: {list(phases)}") + return BAD_CALL + attempts = phases[args.phase] + if launching: + open_attempt(attempts, args) elif not attempts: print(f"no launched attempt of {args.phase} to close") return BAD_CALL @@ -388,6 +657,33 @@ def cmd_attempt(args: argparse.Namespace) -> int: return 0 +def run_finished(state: dict) -> bool: + """Whether an advance was recorded on the last declared phase.""" + phases = list(state.get("phases", {})) + if not phases: + return False + return any( + isinstance(d, dict) and d.get("action") == "advance" and d.get("phase") == phases[-1] + for d in state.get("decisions", []) + ) + + +def budget_lines(state: dict) -> list[str]: + """Attempts against each phase's budget and the total against the ceiling.""" + budgets = state.get("budgets", {}) + lines = [] + total = 0 + for phase, attempts in state.get("phases", {}).items(): + budget = budgets.get(phase, DEFAULT_ATTEMPTS) + total += len(attempts) + mark = " (exhausted)" if len(attempts) >= budget else "" + lines.append(f"{phase}: {len(attempts)}/{budget} attempts{mark}") + ceiling = state.get("ceiling", DEFAULT_CEILING) + mark = " (exhausted)" if total >= ceiling else "" + lines.append(f"total: {total}/{ceiling} attempts{mark}") + return lines + + def cmd_show(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: @@ -398,8 +694,11 @@ def cmd_show(args: argparse.Namespace) -> int: continue for index, attempt in enumerate(attempts, start=1): print(f"{phase} attempt {index}: {attempt.get('status', 'unknown')}") + for line in budget_lines(state): + print(line) for repair in state.get("repairs", []): print(f"repair on {repair['phase']} attempt {repair['attempt']}: {repair['description']}") + print(f"finished: {'yes' if run_finished(state) else 'no'}") return 0 @@ -424,22 +723,33 @@ def build_parser() -> argparse.ArgumentParser: init_p.add_argument("plan") init_p.add_argument("--request", required=True) init_p.add_argument("--base", required=True) + init_p.add_argument("--phases") + init_p.add_argument("--attempts", action="append") + init_p.add_argument("--ceiling", type=int) handoff_p = sub.add_parser("handoff") handoff_p.add_argument("plan") handoff_p.add_argument("--branch", required=True) - handoff_p.add_argument("--spec") + handoff_p.add_argument("--seal", action="append") handoff_p.add_argument("--dirty", action="append") + handoff_p.add_argument("--config") + handoff_p.add_argument("--pipeline") diff_p = sub.add_parser("diff-spec") diff_p.add_argument("plan") - diff_p.add_argument("--spec") + + diff_config_p = sub.add_parser("diff-config") + diff_config_p.add_argument("plan") + diff_config_p.add_argument("--config") + diff_config_p.add_argument("--pipeline") attempt_p = sub.add_parser("attempt") attempt_p.add_argument("plan") - attempt_p.add_argument("phase", choices=PHASES) - attempt_p.add_argument("--status", choices=ATTEMPT_STATUSES, default="launched") + attempt_p.add_argument("phase") + attempt_p.add_argument("--status", choices=ATTEMPT_STATUSES) attempt_p.add_argument("--result") + attempt_p.add_argument("--type") + attempt_p.add_argument("--skill") record_p = sub.add_parser("record") record_p.add_argument("plan") @@ -459,6 +769,7 @@ def main(argv: list[str]) -> int: "init": cmd_init, "handoff": cmd_handoff, "diff-spec": cmd_diff_spec, + "diff-config": cmd_diff_config, "attempt": cmd_attempt, "record": cmd_record, "check-result": cmd_check_result, diff --git a/plugins/factory/references/factory-run.md b/plugins/factory/references/factory-run.md index 36b0103..a2ffc79 100644 --- a/plugins/factory/references/factory-run.md +++ b/plugins/factory/references/factory-run.md @@ -6,10 +6,11 @@ Owned by the `run` skill. Every phase copy (`scope`, `scope-review`, `build`, `s `.dev/{plan}/factory-run.json`, written only by the `run` skill, via `run-state.py`. It exists from `init` on, before any branch. Fields: -- `plan`, `request`, `base` (the default branch at `init`). -- `branch`, `spec_sha256`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `dirty_files` - all written at `handoff`. -- `phases`: `{scope, scope-review, build, ship}`, each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: }`. The list is the attempt numbering: an attempt's number is its 1-based position, it is appended as `launched` before the phase is launched, and its result file is `results/{phase}-{that number}.json`. A trailing `launched` entry on resume therefore means a session died mid-phase, which is what tells that case apart from a phase never started (an empty list). -- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. The list's end is also the run's terminal marker: a run is finished when its last entry is an `end`, or an `advance` on `ship`. There is no separate done field. +- `schema` (currently `2`), `plan`, `request`, `base` (the default branch at `init`). A state file with no `schema` is read as the default four-phase pipeline, three attempts per phase and a ceiling of 12, so runs already in flight keep resuming. +- `branch`, `seals`, `sealed_nothing`, `scenario_texts` (per change set), `not_doing_lines` (the `⊘` lines), `config_path`, `config_sha256`, `pipeline`, `dirty_files` - all written at `handoff`. `seals` lists `{path, sha256, notation}` per sealed file: the paths come from the `seals` of the phases that ran before the go, and `notation` marks a file in the dev spec notation, whose scenario texts and `⊘` lines are what `diff-spec` compares, where any other sealed file is compared by sha256 alone. With no `--seal`, `seals` is empty and `sealed_nothing` is true, and `diff-spec` reports that nothing is watched. `pipeline` is the ordered `{id, type, skill}` list from `factory-config.py show --resolved --json`, and `diff-config` names the phase that was inserted, removed, reordered, retyped or repointed. +- `phases`: one key per phase of this run, in declared order (seeded at `init` from `--phases`, the built-in four when omitted), each an ordered list of attempts. Each attempt: `{status: launched | done | failed | stopped, result: , opened, closed, type, skill}`, where `opened` and `type` and `skill` are written when the attempt is launched (`attempt {plan} {phase} --type --skill `) and `closed` when it is closed. A `failed` close with no result file records the reason `no result file`. The list is the attempt numbering: an attempt's number is its 1-based position, it is appended as `launched` before the phase is launched, and its result file is `results/{phase}-{that number}.json`. A trailing `launched` entry on resume therefore means a session died mid-phase, which is what tells that case apart from a phase never started (an empty list). +- `budgets` (attempts allowed per phase) and `ceiling` (attempts allowed across the run), seeded at `init` from the resolved config. `show` reports each as exhausted once reached and `attempt` never refuses one: the numbers are prose the orchestrator follows. +- `decisions`: an ordered list of `{phase, attempt, action, rationale, evidence: []}` with `action` in `advance`, `repair`, `relaunch`, `end`. The list's end is also the run's terminal marker: a run is finished when its last entry is an `end`, or an `advance` on the last declared phase, which `show` prints as `finished: yes`. There is no separate done field. - `repairs`: an ordered list of `{phase, attempt, description, files: [], evidence: []}`. ## Result envelope: `factory.result/1` diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index 575758c..a5fd108 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -39,7 +39,7 @@ Launch the copied `scope` skill inline, in this session: read its `SKILL.md` at ## 4. Handoff -At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan}` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). +At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan} --seal .dev/{plan}/spec.md` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). ## 5. Per phase: launch, judge, act diff --git a/plugins/factory/skills/run/scripts/run-state.py b/plugins/factory/skills/run/scripts/run-state.py index 9468d5d..f1a1164 100755 --- a/plugins/factory/skills/run/scripts/run-state.py +++ b/plugins/factory/skills/run/scripts/run-state.py @@ -1,21 +1,30 @@ #!/usr/bin/env python3 -"""Own factory-run.json: init, handoff, diff-spec, attempt, record, check-result, show. +"""Own factory-run.json: init, handoff, diff-spec, diff-config, attempt, record, check-result, show. Usage: run-state.py init --request --base - run-state.py handoff --branch [--spec ] [--dirty ...] - run-state.py diff-spec [--spec ] + [--phases a,b,c] [--attempts = ...] [--ceiling ] + run-state.py handoff --branch [--seal ...] [--dirty ...] + [--config ] [--pipeline ] + run-state.py diff-spec + run-state.py diff-config [--config ] [--pipeline ] run-state.py attempt [--status launched|done|failed|stopped] [--result ] + [--type --skill ] # type and skill only on a launch run-state.py record # reads one JSON decision object on stdin run-state.py check-result run-state.py show Exit codes, which the orchestrator branches on: - init 0 written; 1 a state file already exists - handoff 0 recorded; 1 no spec.md at the path; 3 unusable state file - diff-spec 0 no drift; 1 drift found; 3 unusable state file or no spec.md - attempt 0 recorded; 3 unusable state file, no launched attempt to close, - or an attempt already closed + init 0 written; 1 a state file already exists; 3 unusable --phases, + --attempts or --ceiling + handoff 0 recorded (sealing nothing when no --seal is given); 1 a sealed + file is missing; 3 unusable state file or --pipeline file + diff-spec 0 no drift, or nothing was sealed; 1 drift found; 3 unusable state file + diff-config 0 no drift, or no config file to watch; 1 a phase moved; 3 unusable + state file or --pipeline file + attempt 0 recorded; 3 unusable state file, a phase outside this run's list, + --type or --skill on a close, no launched attempt to close, or an + attempt already closed record 0 recorded; 1 action outside the four; 3 unreadable stdin JSON or unusable state file check-result 0 done; 1 failed, a missing or unparseable result file included; @@ -39,6 +48,7 @@ import re import subprocess import sys +from datetime import datetime, timezone from pathlib import Path from typing import NoReturn @@ -48,9 +58,13 @@ NOT_DOING = re.compile(r"^\s*(?:[-*]\s+)?⊘\s") ACTIONS = {"advance", "repair", "relaunch", "end"} -PHASES = ("scope", "scope-review", "build", "ship") +DEFAULT_PHASES = ("scope", "scope-review", "build", "ship") +DEFAULT_ATTEMPTS = 3 +DEFAULT_CEILING = 12 +SCHEMA = 2 ATTEMPT_STATUSES = ("launched", "done", "failed", "stopped") STATE_NAME = "factory-run.json" +DEFAULT_CONFIG = Path(".factory") / "config.yaml" BAD_CALL = 3 STATE_SHAPE = { "phases": dict, @@ -58,6 +72,9 @@ "not_doing_lines": list, "decisions": list, "repairs": list, + "seals": list, + "pipeline": list, + "budgets": dict, } @@ -65,8 +82,12 @@ def state_path(plan: str) -> Path: return Path(".dev") / plan / STATE_NAME -def spec_path(plan: str, override: str | None) -> Path: - return Path(override) if override else Path(".dev") / plan / "spec.md" +def spec_path(plan: str) -> Path: + return Path(".dev") / plan / "spec.md" + + +def now() -> str: + return datetime.now(timezone.utc).isoformat(timespec="seconds") def shape_problem(state: dict) -> str: @@ -85,6 +106,19 @@ def shape_problem(state: dict) -> str: return f"phases.{phase} is not a list of attempts" if any(not isinstance(attempt, dict) for attempt in attempts): return f"phases.{phase} holds an attempt that is not an object" + return budget_problem(state) + + +def is_count(value: object) -> bool: + return isinstance(value, int) and not isinstance(value, bool) + + +def budget_problem(state: dict) -> str: + for phase, budget in state.get("budgets", {}).items(): + if not is_count(budget): + return f"budgets.{phase} is not a number" + if "ceiling" in state and not is_count(state["ceiling"]): + return "ceiling is not a number" return "" @@ -169,22 +203,58 @@ def git_dirty_files() -> list[str]: return [line[3:].strip() for line in out.splitlines() if line.strip()] +def phase_ids(raw: str | None) -> list[str]: + """The ordered phase ids of a run: --phases, or the built-in four.""" + if raw is None: + return list(DEFAULT_PHASES) + ids = [part.strip() for part in raw.split(",")] + if not ids or any(not part for part in ids) or len(set(ids)) != len(ids): + raise ValueError(f"--phases {raw!r}: expected distinct, non-empty ids separated by commas") + return ids + + +def phase_budgets(phases: list[str], pairs: list[str] | None) -> dict[str, int]: + """Per-phase attempt budgets: the default, overridden by --attempts =.""" + budgets = {phase: DEFAULT_ATTEMPTS for phase in phases} + for pair in pairs or []: + phase, _, count = pair.partition("=") + if phase not in budgets or not count.isdigit() or int(count) < 1: + raise ValueError(f"--attempts {pair!r}: expected = for a phase in this run") + budgets[phase] = int(count) + return budgets + + def cmd_init(args: argparse.Namespace) -> int: path = state_path(args.plan) if path.exists(): print(f"factory-run.json already exists at {path}") return 1 + try: + phases = phase_ids(args.phases) + budgets = phase_budgets(phases, args.attempts) + except ValueError as exc: + print(f"bad call: {exc}") + return BAD_CALL + if args.ceiling is not None and args.ceiling < 1: + print(f"bad call: --ceiling {args.ceiling}: expected a positive number") + return BAD_CALL path.parent.mkdir(parents=True, exist_ok=True) state = { + "schema": SCHEMA, "plan": args.plan, "request": args.request, "base": args.base, "branch": None, - "spec_sha256": None, + "seals": [], "scenario_texts": {}, "not_doing_lines": [], + "config_path": None, + "config_sha256": None, + "pipeline": [], "dirty_files": [], - "phases": {"scope": [], "scope-review": [], "build": [], "ship": []}, + "phases": {phase: [] for phase in phases}, + "budgets": budgets, + "ceiling": args.ceiling if args.ceiling is not None else DEFAULT_CEILING, "decisions": [], "repairs": [], } @@ -193,23 +263,97 @@ def cmd_init(args: argparse.Namespace) -> int: return 0 +def sha256_of(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def has_dev_notation(lines: list[str]) -> bool: + """Whether a sealed file is a spec in the dev notation: a Change plan or a ⊘ line.""" + if not_doing_lines(lines): + return True + for line in lines: + heading = HEADING.match(line) + if heading and heading.group(1).lower() == "change plan": + return True + return False + + +def read_lines(path: Path) -> list[str]: + return path.read_text(encoding="utf-8", errors="replace").splitlines() + + +def merge_notation(paths: list[Path]) -> tuple[dict[str, list[str]], list[str]]: + """Scenario texts and ⊘ lines merged across every sealed spec in the dev notation.""" + texts: dict[str, list[str]] = {} + not_doing: list[str] = [] + for path in paths: + lines = read_lines(path) + for change_set, scenarios in scenario_texts(lines).items(): + texts.setdefault(change_set, []).extend(scenarios) + not_doing.extend(not_doing_lines(lines)) + return texts, not_doing + + +def seal_records(paths: list[Path]) -> list[dict]: + return [ + {"path": str(path), "sha256": sha256_of(path), "notation": has_dev_notation(read_lines(path))} + for path in paths + ] + + +def load_pipeline(path: str) -> list[dict] | None: + """The ids, types and skill paths of a `show --resolved --json` file, or None.""" + try: + data = json.loads(Path(path).read_text(encoding="utf-8")) + return [ + {"id": phase["id"], "type": phase["type"], "skill": phase["skill"]} + for phase in data["phases"] + ] + except (OSError, json.JSONDecodeError, KeyError, TypeError) as exc: + print(f"unusable pipeline file {path}: {exc!r}") + return None + + +def record_config_seal(state: dict, config: str | None, pipeline: str | None) -> bool: + """Seal the config hash and the resolved pipeline; False when --pipeline is unusable.""" + if pipeline is not None: + sealed = load_pipeline(pipeline) + if sealed is None: + return False + state["pipeline"] = sealed + path = Path(config) if config else DEFAULT_CONFIG + state["config_path"] = str(path) + state["config_sha256"] = sha256_of(path) if path.is_file() else None + return True + + def cmd_handoff(args: argparse.Namespace) -> int: - spec = spec_path(args.plan, args.spec) - if not spec.is_file(): - print(f"no spec.md at {spec}") - return 1 + paths = [Path(seal) for seal in args.seal or []] + for path in paths: + if not path.is_file(): + print(f"no sealed file at {path}") + return 1 state = load_state(args.plan) if state is None: return BAD_CALL - digest = hashlib.sha256(spec.read_bytes()).hexdigest() - texts, not_doing = scenario_texts_and_not_doing(spec) + if not record_config_seal(state, args.config, args.pipeline): + return BAD_CALL + records = seal_records(paths) + texts, not_doing = merge_notation([Path(r["path"]) for r in records if r["notation"]]) state["branch"] = args.branch - state["spec_sha256"] = digest + state["seals"] = records + state["sealed_nothing"] = not records state["scenario_texts"] = texts state["not_doing_lines"] = not_doing state["dirty_files"] = args.dirty if args.dirty else git_dirty_files() save_state(args.plan, state) - print(f"handoff recorded: {sum(len(v) for v in texts.values())} scenario(s), {len(not_doing)} not-doing line(s)") + if not records: + print("handoff recorded: sealed nothing") + else: + print( + f"handoff recorded: {len(records)} file(s) sealed, " + f"{sum(len(v) for v in texts.values())} scenario(s), {len(not_doing)} not-doing line(s)" + ) return 0 @@ -232,19 +376,124 @@ def not_doing_drift(stored: list[str], not_doing: list[str]) -> list[str]: return [f"⊘ line dropped: {line!r}" for line in stored if line not in not_doing] +def sealed_records(state: dict, plan: str) -> list[dict]: + """The sealed files of a run; an older state file sealed `.dev/{plan}/spec.md` alone.""" + if "seals" in state: + return [seal for seal in state["seals"] if isinstance(seal, dict)] + if state.get("spec_sha256"): + return [{"path": str(spec_path(plan)), "sha256": state["spec_sha256"], "notation": True}] + return [] + + +def hash_drift(records: list[dict]) -> list[str]: + """One message per sealed file without dev notation that is missing or changed.""" + problems = [] + for record in records: + path = Path(record.get("path", "")) + if record.get("notation"): + continue + if not path.is_file(): + problems.append(f"changed: sealed file is missing: {path}") + elif sha256_of(path) != record.get("sha256"): + problems.append(f"changed: {path} no longer matches its sealed sha256") + return problems + + +def notation_drift(state: dict, records: list[dict]) -> list[str]: + paths = [Path(r.get("path", "")) for r in records if r.get("notation")] + missing = [f"changed: sealed file is missing: {p}" for p in paths if not p.is_file()] + if missing or not paths: + return missing + texts, not_doing = merge_notation(paths) + problems = scenario_drift(state.get("scenario_texts", {}), texts) + return problems + not_doing_drift(state.get("not_doing_lines", []), not_doing) + + def cmd_diff_spec(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: return BAD_CALL - spec = spec_path(args.plan, args.spec) - if not spec.is_file(): - print(f"no spec.md at {spec}") + records = sealed_records(state, args.plan) + if not records: + print("not watched: nothing was sealed at handoff") + return 0 + problems = hash_drift(records) + notation_drift(state, records) + for message in problems: + print(message) + return 1 if problems else 0 + + +def pipeline_positions(pipeline: list[dict]) -> dict[str, int]: + return {phase.get("id"): index for index, phase in enumerate(pipeline, start=1)} + + +def moved_phases(sealed: list[dict], current: list[dict]) -> list[str]: + """Phases whose place among the phases both pipelines share changed.""" + now_at, then_at = pipeline_positions(current), pipeline_positions(sealed) + common_then = [p["id"] for p in sealed if p["id"] in now_at] + common_now = [p["id"] for p in current if p["id"] in then_at] + return [ + f"phase {pid} reordered: position {then_at[pid]} -> {now_at[pid]}" + for index, pid in enumerate(common_then) + if common_now.index(pid) != index + ] + + +def edited_phases(sealed: list[dict], current: list[dict]) -> list[str]: + """Phases present in both pipelines whose type or skill path changed.""" + before = {p["id"]: p for p in sealed} + problems = [] + for phase in current: + old = before.get(phase["id"]) + if old is None: + continue + if old["type"] != phase["type"]: + problems.append(f"phase {phase['id']} retyped: {old['type']} -> {phase['type']}") + if old["skill"] != phase["skill"]: + problems.append(f"phase {phase['id']} repointed: {old['skill']} -> {phase['skill']}") + return problems + + +def pipeline_drift(sealed: list[dict], current: list[dict]) -> list[str]: + then_ids = {p["id"] for p in sealed} + now_ids = {p["id"] for p in current} + problems = [f"phase {p['id']} removed" for p in sealed if p["id"] not in now_ids] + problems += [f"phase {p['id']} inserted" for p in current if p["id"] not in then_ids] + return problems + moved_phases(sealed, current) + edited_phases(sealed, current) + + +def config_drift(state: dict, config: Path, pipeline: str | None) -> tuple[list[str], str] | None: + """The drift messages and a one-line note, or None when --pipeline is unusable.""" + changed = sha256_of(config) != state.get("config_sha256") + if pipeline is None: + return ([f"config changed: {config} no longer matches its sealed sha256"] if changed else []), "" + current = load_pipeline(pipeline) + if current is None: + return None + sealed = state.get("pipeline", []) + if not sealed: + return [], "not watched: no pipeline was sealed at handoff" + problems = pipeline_drift(sealed, current) + note = "config text changed, resolved pipeline unchanged" if changed and not problems else "" + return problems, note + + +def cmd_diff_config(args: argparse.Namespace) -> int: + state = load_state(args.plan) + if state is None: + return BAD_CALL + config = Path(args.config) if args.config else DEFAULT_CONFIG + if not config.is_file(): + print("not watched: no config file, so the run uses the built-in default pipeline") + return 0 + result = config_drift(state, config, args.pipeline) + if result is None: return BAD_CALL - texts, not_doing = scenario_texts_and_not_doing(spec) - problems = scenario_drift(state.get("scenario_texts", {}), texts) - problems += not_doing_drift(state.get("not_doing_lines", []), not_doing) + problems, note = result for message in problems: print(message) + if note: + print(note) return 1 if problems else 0 @@ -353,7 +602,10 @@ def cmd_check_result(args: argparse.Namespace) -> int: def close_attempt(attempt: dict, status: str, result: str | None) -> None: """Close a launched attempt, recording an unusable result file as a failure.""" attempt["status"] = status + attempt["closed"] = now() if not result: + if status == "failed": + attempt["result"] = {"status": "failed", "reason": "no result file"} return data, reason = read_result(Path(result)) if data is None: @@ -364,13 +616,30 @@ def close_attempt(attempt: dict, status: str, result: str | None) -> None: attempt["result"] = data +def open_attempt(attempts: list[dict], args: argparse.Namespace) -> None: + attempt = {"status": "launched", "result": None, "opened": now()} + if args.type: + attempt["type"] = args.type + if args.skill: + attempt["skill"] = args.skill + attempts.append(attempt) + + def cmd_attempt(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: return BAD_CALL - attempts = state.setdefault("phases", {}).setdefault(args.phase, []) - if args.status == "launched": - attempts.append({"status": "launched", "result": None}) + launching = args.status in (None, "launched") + if not launching and (args.type or args.skill): + print("--type and --skill are only valid when launching an attempt") + return BAD_CALL + phases = state.setdefault("phases", {}) + if args.phase not in phases: + print(f"{args.phase!r} is not a phase of this run: {list(phases)}") + return BAD_CALL + attempts = phases[args.phase] + if launching: + open_attempt(attempts, args) elif not attempts: print(f"no launched attempt of {args.phase} to close") return BAD_CALL @@ -388,6 +657,33 @@ def cmd_attempt(args: argparse.Namespace) -> int: return 0 +def run_finished(state: dict) -> bool: + """Whether an advance was recorded on the last declared phase.""" + phases = list(state.get("phases", {})) + if not phases: + return False + return any( + isinstance(d, dict) and d.get("action") == "advance" and d.get("phase") == phases[-1] + for d in state.get("decisions", []) + ) + + +def budget_lines(state: dict) -> list[str]: + """Attempts against each phase's budget and the total against the ceiling.""" + budgets = state.get("budgets", {}) + lines = [] + total = 0 + for phase, attempts in state.get("phases", {}).items(): + budget = budgets.get(phase, DEFAULT_ATTEMPTS) + total += len(attempts) + mark = " (exhausted)" if len(attempts) >= budget else "" + lines.append(f"{phase}: {len(attempts)}/{budget} attempts{mark}") + ceiling = state.get("ceiling", DEFAULT_CEILING) + mark = " (exhausted)" if total >= ceiling else "" + lines.append(f"total: {total}/{ceiling} attempts{mark}") + return lines + + def cmd_show(args: argparse.Namespace) -> int: state = load_state(args.plan) if state is None: @@ -398,8 +694,11 @@ def cmd_show(args: argparse.Namespace) -> int: continue for index, attempt in enumerate(attempts, start=1): print(f"{phase} attempt {index}: {attempt.get('status', 'unknown')}") + for line in budget_lines(state): + print(line) for repair in state.get("repairs", []): print(f"repair on {repair['phase']} attempt {repair['attempt']}: {repair['description']}") + print(f"finished: {'yes' if run_finished(state) else 'no'}") return 0 @@ -424,22 +723,33 @@ def build_parser() -> argparse.ArgumentParser: init_p.add_argument("plan") init_p.add_argument("--request", required=True) init_p.add_argument("--base", required=True) + init_p.add_argument("--phases") + init_p.add_argument("--attempts", action="append") + init_p.add_argument("--ceiling", type=int) handoff_p = sub.add_parser("handoff") handoff_p.add_argument("plan") handoff_p.add_argument("--branch", required=True) - handoff_p.add_argument("--spec") + handoff_p.add_argument("--seal", action="append") handoff_p.add_argument("--dirty", action="append") + handoff_p.add_argument("--config") + handoff_p.add_argument("--pipeline") diff_p = sub.add_parser("diff-spec") diff_p.add_argument("plan") - diff_p.add_argument("--spec") + + diff_config_p = sub.add_parser("diff-config") + diff_config_p.add_argument("plan") + diff_config_p.add_argument("--config") + diff_config_p.add_argument("--pipeline") attempt_p = sub.add_parser("attempt") attempt_p.add_argument("plan") - attempt_p.add_argument("phase", choices=PHASES) - attempt_p.add_argument("--status", choices=ATTEMPT_STATUSES, default="launched") + attempt_p.add_argument("phase") + attempt_p.add_argument("--status", choices=ATTEMPT_STATUSES) attempt_p.add_argument("--result") + attempt_p.add_argument("--type") + attempt_p.add_argument("--skill") record_p = sub.add_parser("record") record_p.add_argument("plan") @@ -459,6 +769,7 @@ def main(argv: list[str]) -> int: "init": cmd_init, "handoff": cmd_handoff, "diff-spec": cmd_diff_spec, + "diff-config": cmd_diff_config, "attempt": cmd_attempt, "record": cmd_record, "check-result": cmd_check_result, From 2d2d925a40525d03e3060db9356c95f0a9482c3a Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 09:13:29 +0200 Subject: [PATCH 21/30] feat(factory): the orchestrator drives a declared pipeline Change set 4 of config-driven-factory/spec.md: run/SKILL.md, judgment.md and launch.md read phases, types, checks and budgets from factory-config.py instead of naming four phases. Built-in check executables become real plugin-root script paths and check now requires them to exist. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/evals/tests/test_factory_config.py | 37 +++++++++++++++++ factory/references/factory-run.md | 6 +-- factory/references/pipeline-config.md | 2 +- factory/scripts/factory-config.py | 41 +++++++++++++------ factory/skills/run/SKILL.md | 37 ++++++++++------- factory/skills/run/references/judgment.md | 25 +++++------ factory/skills/run/references/launch.md | 21 ++++++---- plugins/factory/references/factory-run.md | 6 +-- plugins/factory/references/pipeline-config.md | 2 +- plugins/factory/scripts/factory-config.py | 41 +++++++++++++------ plugins/factory/skills/run/SKILL.md | 37 ++++++++++------- .../factory/skills/run/references/judgment.md | 25 +++++------ .../factory/skills/run/references/launch.md | 21 ++++++---- 13 files changed, 199 insertions(+), 102 deletions(-) diff --git a/factory/evals/tests/test_factory_config.py b/factory/evals/tests/test_factory_config.py index 8fbb404..390426d 100644 --- a/factory/evals/tests/test_factory_config.py +++ b/factory/evals/tests/test_factory_config.py @@ -186,6 +186,29 @@ def test_check_executable_with_plan_dir(self) -> None: self.assertEqual(result.returncode, 1) self.assertIn("${plan_dir}", result.stdout) + def test_missing_skill_file_and_missing_executable_do_not_resolve(self) -> None: + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {}}, + "phases": [ + { + "id": "a", + "type": "t", + "skill": "${repo_root}/nope/SKILL.md", + "checks": [["${repo_root}/nope.py"]], + "unattended_safe": True, + } + ], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("nope/SKILL.md", result.stdout) + self.assertIn("nope.py", result.stdout) + self.assertIn("does not resolve", result.stdout) + def test_reserved_phase_id(self) -> None: make_skill(self.repo, "x/SKILL.md") write_json_doc( @@ -229,6 +252,8 @@ def test_requires_with_colon_in_double_quotes(self) -> None: def test_nested_checks_both_lists_shown(self) -> None: make_skill(self.repo, "x/SKILL.md") + make_skill(self.repo, "lint-spec.py") + make_skill(self.repo, "check-tests.py") write_json_doc( self.repo, { @@ -395,6 +420,8 @@ def test_set_defaults_attempts_reflected_in_show_resolved(self) -> None: def test_set_types_checks_with_items(self) -> None: make_skill(self.repo, "x/SKILL.md") + make_skill(self.repo, "lint-spec.py") + make_skill(self.repo, "check-tests.py") write_json_doc( self.repo, { @@ -524,6 +551,8 @@ def test_relative_path_escaping_repo_root(self) -> None: def test_phase_omitting_checks_inherits_type_checks(self) -> None: make_skill(self.repo, "x/SKILL.md") + make_skill(self.repo, "lint-spec.py") + make_skill(self.repo, "check-tests.py") write_json_doc( self.repo, { @@ -537,6 +566,8 @@ def test_phase_omitting_checks_inherits_type_checks(self) -> None: def test_phase_checks_replace_type_checks(self) -> None: make_skill(self.repo, "x/SKILL.md") + make_skill(self.repo, "lint-spec.py") + make_skill(self.repo, "check-tests.py") write_json_doc( self.repo, { @@ -561,6 +592,12 @@ def test_builtin_pipeline_exactly_three_checks_and_requires_prose(self) -> None: result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") self.assertEqual(result.returncode, 0, result.stdout) self.assertEqual(result.stdout.count("check:"), 3) + plugin = str(REPO_ROOT / "factory") + for line in result.stdout.splitlines(): + if line.strip().startswith("check:"): + executable = eval(line.split("check:", 1)[1])[0] # noqa: S307 - our own printed list + self.assertTrue(executable.startswith(plugin + "/"), executable) + self.assertTrue(Path(executable).is_file(), executable) self.assertIn("gh pr view", result.stdout) self.assertIn("git ls-remote", result.stdout) self.assertIn("Verdict: APPROVED", result.stdout) diff --git a/factory/references/factory-run.md b/factory/references/factory-run.md index a2ffc79..43cf943 100644 --- a/factory/references/factory-run.md +++ b/factory/references/factory-run.md @@ -45,16 +45,16 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las After the go, no phase skill asks a person anything. An escalation that dev's version would put outside the run is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. -`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. `run/SKILL.md` is the only file in the scan root that may use `` / ``, for its pre-go lines; the markers must balance and each must stand alone on its line. No phase copy may use them, because no phase copy has anything to route. +The interactive phases - `scope` in the built-in pipeline - are the one exception: they run inline in the orchestrator's own session, keep their interview, and are outside `check_factory_unattended`'s scan root. A phase body the factory did not write is covered by no check: its author declares it `unattended_safe` in the config, and a phase that blocks on a person is neither prevented nor detected. `run/SKILL.md` is the only file in the scan root that may use `` / ``, for its pre-go lines; the markers must balance and each must stand alone on its line. No phase copy may use them, because no phase copy has anything to route. ## Launch prompt shape Each phase subagent's prompt carries, in order: -1. The phase's skill path: the absolute path to `factory/skills/{phase}/SKILL.md`, with the instruction to read `{phase}-skill-root` as that path's parent directory (a subagent reading a file cannot resolve `{phase}-skill-root}`-style placeholders on its own). +1. The phase's skill path: the absolute resolved `skill` path from `factory-config.py show --resolved`, with the instruction to read that path's parent directory wherever the file uses a skill-root placeholder (a subagent reading a file cannot resolve `{phase}-skill-root`-style placeholders on its own). Any skill can be wrapped this way, ours or a team's own. 2. The plan name and the plan directory's absolute path. 3. The attempt number, which is the one `run-state.py attempt` printed when it recorded this launch, never a number counted by hand or read off a file in `results/`. 4. On a relaunch: the previous attempt's reason and explicit guidance naming what it did and what is required instead - never "try again". 5. The scratch root for this attempt: `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/`. 6. The result path this attempt must write: `.dev/{plan}/results/{phase}-{attempt}.json`. -7. The instruction that this phase never launches another phase and, after the go, never asks a person anything - decide under the unattended policy above and record it. +7. The instruction that this phase never names or launches the next phase and, after the go, never asks a person anything - decide under the unattended policy above and record it. `launch.md` carries the exact wording. The wrapper imposes the result shape on a phase; it does not guarantee it, and a phase that writes no result is read as a failed attempt. diff --git a/factory/references/pipeline-config.md b/factory/references/pipeline-config.md index d72662b..2b75602 100644 --- a/factory/references/pipeline-config.md +++ b/factory/references/pipeline-config.md @@ -85,4 +85,4 @@ A list-valued key takes repeated `--item` flags in place of a positional value. ## The built-in default pipeline -Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check` (ship) become `checks` entries; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. +Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/skills/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. diff --git a/factory/scripts/factory-config.py b/factory/scripts/factory-config.py index 98bdc78..a8caed2 100755 --- a/factory/scripts/factory-config.py +++ b/factory/scripts/factory-config.py @@ -56,21 +56,22 @@ "interactive": True, "requires": "the go was recorded (the interview happened inline in this session, " "so the orchestrator already knows)", - "checks": [["lint-spec.py", "${plan_dir}/spec.md"]], + "checks": [["${plugin_root}/skills/scope/scripts/lint-spec.py", "${plan_dir}/spec.md"]], "seals": ["${plan_dir}/spec.md"], "attempts": 3, }, "review": { "interactive": False, "requires": 'spec-review_N.md contains a line whose whole text is exactly ' - '"Verdict: APPROVED" - nothing after it - with a "Rounds:" line', + '"Verdict: APPROVED" - nothing after it - with a "Rounds:" line, and the spec ' + 'still lints clean with lint-spec.py', "attempts": 3, }, "implement": { "interactive": False, "requires": "the spec's Validation block is green, and commits since handoff " "match the change plan's numbering", - "checks": [["check-tests.py", "${plan_dir}"]], + "checks": [["${plugin_root}/skills/build/scripts/check-tests.py", "${plan_dir}"]], "attempts": 3, }, "ship": { @@ -78,7 +79,7 @@ "requires": "review_N.md carries a verdict for HEAD and gh pr view shows the PR " "open with head equal to HEAD, or (without a GitHub remote) git ls-remote origin " "shows the branch at HEAD and pr.md is written", - "checks": [["pr-evidence.py", "check"]], + "checks": [["${plugin_root}/skills/ship/scripts/pr-evidence.py", "check", "${plan_dir}/pr.md"]], "attempts": 3, }, } @@ -455,22 +456,40 @@ def _normalize_under_root(raw: str, roots: dict) -> str: return text -def _resolves_under_allowlist(path_text: str, roots: dict) -> bool: +def _resolved_path(path_text: str, roots: dict): + """The absolute path a skill or check executable names, or None when unresolvable.""" resolved = Path(_normalize_under_root(path_text, roots)) try: - resolved = resolved.resolve() if resolved.is_absolute() else (roots["repo_root"] / resolved).resolve() + if not resolved.is_absolute(): + resolved = Path(roots["repo_root"]) / resolved + return resolved.resolve() except (OSError, RuntimeError): + return None + + +def _resolves_under_allowlist(path_text: str, roots: dict) -> bool: + resolved = _resolved_path(path_text, roots) + if resolved is None: return False for root_key in ("plugin_root", "repo_root"): - root = Path(roots[root_key]).resolve() try: - resolved.relative_to(root) + resolved.relative_to(Path(roots[root_key]).resolve()) return True except ValueError: continue return False +def _path_findings(pid: str, what: str, path_text: str, roots: dict) -> list: + """Findings for a skill path or check executable outside the roots or not a file.""" + if not _resolves_under_allowlist(path_text, roots): + return [f"phase {pid!r}: {what} {path_text!r} is outside the plugin and repo roots"] + resolved = _resolved_path(path_text, roots) + if resolved is None or not resolved.is_file(): + return [f"phase {pid!r}: {what} {path_text!r} does not resolve to a file"] + return [] + + def sha256_of(path: Path) -> str: return hashlib.sha256(path.read_bytes()).hexdigest() @@ -589,8 +608,7 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): type_def = types.get(ptype, BUILTIN_TYPES.get(ptype, {})) skill_raw = phase.get("skill", "") - if not _resolves_under_allowlist(skill_raw, roots): - findings.append(f"phase {pid!r}: skill path {skill_raw!r} does not resolve under the plugin or repo root") + findings.extend(_path_findings(pid, "skill path", skill_raw, roots)) skill_resolved = substitute(skill_raw, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) checks_raw = phase["checks"] if "checks" in phase else type_def.get("checks", []) @@ -613,8 +631,7 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): for ph in placeholders_in(element): if ph not in CHECK_PLACEHOLDERS: findings.append(f"phase {pid!r}: check {entry!r} uses unsupported placeholder {ph}") - if not _resolves_under_allowlist(executable, roots): - findings.append(f"phase {pid!r}: check executable {executable!r} does not resolve under the plugin or repo root") + findings.extend(_path_findings(pid, "check executable", executable, roots)) resolved_entry = [substitute(e, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) for e in entry] resolved_checks.append(resolved_entry) diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index 8ab315e..bf61910 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -1,16 +1,19 @@ --- name: run -description: Take a request through an interactive scope, then run an unattended scope-review, build, and ship as phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use to run the factory end to end on Claude Code or Codex. +description: Drive the pipeline declared in .factory/config.yaml, or the built-in scope, scope-review, build and ship when there is none - the interactive phases inline, then every other phase as an unattended subagent, judging each against its type's declared contract and repairing or relaunching on failure, ending in a pull request or a report. Use to run the factory end to end on Claude Code or Codex. disable-model-invocation: true --- # Run -You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase invoke another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. +You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase launch another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. +The pipeline is data: an ordered list of typed phases in `.factory/config.yaml`, or the built-in default (`scope`, `scope-review`, `build`, `ship`) when that file is absent. You never infer what a phase means - everything you do with one comes from its type's declared axes, read via `factory-config.py show --resolved` ([references/pipeline-config.md](../../references/pipeline-config.md)). +`${plugin_root}` is `{run-skill-root}/../..`, and `factory-config.py` resolves it from its own location, the same directory. ## 1. Preflight Confirm the host has a subagent tool per the transport ladder in [references/launch.md](references/launch.md); with none, stop before the interview so nobody pays for a scope that cannot run. +Run `python3 {run-skill-root}/../../scripts/factory-config.py check`. It is the single validator, so judge nothing about the file yourself: on exit 1 print its findings verbatim and stop before the interview. When a finding says a skill path does not resolve, also name the command that installs the defaults, `python3 {run-skill-root}/../../scripts/factory-config.py inject`. Check that `.dev/` is git-ignored in the consuming repository (`git check-ignore -q .dev/`); when it is not, append the entry to `.gitignore` yourself, and once the run branch exists at handoff, record the write as a repair with the check output as evidence (D-plan-files-ignored). The two hard stops in section 5 bind this write too. Resolve your own base directory from the host's injected "Base directory for this skill" line. @@ -20,11 +23,11 @@ If two unfinished state files exist under `.dev/` with no branch match, or a rec ## 2. Init or resume -A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it, create `.dev/{plan}/` with `request.md` holding the request, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch}` (D-plan-slug, D-request-input). +A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it and create `.dev/{plan}/` with `request.md` holding the request. Write `python3 {run-skill-root}/../../scripts/factory-config.py show --resolved --json` to `.dev/{plan}/pipeline.json`, read `show --resolved` for each phase's type, `attempts`, `checks`, `requires`, `seals`, class and skill path plus the `ceiling` and the `go`, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch} --phases {ids in order, comma-separated} --attempts {phase}={n} ... --ceiling {ceiling}` (D-plan-slug, D-request-input). No request and a run branch checked out: resume that plan. -No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when its `decisions` list ends in an `end` action, or in an `advance` on `ship`; nothing else marks a run finished, so every other state file is unfinished. +No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when `run-state.py show {plan}` prints `finished: yes` - its `decisions` list ends in an `end` action, or in an `advance` on the last declared phase; nothing else marks a run finished, so every other state file is unfinished. -Resuming, launch nothing before reading the history. Run `run-state.py show {plan}`: it prints each phase's attempts in order with their statuses, and those numbers are the run's only attempt numbers. Take the highest-numbered attempt of the earliest phase you have not accepted as `done`, and act on its status: +Resuming, launch nothing before reading the history. Write a fresh `show --resolved --json` to `.dev/{plan}/pipeline-now.json` and run `run-state.py diff-config {plan} --pipeline .dev/{plan}/pipeline-now.json` and `run-state.py diff-spec {plan}`; their output is evidence for the next judgment, not a block. Run `run-state.py show {plan}`: it prints each phase's attempts in order with their statuses, and those numbers are the run's only attempt numbers. Take the highest-numbered attempt of the earliest phase you have not accepted as `done`, and act on its status: - `launched`, and `.dev/{plan}/results/{phase}-{attempt}.json` exists for that attempt's own number: the phase finished and the session died before its result was read. Pick section 5 up at step 4, `check-result` on that exact path. - `launched`, and no result file at that attempt's own number: the attempt died mid-phase. Close it with `run-state.py attempt {plan} {phase} --status failed` and no `--result`, so it is a failure with the reason "no result file" and no other attempt's result is read into it. Then take a new attempt from section 5 step 1, its guidance saying to start from the artifacts already on disk and not redo finished work. @@ -32,30 +35,34 @@ Resuming, launch nothing before reading the history. Run `run-state.py show {pla A result file belongs to the attempt number in its name and to no other. Never infer an attempt number from what is in `results/`: after a crash, the newest file there is the last attempt that finished, not the one that was running. -## 3. Scope, inline +## 3. The interactive prefix, inline + +The phases whose type is `interactive` form a contiguous prefix of the pipeline; they run inline in this session, in order, because a subagent has no user-input tool. -Launch the copied `scope` skill inline, in this session: read its `SKILL.md` at the absolute path `{run-skill-root}/../scope/SKILL.md`, telling it to read `{scope-skill-root}` as that path's parent directory. It keeps its interview and ends with one explicit go question you ask the user. +For each one, bracket it the way a launched phase is: run `run-state.py attempt {plan} {phase} --type {type} --skill {skill path}` first, read its `SKILL.md` at its resolved skill path (telling it to read that path's parent directory as its skill root), let it keep its interview, and close it with `run-state.py attempt {plan} {phase} --status {done|failed} --result .dev/{plan}/results/{phase}-{attempt}.json`. The last phase of the prefix ends with one explicit go question you ask the user. +A pipeline whose prefix is empty has no go: it branches and seals before its first launch. + ## 4. Handoff -At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan} --seal .dev/{plan}/spec.md` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). +At the go - or before the first launch when there is no prefix - create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan}` with one `--seal {path}` per path in the `seals` of each prefix phase (`${plan_dir}` is `.dev/{plan}`), plus `--config .factory/config.yaml --pipeline .dev/{plan}/pipeline.json`. It records each sealed file's hash and, for a spec in the dev notation, its scenario texts and `⊘` lines, the config hash with the resolved pipeline, and the checkout's dirty files (D-checkout, D-handoff-seal). With no `seals` it seals nothing and `diff-spec` says so. ## 5. Per phase: launch, judge, act -For each phase in order (`scope-review`, `build`, `ship`): +For each phase after the prefix, in declared order: -1. Take the attempt number first: `run-state.py attempt {plan} {phase}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. +1. Take the attempt number first: `run-state.py attempt {plan} {phase} --type {type} --skill {skill path}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. 2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. 3. Wait for the host's completion signal - a batch in flight is not over. 4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. -5. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). +5. Judge the phase per [references/judgment.md](references/judgment.md): run its `checks`, then weigh its `requires`. 6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. -7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. -8. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. +7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run. Follow the phase's budget and the run's ceiling as printed by `run-state.py show {plan}`; `attempt` never refuses one, so honouring it is yours. +8. Before advancing, run `run-state.py diff-spec {plan}` and `run-state.py diff-config {plan} --pipeline .dev/{plan}/pipeline-now.json`; a dropped or reworded sealed scenario, `⊘` line or phase is evidence for the next judgment, not an automatic block. ## 6. Closing report -The last decision is recorded before the report is printed - the `end`, or the `advance` that accepts `ship` - because that entry is the only thing that marks the run finished for a later resume. -Print: phases, attempts, decisions, repairs, and the PR link or the exact action a person must take. Read only result files, check outputs, `git log`, and the state file for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). +The last decision is recorded before the report is printed - the `end`, or the `advance` that accepts the last declared phase - because that entry is the only thing that marks the run finished for a later resume. +Print, per phase: its type, the skill path that ran, its attempts with opened and closed times, and the decisions and repairs, then the PR link or the exact action a person must take. Read only the state file (`run-state.py show`), result files, check outputs and `git log` for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). diff --git a/factory/skills/run/references/judgment.md b/factory/skills/run/references/judgment.md index f30a2e0..a58c545 100644 --- a/factory/skills/run/references/judgment.md +++ b/factory/skills/run/references/judgment.md @@ -1,22 +1,23 @@ # Judgment -Per phase, the evidence commands the orchestrator re-runs as proof and what a `done` looks like. `run-state.py check-result` gives the result file's own claim; these checks are the independent evidence read before trusting it. +How the orchestrator decides a phase is done, from what its type declares. `run-state.py check-result` gives the result file's own claim; the type's `checks` and `requires` are the independent evidence read before trusting it. Read them from `factory-config.py show --resolved`. -## scope +## Reading a type -`done`: `lint-spec.py .dev/{plan}/spec.md` reports clean, and the go was recorded (the interview happened inline in this session, so the orchestrator already knows). +- **`checks`**: each entry is an argv list. Substitute `${plan_dir}` (`.dev/{plan}`), `${phase}` and `${attempt}` in each element, then run the list element by element, never composed into a shell string. `check` already refused any executable element carrying `${plan_dir}`, `${phase}` or `${attempt}`, so only the arguments vary. An entry that exits non-zero is a failed check whose output is the reason. +- **`requires`**: the outcome contract in prose. Judge each criterion it names by running the tool it names yourself (`git`, `gh`, `grep`), as the built-in types below do. A type declaring no `checks` is judged on its result envelope and its `requires` alone; that is weaker evidence, so weigh it as such. +- **`attempts`** and the run's `ceiling`: the budgets the failure ladder follows. +- A phase never reaches `done` because its own result says so. Its result claims; the checks and `requires` decide. -## scope-review +## The built-in types -`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a line whose whole text is exactly `Verdict: APPROVED` - nothing after it - with a `Rounds:` line. Check it as a whole line, `grep -cx 'Verdict: APPROVED'`, never as a substring: a longer verdict such as `Verdict: APPROVED WITH DEFERRALS` is not a `done`. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. +`interview` (the `scope` phase): `done` when `lint-spec.py` reports the spec clean and the go was recorded (the interview happened inline in this session, so the orchestrator already knows). -## build +`review` (`scope-review`): `done` when `spec-review_N.md` contains a line whose whole text is exactly `Verdict: APPROVED` - nothing after it - with a `Rounds:` line, and the spec still lints clean. Check it as a whole line, `grep -cx 'Verdict: APPROVED'`, never as a substring: a longer verdict such as `Verdict: APPROVED WITH DEFERRALS` is not a `done`. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. -`done`: `check-tests.py .dev/{plan}` clean, the spec's Validation block green, commits since handoff match the change plan's numbering, and (when the spec has `[e2e]` scenarios) the e2e report's data block shows every scenario passed. +`implement` (`build`): `done` when `check-tests.py` reports clean, the spec's Validation block is green, commits since handoff match the change plan's numbering, and (when the spec has `[e2e]` scenarios) the e2e report's data block shows every scenario passed. Only `check-tests.py` is a declared check; the rest is `requires`. -## ship - -`done` has two endings, decided by what `origin` resolves to: +`ship` (`ship`) has two endings, decided by what `origin` resolves to: - **A GitHub remote**: `pr-evidence.py check` clean, `review_N.md` carries a verdict for HEAD, and `gh pr view` shows the PR open with head equal to HEAD. - **A remote no GitHub host backs** (the fixture's bare origin): `gh pr create` cannot succeed, so ship cannot reach `done` at all. The evidence is `git ls-remote origin` showing the branch at HEAD and `pr.md` written to disk. The orchestrator ends the run with a report naming the pull request as the one unfinished step - this is the fixture's expected ending, not a factory fault, and is judged under the third `gh` fault category below, never as a park. @@ -25,9 +26,9 @@ Per phase, the evidence commands the orchestrator re-runs as proof and what a `d Cheapest sufficient step, recorded before it is acted on: -1. **Repair.** Fix it yourself and re-run the phase's checks - at most two repairs per attempt; the third fix is a relaunch and counts as a new attempt. A repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches. +1. **Repair.** Only for a built-in phase or an injected copy that still matches its manifest - what `show --resolved` prints as class `built-in` or `injected`. Fix it yourself and re-run the phase's checks - at most two repairs per attempt; the third fix is a relaunch and counts as a new attempt. A repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches. A foreign phase is never repaired: you never infer the meaning of an artifact you did not define, so a foreign failure goes straight to relaunch or end. 2. **Relaunch.** A fresh subagent, guidance naming what the previous attempt did and what is required instead - never "try again". -3. **End.** Three attempts of one phase without a `done` the orchestrator accepts: end the run with a report naming the last reason, the artifacts, and what a person could do. +3. **End.** The phase's attempts reach its `attempts` budget (three for every built-in type) without a `done` the orchestrator accepts, or the run's attempts reach its `ceiling`: end the run with a report naming every phase's attempts, the last reason, the artifacts, and what a person could do. Every repair is committed on the run branch and bound by the two hard stops: never rewrite history, force-push, or delete outside the run branch. diff --git a/factory/skills/run/references/launch.md b/factory/skills/run/references/launch.md index 2e9b7ff..a9c8137 100644 --- a/factory/skills/run/references/launch.md +++ b/factory/skills/run/references/launch.md @@ -9,12 +9,16 @@ The host transport ladder for one phase agent, and the exact prompt template. Mi 3. **opencode**: the `task` tool. 4. **Unavailable**: stop before the interview - nobody pays for a scope that cannot run. +## The wrapper + +The template below wraps any phase skill, ours or a team's own, and imposes the result shape on it: the result file at `results/{phase}-{attempt}.json`, schema `factory.result/1`, written as the phase's last action. It imposes the shape, it does not guarantee it - a skill with strong opinions about its own output can defeat it, so a phase that writes nothing is read as a failed attempt, and the phase's declared `checks` verify the outcome independently of what the result says. + ## Prompt template ``` -You implement exactly one factory phase. Read {phase}-skill-root/SKILL.md at -the absolute path {skill_path}; treat {skill_path}'s parent directory as -{phase}-skill-root when the file uses that placeholder. Follow it exactly. +You run exactly one factory phase. Read the skill at the absolute path +{skill_path}, and treat {skill_path}'s parent directory as the skill's own +root when the file uses a root placeholder. Follow it exactly. Plan: {plan}, at .dev/{plan}/ (absolute: {plan_dir}). Attempt: {attempt}, the number run-state.py just printed for this launch. @@ -23,10 +27,11 @@ Guidance: {what to do differently, never "try again"}.} Scratch root for this attempt: /tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/. Write your result to .dev/{plan}/results/{phase}-{attempt}.json as your last -action, per the schema in factory-run.md, echoing skill_path back exactly as -given above. +action: a JSON object with "schema": "factory.result/1", "phase", "status" +(done, failed or stopped), "reason" when not done, and skill_path echoed back +exactly as given above; the full schema is in factory-run.md. -This phase never launches another phase, and after the go it never asks a -person anything - decide under the unattended policy in factory-run.md and -record the decision in auto_decided. +Never name or launch the next phase. After the go you never ask a person +anything - decide under the unattended policy in factory-run.md and record +the decision in auto_decided. ``` diff --git a/plugins/factory/references/factory-run.md b/plugins/factory/references/factory-run.md index a2ffc79..43cf943 100644 --- a/plugins/factory/references/factory-run.md +++ b/plugins/factory/references/factory-run.md @@ -45,16 +45,16 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las After the go, no phase skill asks a person anything. An escalation that dev's version would put outside the run is instead: decided under the recommended option and recorded in `auto_decided`, or reported as `failed`/`stopped` with the reason a person would need. The two hard stops (`secret.found`, `action.destructive`) always report `stopped`; the orchestrator never overrides them. -`scope` is the one exception: it runs inline in the orchestrator's own session, keeps its interview, and is outside `check_factory_unattended`'s scan root. `run/SKILL.md` is the only file in the scan root that may use `` / ``, for its pre-go lines; the markers must balance and each must stand alone on its line. No phase copy may use them, because no phase copy has anything to route. +The interactive phases - `scope` in the built-in pipeline - are the one exception: they run inline in the orchestrator's own session, keep their interview, and are outside `check_factory_unattended`'s scan root. A phase body the factory did not write is covered by no check: its author declares it `unattended_safe` in the config, and a phase that blocks on a person is neither prevented nor detected. `run/SKILL.md` is the only file in the scan root that may use `` / ``, for its pre-go lines; the markers must balance and each must stand alone on its line. No phase copy may use them, because no phase copy has anything to route. ## Launch prompt shape Each phase subagent's prompt carries, in order: -1. The phase's skill path: the absolute path to `factory/skills/{phase}/SKILL.md`, with the instruction to read `{phase}-skill-root` as that path's parent directory (a subagent reading a file cannot resolve `{phase}-skill-root}`-style placeholders on its own). +1. The phase's skill path: the absolute resolved `skill` path from `factory-config.py show --resolved`, with the instruction to read that path's parent directory wherever the file uses a skill-root placeholder (a subagent reading a file cannot resolve `{phase}-skill-root`-style placeholders on its own). Any skill can be wrapped this way, ours or a team's own. 2. The plan name and the plan directory's absolute path. 3. The attempt number, which is the one `run-state.py attempt` printed when it recorded this launch, never a number counted by hand or read off a file in `results/`. 4. On a relaunch: the previous attempt's reason and explicit guidance naming what it did and what is required instead - never "try again". 5. The scratch root for this attempt: `/tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/`. 6. The result path this attempt must write: `.dev/{plan}/results/{phase}-{attempt}.json`. -7. The instruction that this phase never launches another phase and, after the go, never asks a person anything - decide under the unattended policy above and record it. +7. The instruction that this phase never names or launches the next phase and, after the go, never asks a person anything - decide under the unattended policy above and record it. `launch.md` carries the exact wording. The wrapper imposes the result shape on a phase; it does not guarantee it, and a phase that writes no result is read as a failed attempt. diff --git a/plugins/factory/references/pipeline-config.md b/plugins/factory/references/pipeline-config.md index d72662b..2b75602 100644 --- a/plugins/factory/references/pipeline-config.md +++ b/plugins/factory/references/pipeline-config.md @@ -85,4 +85,4 @@ A list-valued key takes repeated `--item` flags in place of a positional value. ## The built-in default pipeline -Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check` (ship) become `checks` entries; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. +Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/skills/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. diff --git a/plugins/factory/scripts/factory-config.py b/plugins/factory/scripts/factory-config.py index 98bdc78..a8caed2 100755 --- a/plugins/factory/scripts/factory-config.py +++ b/plugins/factory/scripts/factory-config.py @@ -56,21 +56,22 @@ "interactive": True, "requires": "the go was recorded (the interview happened inline in this session, " "so the orchestrator already knows)", - "checks": [["lint-spec.py", "${plan_dir}/spec.md"]], + "checks": [["${plugin_root}/skills/scope/scripts/lint-spec.py", "${plan_dir}/spec.md"]], "seals": ["${plan_dir}/spec.md"], "attempts": 3, }, "review": { "interactive": False, "requires": 'spec-review_N.md contains a line whose whole text is exactly ' - '"Verdict: APPROVED" - nothing after it - with a "Rounds:" line', + '"Verdict: APPROVED" - nothing after it - with a "Rounds:" line, and the spec ' + 'still lints clean with lint-spec.py', "attempts": 3, }, "implement": { "interactive": False, "requires": "the spec's Validation block is green, and commits since handoff " "match the change plan's numbering", - "checks": [["check-tests.py", "${plan_dir}"]], + "checks": [["${plugin_root}/skills/build/scripts/check-tests.py", "${plan_dir}"]], "attempts": 3, }, "ship": { @@ -78,7 +79,7 @@ "requires": "review_N.md carries a verdict for HEAD and gh pr view shows the PR " "open with head equal to HEAD, or (without a GitHub remote) git ls-remote origin " "shows the branch at HEAD and pr.md is written", - "checks": [["pr-evidence.py", "check"]], + "checks": [["${plugin_root}/skills/ship/scripts/pr-evidence.py", "check", "${plan_dir}/pr.md"]], "attempts": 3, }, } @@ -455,22 +456,40 @@ def _normalize_under_root(raw: str, roots: dict) -> str: return text -def _resolves_under_allowlist(path_text: str, roots: dict) -> bool: +def _resolved_path(path_text: str, roots: dict): + """The absolute path a skill or check executable names, or None when unresolvable.""" resolved = Path(_normalize_under_root(path_text, roots)) try: - resolved = resolved.resolve() if resolved.is_absolute() else (roots["repo_root"] / resolved).resolve() + if not resolved.is_absolute(): + resolved = Path(roots["repo_root"]) / resolved + return resolved.resolve() except (OSError, RuntimeError): + return None + + +def _resolves_under_allowlist(path_text: str, roots: dict) -> bool: + resolved = _resolved_path(path_text, roots) + if resolved is None: return False for root_key in ("plugin_root", "repo_root"): - root = Path(roots[root_key]).resolve() try: - resolved.relative_to(root) + resolved.relative_to(Path(roots[root_key]).resolve()) return True except ValueError: continue return False +def _path_findings(pid: str, what: str, path_text: str, roots: dict) -> list: + """Findings for a skill path or check executable outside the roots or not a file.""" + if not _resolves_under_allowlist(path_text, roots): + return [f"phase {pid!r}: {what} {path_text!r} is outside the plugin and repo roots"] + resolved = _resolved_path(path_text, roots) + if resolved is None or not resolved.is_file(): + return [f"phase {pid!r}: {what} {path_text!r} does not resolve to a file"] + return [] + + def sha256_of(path: Path) -> str: return hashlib.sha256(path.read_bytes()).hexdigest() @@ -589,8 +608,7 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): type_def = types.get(ptype, BUILTIN_TYPES.get(ptype, {})) skill_raw = phase.get("skill", "") - if not _resolves_under_allowlist(skill_raw, roots): - findings.append(f"phase {pid!r}: skill path {skill_raw!r} does not resolve under the plugin or repo root") + findings.extend(_path_findings(pid, "skill path", skill_raw, roots)) skill_resolved = substitute(skill_raw, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) checks_raw = phase["checks"] if "checks" in phase else type_def.get("checks", []) @@ -613,8 +631,7 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): for ph in placeholders_in(element): if ph not in CHECK_PLACEHOLDERS: findings.append(f"phase {pid!r}: check {entry!r} uses unsupported placeholder {ph}") - if not _resolves_under_allowlist(executable, roots): - findings.append(f"phase {pid!r}: check executable {executable!r} does not resolve under the plugin or repo root") + findings.extend(_path_findings(pid, "check executable", executable, roots)) resolved_entry = [substitute(e, {"plugin_root": roots["plugin_root"], "repo_root": roots["repo_root"]}) for e in entry] resolved_checks.append(resolved_entry) diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index a5fd108..c752e20 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -1,15 +1,18 @@ --- name: run -description: Take a request through an interactive scope, then run an unattended scope-review, build, and ship as phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use to run the factory end to end on Claude Code or Codex. +description: Drive the pipeline declared in .factory/config.yaml, or the built-in scope, scope-review, build and ship when there is none - the interactive phases inline, then every other phase as an unattended subagent, judging each against its type's declared contract and repairing or relaunching on failure, ending in a pull request or a report. Use to run the factory end to end on Claude Code or Codex. --- # Run -You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase invoke another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. +You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase launch another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. +The pipeline is data: an ordered list of typed phases in `.factory/config.yaml`, or the built-in default (`scope`, `scope-review`, `build`, `ship`) when that file is absent. You never infer what a phase means - everything you do with one comes from its type's declared axes, read via `factory-config.py show --resolved` ([references/pipeline-config.md](../../references/pipeline-config.md)). +`${plugin_root}` is `{run-skill-root}/../..`, and `factory-config.py` resolves it from its own location, the same directory. ## 1. Preflight Confirm the host has a subagent tool per the transport ladder in [references/launch.md](references/launch.md); with none, stop before the interview so nobody pays for a scope that cannot run. +Run `python3 {run-skill-root}/../../scripts/factory-config.py check`. It is the single validator, so judge nothing about the file yourself: on exit 1 print its findings verbatim and stop before the interview. When a finding says a skill path does not resolve, also name the command that installs the defaults, `python3 {run-skill-root}/../../scripts/factory-config.py inject`. Check that `.dev/` is git-ignored in the consuming repository (`git check-ignore -q .dev/`); when it is not, append the entry to `.gitignore` yourself, and once the run branch exists at handoff, record the write as a repair with the check output as evidence (D-plan-files-ignored). The two hard stops in section 5 bind this write too. Resolve your own base directory from the host's injected "Base directory for this skill" line. @@ -19,11 +22,11 @@ If two unfinished state files exist under `.dev/` with no branch match, or a rec ## 2. Init or resume -A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it, create `.dev/{plan}/` with `request.md` holding the request, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch}` (D-plan-slug, D-request-input). +A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it and create `.dev/{plan}/` with `request.md` holding the request. Write `python3 {run-skill-root}/../../scripts/factory-config.py show --resolved --json` to `.dev/{plan}/pipeline.json`, read `show --resolved` for each phase's type, `attempts`, `checks`, `requires`, `seals`, class and skill path plus the `ceiling` and the `go`, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch} --phases {ids in order, comma-separated} --attempts {phase}={n} ... --ceiling {ceiling}` (D-plan-slug, D-request-input). No request and a run branch checked out: resume that plan. -No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when its `decisions` list ends in an `end` action, or in an `advance` on `ship`; nothing else marks a run finished, so every other state file is unfinished. +No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when `run-state.py show {plan}` prints `finished: yes` - its `decisions` list ends in an `end` action, or in an `advance` on the last declared phase; nothing else marks a run finished, so every other state file is unfinished. -Resuming, launch nothing before reading the history. Run `run-state.py show {plan}`: it prints each phase's attempts in order with their statuses, and those numbers are the run's only attempt numbers. Take the highest-numbered attempt of the earliest phase you have not accepted as `done`, and act on its status: +Resuming, launch nothing before reading the history. Write a fresh `show --resolved --json` to `.dev/{plan}/pipeline-now.json` and run `run-state.py diff-config {plan} --pipeline .dev/{plan}/pipeline-now.json` and `run-state.py diff-spec {plan}`; their output is evidence for the next judgment, not a block. Run `run-state.py show {plan}`: it prints each phase's attempts in order with their statuses, and those numbers are the run's only attempt numbers. Take the highest-numbered attempt of the earliest phase you have not accepted as `done`, and act on its status: - `launched`, and `.dev/{plan}/results/{phase}-{attempt}.json` exists for that attempt's own number: the phase finished and the session died before its result was read. Pick section 5 up at step 4, `check-result` on that exact path. - `launched`, and no result file at that attempt's own number: the attempt died mid-phase. Close it with `run-state.py attempt {plan} {phase} --status failed` and no `--result`, so it is a failure with the reason "no result file" and no other attempt's result is read into it. Then take a new attempt from section 5 step 1, its guidance saying to start from the artifacts already on disk and not redo finished work. @@ -31,30 +34,34 @@ Resuming, launch nothing before reading the history. Run `run-state.py show {pla A result file belongs to the attempt number in its name and to no other. Never infer an attempt number from what is in `results/`: after a crash, the newest file there is the last attempt that finished, not the one that was running. -## 3. Scope, inline +## 3. The interactive prefix, inline + +The phases whose type is `interactive` form a contiguous prefix of the pipeline; they run inline in this session, in order, because a subagent has no user-input tool. -Launch the copied `scope` skill inline, in this session: read its `SKILL.md` at the absolute path `{run-skill-root}/../scope/SKILL.md`, telling it to read `{scope-skill-root}` as that path's parent directory. It keeps its interview and ends with one explicit go question you ask the user. +For each one, bracket it the way a launched phase is: run `run-state.py attempt {plan} {phase} --type {type} --skill {skill path}` first, read its `SKILL.md` at its resolved skill path (telling it to read that path's parent directory as its skill root), let it keep its interview, and close it with `run-state.py attempt {plan} {phase} --status {done|failed} --result .dev/{plan}/results/{phase}-{attempt}.json`. The last phase of the prefix ends with one explicit go question you ask the user. +A pipeline whose prefix is empty has no go: it branches and seals before its first launch. + ## 4. Handoff -At the go: create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan} --seal .dev/{plan}/spec.md` to record the spec's sha256, its scenario texts per change set, its `⊘` lines, and the checkout's dirty files (D-checkout, D-handoff-seal). +At the go - or before the first launch when there is no prefix - create branch `factory/{plan}` from the base branch. Run `run-state.py handoff {plan} --branch factory/{plan}` with one `--seal {path}` per path in the `seals` of each prefix phase (`${plan_dir}` is `.dev/{plan}`), plus `--config .factory/config.yaml --pipeline .dev/{plan}/pipeline.json`. It records each sealed file's hash and, for a spec in the dev notation, its scenario texts and `⊘` lines, the config hash with the resolved pipeline, and the checkout's dirty files (D-checkout, D-handoff-seal). With no `seals` it seals nothing and `diff-spec` says so. ## 5. Per phase: launch, judge, act -For each phase in order (`scope-review`, `build`, `ship`): +For each phase after the prefix, in declared order: -1. Take the attempt number first: `run-state.py attempt {plan} {phase}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. +1. Take the attempt number first: `run-state.py attempt {plan} {phase} --type {type} --skill {skill path}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. 2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. 3. Wait for the host's completion signal - a batch in flight is not over. 4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. -5. Run the phase's evidence checks from [references/judgment.md](references/judgment.md). +5. Judge the phase per [references/judgment.md](references/judgment.md): run its `checks`, then weigh its `requires`. 6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. -7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run after the third accepted failure. -8. Before advancing, run `run-state.py diff-spec {plan}`; a dropped or reworded approved scenario or `⊘` line is evidence for the next judgment, not an automatic block. +7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run. Follow the phase's budget and the run's ceiling as printed by `run-state.py show {plan}`; `attempt` never refuses one, so honouring it is yours. +8. Before advancing, run `run-state.py diff-spec {plan}` and `run-state.py diff-config {plan} --pipeline .dev/{plan}/pipeline-now.json`; a dropped or reworded sealed scenario, `⊘` line or phase is evidence for the next judgment, not an automatic block. ## 6. Closing report -The last decision is recorded before the report is printed - the `end`, or the `advance` that accepts `ship` - because that entry is the only thing that marks the run finished for a later resume. -Print: phases, attempts, decisions, repairs, and the PR link or the exact action a person must take. Read only result files, check outputs, `git log`, and the state file for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). +The last decision is recorded before the report is printed - the `end`, or the `advance` that accepts the last declared phase - because that entry is the only thing that marks the run finished for a later resume. +Print, per phase: its type, the skill path that ran, its attempts with opened and closed times, and the decisions and repairs, then the PR link or the exact action a person must take. Read only the state file (`run-state.py show`), result files, check outputs and `git log` for this - never a subagent transcript, the full spec, or implementation notes unless a judgment needs a specific section (D-orchestrator-context). diff --git a/plugins/factory/skills/run/references/judgment.md b/plugins/factory/skills/run/references/judgment.md index f30a2e0..a58c545 100644 --- a/plugins/factory/skills/run/references/judgment.md +++ b/plugins/factory/skills/run/references/judgment.md @@ -1,22 +1,23 @@ # Judgment -Per phase, the evidence commands the orchestrator re-runs as proof and what a `done` looks like. `run-state.py check-result` gives the result file's own claim; these checks are the independent evidence read before trusting it. +How the orchestrator decides a phase is done, from what its type declares. `run-state.py check-result` gives the result file's own claim; the type's `checks` and `requires` are the independent evidence read before trusting it. Read them from `factory-config.py show --resolved`. -## scope +## Reading a type -`done`: `lint-spec.py .dev/{plan}/spec.md` reports clean, and the go was recorded (the interview happened inline in this session, so the orchestrator already knows). +- **`checks`**: each entry is an argv list. Substitute `${plan_dir}` (`.dev/{plan}`), `${phase}` and `${attempt}` in each element, then run the list element by element, never composed into a shell string. `check` already refused any executable element carrying `${plan_dir}`, `${phase}` or `${attempt}`, so only the arguments vary. An entry that exits non-zero is a failed check whose output is the reason. +- **`requires`**: the outcome contract in prose. Judge each criterion it names by running the tool it names yourself (`git`, `gh`, `grep`), as the built-in types below do. A type declaring no `checks` is judged on its result envelope and its `requires` alone; that is weaker evidence, so weigh it as such. +- **`attempts`** and the run's `ceiling`: the budgets the failure ladder follows. +- A phase never reaches `done` because its own result says so. Its result claims; the checks and `requires` decide. -## scope-review +## The built-in types -`done`: `lint-spec.py` clean, and `spec-review_N.md` contains a line whose whole text is exactly `Verdict: APPROVED` - nothing after it - with a `Rounds:` line. Check it as a whole line, `grep -cx 'Verdict: APPROVED'`, never as a substring: a longer verdict such as `Verdict: APPROVED WITH DEFERRALS` is not a `done`. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. +`interview` (the `scope` phase): `done` when `lint-spec.py` reports the spec clean and the go was recorded (the interview happened inline in this session, so the orchestrator already knows). -## build +`review` (`scope-review`): `done` when `spec-review_N.md` contains a line whose whole text is exactly `Verdict: APPROVED` - nothing after it - with a `Rounds:` line, and the spec still lints clean. Check it as a whole line, `grep -cx 'Verdict: APPROVED'`, never as a substring: a longer verdict such as `Verdict: APPROVED WITH DEFERRALS` is not a `done`. A `failed` result with reason `rescope` is not repaired: it ends the run naming what a future `scope` run must revisit. -`done`: `check-tests.py .dev/{plan}` clean, the spec's Validation block green, commits since handoff match the change plan's numbering, and (when the spec has `[e2e]` scenarios) the e2e report's data block shows every scenario passed. +`implement` (`build`): `done` when `check-tests.py` reports clean, the spec's Validation block is green, commits since handoff match the change plan's numbering, and (when the spec has `[e2e]` scenarios) the e2e report's data block shows every scenario passed. Only `check-tests.py` is a declared check; the rest is `requires`. -## ship - -`done` has two endings, decided by what `origin` resolves to: +`ship` (`ship`) has two endings, decided by what `origin` resolves to: - **A GitHub remote**: `pr-evidence.py check` clean, `review_N.md` carries a verdict for HEAD, and `gh pr view` shows the PR open with head equal to HEAD. - **A remote no GitHub host backs** (the fixture's bare origin): `gh pr create` cannot succeed, so ship cannot reach `done` at all. The evidence is `git ls-remote origin` showing the branch at HEAD and `pr.md` written to disk. The orchestrator ends the run with a report naming the pull request as the one unfinished step - this is the fixture's expected ending, not a factory fault, and is judged under the third `gh` fault category below, never as a park. @@ -25,9 +26,9 @@ Per phase, the evidence commands the orchestrator re-runs as proof and what a `d Cheapest sufficient step, recorded before it is acted on: -1. **Repair.** Fix it yourself and re-run the phase's checks - at most two repairs per attempt; the third fix is a relaunch and counts as a new attempt. A repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches. +1. **Repair.** Only for a built-in phase or an injected copy that still matches its manifest - what `show --resolved` prints as class `built-in` or `injected`. Fix it yourself and re-run the phase's checks - at most two repairs per attempt; the third fix is a relaunch and counts as a new attempt. A repair after ship's review moves HEAD, so the review no longer names HEAD and ship relaunches. A foreign phase is never repaired: you never infer the meaning of an artifact you did not define, so a foreign failure goes straight to relaunch or end. 2. **Relaunch.** A fresh subagent, guidance naming what the previous attempt did and what is required instead - never "try again". -3. **End.** Three attempts of one phase without a `done` the orchestrator accepts: end the run with a report naming the last reason, the artifacts, and what a person could do. +3. **End.** The phase's attempts reach its `attempts` budget (three for every built-in type) without a `done` the orchestrator accepts, or the run's attempts reach its `ceiling`: end the run with a report naming every phase's attempts, the last reason, the artifacts, and what a person could do. Every repair is committed on the run branch and bound by the two hard stops: never rewrite history, force-push, or delete outside the run branch. diff --git a/plugins/factory/skills/run/references/launch.md b/plugins/factory/skills/run/references/launch.md index 2e9b7ff..a9c8137 100644 --- a/plugins/factory/skills/run/references/launch.md +++ b/plugins/factory/skills/run/references/launch.md @@ -9,12 +9,16 @@ The host transport ladder for one phase agent, and the exact prompt template. Mi 3. **opencode**: the `task` tool. 4. **Unavailable**: stop before the interview - nobody pays for a scope that cannot run. +## The wrapper + +The template below wraps any phase skill, ours or a team's own, and imposes the result shape on it: the result file at `results/{phase}-{attempt}.json`, schema `factory.result/1`, written as the phase's last action. It imposes the shape, it does not guarantee it - a skill with strong opinions about its own output can defeat it, so a phase that writes nothing is read as a failed attempt, and the phase's declared `checks` verify the outcome independently of what the result says. + ## Prompt template ``` -You implement exactly one factory phase. Read {phase}-skill-root/SKILL.md at -the absolute path {skill_path}; treat {skill_path}'s parent directory as -{phase}-skill-root when the file uses that placeholder. Follow it exactly. +You run exactly one factory phase. Read the skill at the absolute path +{skill_path}, and treat {skill_path}'s parent directory as the skill's own +root when the file uses a root placeholder. Follow it exactly. Plan: {plan}, at .dev/{plan}/ (absolute: {plan_dir}). Attempt: {attempt}, the number run-state.py just printed for this launch. @@ -23,10 +27,11 @@ Guidance: {what to do differently, never "try again"}.} Scratch root for this attempt: /tmp/{project-slug}/factory/{plan}/{phase}-{attempt}/. Write your result to .dev/{plan}/results/{phase}-{attempt}.json as your last -action, per the schema in factory-run.md, echoing skill_path back exactly as -given above. +action: a JSON object with "schema": "factory.result/1", "phase", "status" +(done, failed or stopped), "reason" when not done, and skill_path echoed back +exactly as given above; the full schema is in factory-run.md. -This phase never launches another phase, and after the go it never asks a -person anything - decide under the unattended policy in factory-run.md and -record the decision in auto_decided. +Never name or launch the next phase. After the go you never ask a person +anything - decide under the unattended policy in factory-run.md and record +the decision in auto_decided. ``` From 671440f5777ff7c5936b43fde8c621cd933eb14a Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 09:24:16 +0200 Subject: [PATCH 22/30] refactor(factory): phases stop being invocable skills Change set 5 of config-driven-factory/spec.md: the four phase bodies move to factory/phases/ so run is the plugin's only invocable skill. The generator, F02 and F03, the path resolvers, docs and manifests follow the move, and F02 now holds every body and the launch wrapper to the result contract. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- .claude-plugin/marketplace.json | 2 +- CLAUDE.md | 8 +- README.md | 2 +- docs/architecture.md | 13 +-- factory/.claude-plugin/plugin.json | 2 +- factory/README.md | 10 +- factory/evals/README.md | 4 +- factory/evals/tests/test_factory_checks.py | 84 ++++++++++++++++- .../evals/tests/test_protocol_and_copies.py | 8 +- factory/{skills => phases}/build/SKILL.md | 0 .../build/references/e2e-report.md | 0 .../build/references/layers.md | 0 .../build/references/mocking.md | 0 .../build/references/parallel.md | 0 .../build/references/tests.md | 0 .../build/scripts/check-tests.py | 0 .../build/templates/e2e-report.html | 0 .../{skills => phases}/scope-review/SKILL.md | 0 .../scope-review/references/lenses.md | 0 factory/{skills => phases}/scope/SKILL.md | 0 .../scope/references/bootstrap.md | 0 .../scope/references/data-schema.md | 0 .../scope/references/reverse-mode.md | 0 .../scope/scripts/lint-spec.py | 0 .../scope/templates/spec.html | 0 factory/{skills => phases}/ship/SKILL.md | 0 .../ship/references/data-schema.md | 0 .../ship/references/gauntlet.md | 0 .../ship/references/lenses.md | 0 .../ship/references/orchestration-heavy.md | 0 .../ship/references/orchestration.md | 0 .../ship/references/pull-request.md | 0 .../ship/references/remediation.md | 0 .../ship/references/report-format.md | 0 .../ship/references/tools.md | 0 .../ship/scripts/aggregate-findings.py | 0 .../ship/scripts/pr-evidence.py | 0 .../ship/scripts/run-pi-agents.sh | 0 .../ship/templates/review.html | 0 factory/references/factory-run.md | 2 +- factory/references/pipeline-config.md | 2 +- factory/scripts/factory-config.py | 14 +-- plugins/factory/.codex-plugin/plugin.json | 6 +- .../factory/{skills => phases}/build/SKILL.md | 0 .../build/references/e2e-report.md | 0 .../build/references/layers.md | 0 .../build/references/mocking.md | 0 .../build/references/parallel.md | 0 .../build/references/tests.md | 0 .../build/scripts/check-tests.py | 0 .../build/templates/e2e-report.html | 0 .../{skills => phases}/scope-review/SKILL.md | 0 .../scope-review/references/lenses.md | 0 .../factory/{skills => phases}/scope/SKILL.md | 0 .../scope/references/bootstrap.md | 0 .../scope/references/data-schema.md | 0 .../scope/references/reverse-mode.md | 0 .../scope/scripts/lint-spec.py | 0 .../scope/templates/spec.html | 0 .../factory/{skills => phases}/ship/SKILL.md | 0 .../ship/references/data-schema.md | 0 .../ship/references/gauntlet.md | 0 .../ship/references/lenses.md | 0 .../ship/references/orchestration-heavy.md | 0 .../ship/references/orchestration.md | 0 .../ship/references/pull-request.md | 0 .../ship/references/remediation.md | 0 .../ship/references/report-format.md | 0 .../ship/references/tools.md | 0 .../ship/scripts/aggregate-findings.py | 0 .../ship/scripts/pr-evidence.py | 0 .../ship/scripts/run-pi-agents.sh | 0 .../ship/templates/review.html | 0 plugins/factory/references/factory-run.md | 2 +- plugins/factory/references/pipeline-config.md | 2 +- plugins/factory/scripts/factory-config.py | 14 +-- .../factory/skills/build/agents/openai.yaml | 6 -- plugins/factory/skills/run/agents/openai.yaml | 2 +- .../skills/scope-review/agents/openai.yaml | 6 -- .../factory/skills/scope/agents/openai.yaml | 6 -- .../factory/skills/ship/agents/openai.yaml | 6 -- scripts/build_codex_plugin.py | 36 +++----- scripts/validate.sh | 91 +++++++++++++++---- 83 files changed, 219 insertions(+), 109 deletions(-) rename factory/{skills => phases}/build/SKILL.md (100%) rename factory/{skills => phases}/build/references/e2e-report.md (100%) rename factory/{skills => phases}/build/references/layers.md (100%) rename factory/{skills => phases}/build/references/mocking.md (100%) rename factory/{skills => phases}/build/references/parallel.md (100%) rename factory/{skills => phases}/build/references/tests.md (100%) rename factory/{skills => phases}/build/scripts/check-tests.py (100%) rename factory/{skills => phases}/build/templates/e2e-report.html (100%) rename factory/{skills => phases}/scope-review/SKILL.md (100%) rename factory/{skills => phases}/scope-review/references/lenses.md (100%) rename factory/{skills => phases}/scope/SKILL.md (100%) rename factory/{skills => phases}/scope/references/bootstrap.md (100%) rename factory/{skills => phases}/scope/references/data-schema.md (100%) rename factory/{skills => phases}/scope/references/reverse-mode.md (100%) rename factory/{skills => phases}/scope/scripts/lint-spec.py (100%) rename factory/{skills => phases}/scope/templates/spec.html (100%) rename factory/{skills => phases}/ship/SKILL.md (100%) rename factory/{skills => phases}/ship/references/data-schema.md (100%) rename factory/{skills => phases}/ship/references/gauntlet.md (100%) rename factory/{skills => phases}/ship/references/lenses.md (100%) rename factory/{skills => phases}/ship/references/orchestration-heavy.md (100%) rename factory/{skills => phases}/ship/references/orchestration.md (100%) rename factory/{skills => phases}/ship/references/pull-request.md (100%) rename factory/{skills => phases}/ship/references/remediation.md (100%) rename factory/{skills => phases}/ship/references/report-format.md (100%) rename factory/{skills => phases}/ship/references/tools.md (100%) rename factory/{skills => phases}/ship/scripts/aggregate-findings.py (100%) rename factory/{skills => phases}/ship/scripts/pr-evidence.py (100%) rename factory/{skills => phases}/ship/scripts/run-pi-agents.sh (100%) rename factory/{skills => phases}/ship/templates/review.html (100%) rename plugins/factory/{skills => phases}/build/SKILL.md (100%) rename plugins/factory/{skills => phases}/build/references/e2e-report.md (100%) rename plugins/factory/{skills => phases}/build/references/layers.md (100%) rename plugins/factory/{skills => phases}/build/references/mocking.md (100%) rename plugins/factory/{skills => phases}/build/references/parallel.md (100%) rename plugins/factory/{skills => phases}/build/references/tests.md (100%) rename plugins/factory/{skills => phases}/build/scripts/check-tests.py (100%) rename plugins/factory/{skills => phases}/build/templates/e2e-report.html (100%) rename plugins/factory/{skills => phases}/scope-review/SKILL.md (100%) rename plugins/factory/{skills => phases}/scope-review/references/lenses.md (100%) rename plugins/factory/{skills => phases}/scope/SKILL.md (100%) rename plugins/factory/{skills => phases}/scope/references/bootstrap.md (100%) rename plugins/factory/{skills => phases}/scope/references/data-schema.md (100%) rename plugins/factory/{skills => phases}/scope/references/reverse-mode.md (100%) rename plugins/factory/{skills => phases}/scope/scripts/lint-spec.py (100%) rename plugins/factory/{skills => phases}/scope/templates/spec.html (100%) rename plugins/factory/{skills => phases}/ship/SKILL.md (100%) rename plugins/factory/{skills => phases}/ship/references/data-schema.md (100%) rename plugins/factory/{skills => phases}/ship/references/gauntlet.md (100%) rename plugins/factory/{skills => phases}/ship/references/lenses.md (100%) rename plugins/factory/{skills => phases}/ship/references/orchestration-heavy.md (100%) rename plugins/factory/{skills => phases}/ship/references/orchestration.md (100%) rename plugins/factory/{skills => phases}/ship/references/pull-request.md (100%) rename plugins/factory/{skills => phases}/ship/references/remediation.md (100%) rename plugins/factory/{skills => phases}/ship/references/report-format.md (100%) rename plugins/factory/{skills => phases}/ship/references/tools.md (100%) rename plugins/factory/{skills => phases}/ship/scripts/aggregate-findings.py (100%) rename plugins/factory/{skills => phases}/ship/scripts/pr-evidence.py (100%) rename plugins/factory/{skills => phases}/ship/scripts/run-pi-agents.sh (100%) rename plugins/factory/{skills => phases}/ship/templates/review.html (100%) delete mode 100644 plugins/factory/skills/build/agents/openai.yaml delete mode 100644 plugins/factory/skills/scope-review/agents/openai.yaml delete mode 100644 plugins/factory/skills/scope/agents/openai.yaml delete mode 100644 plugins/factory/skills/ship/agents/openai.yaml diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 8b3a74a..1b0725e 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -15,7 +15,7 @@ { "name": "factory", "source": "./factory", - "description": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself, ending in a pull request or a report." + "description": "An orchestrator skill that drives a pipeline of typed phases declared in .factory/config.yaml (the built-in scope, scope-review, build, and ship by default), judging each phase's completion itself, ending in a pull request or a report." } ] } diff --git a/CLAUDE.md b/CLAUDE.md index 278ee63..b405514 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -10,8 +10,10 @@ This is a monorepo for Tobrun's Claude Code, Codex, and Pi skills. The root `plugins/`. The root `package.json` exposes the `dev` source skills as a Pi package. There are two plugins: `dev`, the hand-invoked development workflow, and `factory`, whose `run` skill is the one sanctioned exception to "skills -never invoke each other" - it launches the copied `scope`, `scope-review`, -`build`, and `ship` phase skills by path and judges their completion itself. +never invoke each other" - it drives the pipeline declared in the consuming +repository's `.factory/config.yaml`, or the built-in `scope`, `scope-review`, +`build`, and `ship` phase bodies under `factory/phases/` by path, and judges +each phase's completion itself. `run` is the plugin's only invocable skill. ## Repository Structure @@ -48,7 +50,7 @@ never invoke each other" - it launches the copied `scope`, `scope-review`, - All changes must pass `scripts/validate.sh` before committing. - Every plugin directory name must match its `plugin.json` name and marketplace entry name. - Every `SKILL.md` must have YAML frontmatter with `name` and `description`. -- Every `SKILL.md` must set `disable-model-invocation: true`; all skills in this repo are human-triggered only, and skills recommend the next step instead of invoking each other - except the factory `run` skill, the one sanctioned invoker, which launches its copied phase skills by path. +- Every `SKILL.md` must set `disable-model-invocation: true`; all skills in this repo are human-triggered only, and skills recommend the next step instead of invoking each other - except the factory `run` skill, the one sanctioned invoker, which launches its phase bodies (`factory/phases/`, not skills) by path. - Plan files under `.dev/` are never committed. - Do not edit `plugins/` directly. Run `python3 scripts/build_codex_plugin.py` after changing `dev/`; the generator builds every configured diff --git a/README.md b/README.md index 462be10..195027c 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,7 @@ | Plugin | Use When | Tools | | ------ | -------- | ----- | | [dev](dev/) | A test-focused development workflow for Claude Code, Codex, opencode, and Pi. | `scope`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz` | -| [factory](factory/) | Take a request from scope to a shipped pull request unattended, on Claude Code or Codex. | `run`, `scope`, `scope-review`, `build`, `ship` | +| [factory](factory/) | Take a request from scope to a shipped pull request unattended, on Claude Code or Codex. | `run` | ## Claude Code diff --git a/docs/architecture.md b/docs/architecture.md index 906df2e..754647c 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -17,7 +17,7 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin ret | repo scripts | validation of the whole repository, the plugin generator, and the Pi transport self-test | `scripts/` | every other component | | harden tools | optional local analyses a ship gauntlet can draw on: added lines, coverage-weighted complexity, flaky reruns, mutation | `tools/harden/` | nothing; run by hand against a target repository | | research and todo | plans, findings, and reading notes | `research/`, `todo/` | nothing | -| factory plugin | the `run` orchestrator skill, the four phase copies, their references and scripts | `factory/skills/`, `factory/references/`, `factory/scripts/`, `factory/evals/` | the host's subagent tool, the consuming repository's `.dev/` | +| factory plugin | the `run` orchestrator skill (its only invocable skill), the four built-in phase bodies it reads by path, `factory/scripts/factory-config.py` for the pipeline config, their references and scripts | `factory/skills/`, `factory/phases/`, `factory/references/`, `factory/scripts/`, `factory/evals/` | the host's subagent tool, the consuming repository's .dev/ and .factory/ | ## Flows @@ -27,13 +27,14 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin ret 3. `dev/scripts/skill-metrics.py` measures each run and appends a row to the consuming repository's .dev/metrics.jsonl. ### Building the Codex distribution -1. `scripts/build_codex_plugin.py` copies skills/, references/, and `scripts/` of the source plugin into plugins/{name}/, strips Claude-only frontmatter, and writes agents/openai.yaml. +1. `scripts/build_codex_plugin.py` copies skills/, references/, and `scripts/` of the source plugin (and factory/phases/ for the factory) into plugins/{name}/, strips Claude-only frontmatter, and writes agents/openai.yaml for invocable skills. 2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), the Pi package and transport (P01), and the factory plugin's unattended wording and result protocol; `--check` mode of the generator fails when `plugins/` is stale. ### A factory run -1. A person invokes `/factory:run "a request"` (or a path to one); `run` scopes the change inline with them and ends with one go question. -2. From the go, `run` launches `scope-review`, `build`, and `ship` in turn as fresh-context subagents, judges each result's file against the phase's own deterministic checks, repairs or relaunches on failure, and never lets a phase invoke another. -3. The run ends in an open pull request, or a report naming the exact action a person could take - the pushed branch and the run's state file are what survives a closed session. +1. A person invokes `/factory:run "a request"` (or a path to one); preflight runs factory-config.py check, which classes each phase as built-in, injected or foreign and validates the pipeline from the consuming repository's .factory/config.yaml (the built-in default when there is none), and stops before the interview on any finding. +2. `run` runs the contiguous interactive prefix inline with them (`scope` in the default) and ends it with one go question. +3. From the go, `run` launches each remaining phase in turn as a fresh-context subagent, judges each result's file against its type's declared checks and outcome contract, repairs or relaunches on failure, and never lets a phase launch another. +4. The run ends in an open pull request, or a report naming the exact action a person could take - the pushed branch and the run's state file are what survives a closed session. ## Boundaries @@ -42,7 +43,7 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin ret | Claude Code, Codex, opencode, Pi | host | the skills | each reads the skills in its own format; the generated tree under `plugins/` serves Codex | | git and GitHub | external | the ship skill | branch state and `gh pr` calls made during a ship run | | Jira | external HTTP | dev skills | through `acli`, only when .dev/config.json enables it | -| consuming repository | store | the skills | `.dev/` plan files and `docs/` ledgers | +| consuming repository | store | the skills | `.dev/` plan files, `docs/` ledgers, and the .factory/ config and injected phase copies | ## Cross-cutting diff --git a/factory/.claude-plugin/plugin.json b/factory/.claude-plugin/plugin.json index 23e7118..006abdb 100644 --- a/factory/.claude-plugin/plugin.json +++ b/factory/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "factory", "version": "0.1.0", - "description": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", + "description": "An orchestrator skill that drives a pipeline of typed phases declared in .factory/config.yaml (the built-in scope, scope-review, build, and ship when there is none), running the interactive phases inline and the rest unattended, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", "author": { "name": "Tobrun" } diff --git a/factory/README.md b/factory/README.md index faa3e8b..c77121f 100644 --- a/factory/README.md +++ b/factory/README.md @@ -1,6 +1,6 @@ # factory -An orchestrator skill that takes a request through an interactive `scope`, then runs `scope-review`, `build`, and `ship` as unattended phases on the current checkout - judging each phase's completion itself, repairing or relaunching it on failure, and ending in a pull request or a report. No runner process to install: the whole thing is one skill, `run`, that launches subagents through the host's own subagent tool. +An orchestrator skill that drives a pipeline of typed phases on the current checkout - the interactive ones inline, the rest as unattended subagents - judging each phase's completion itself, repairing or relaunching it on failure, and ending in a pull request or a report. The pipeline is declared in `.factory/config.yaml` in your repository; with no config it runs our default, `scope`, `scope-review`, `build` and `ship`, from the installed plugin. No runner process to install: the whole thing is one skill, `run`, that launches subagents through the host's own subagent tool, so the plugin adds one skill to your namespace and no more. ## Install @@ -9,7 +9,7 @@ An orchestrator skill that takes a request through an interactive `scope`, then ## The run flow -Invoke `/factory:run "a quoted request"` or `/factory:run path/to/request.md`. `run` checks the transport, scopes the change with you inline (the only interactive phase), then ends with one go question. From there it launches `scope-review`, `build`, and `ship` in order, each as a fresh-context subagent, judging every result against the phase's own deterministic checks before advancing. A found secret or a destructive action outside the run branch stops the run immediately; anything else gets repaired, relaunched, or - after three attempts of one phase - ends the run with a report naming what a person could do. +Invoke `/factory:run "a quoted request"` or `/factory:run path/to/request.md`. `run` checks the transport and validates the pipeline with `factory-config.py check`, runs the interactive phases with you inline (`scope` in the default), then ends with one go question. From there it launches the remaining phases (`scope-review`, `build`, and `ship` in the default) in order, each as a fresh-context subagent, judging every result against its type's declared checks and outcome contract before advancing. A found secret or a destructive action outside the run branch stops the run immediately; anything else gets repaired, relaunched, or - after three attempts of one phase - ends the run with a report naming what a person could do. Invoke `/factory:run` with no request to resume: the plan on the checked-out branch, or the single unfinished state file under `.dev/`. @@ -23,3 +23,9 @@ Ignoring `.dev/` in the consuming repository is not a prerequisite: the run's ow ## Clear and resume The run lives as long as the session; `.dev/{plan}/factory-run.json` is what survives a closed one. Clear the session and invoke `/factory:run` again with no request to pick up where it left off, with a fresh context. + +## The pipeline config + +`python3 factory/scripts/factory-config.py init` writes a starter `.factory/config.yaml`; `show --resolved`, `check`, `set` and `unset` read and edit it, and you normally never edit the file by hand. Each phase names a `type` declared under `types`, and a type declares exactly five things: `interactive`, `requires` (the outcome, in prose), `checks` (argv commands that verify it), `seals` (artifacts watched for drift) and `attempts`. The full schema is in [references/pipeline-config.md](references/pipeline-config.md). + +A phase is one of three classes. A built-in phase runs from the installed plugin. An injected phase is a copy under `.factory/skills/` whose hash still matches `.factory/.inject.json`. Anything else is foreign, and a foreign phase must declare `unattended_safe: true`: nothing can detect a phase that blocks on a person, so that declaration is the word of whoever read the skill. diff --git a/factory/evals/README.md b/factory/evals/README.md index 345fa9c..8fe9b7e 100644 --- a/factory/evals/README.md +++ b/factory/evals/README.md @@ -36,13 +36,13 @@ One per host, before build's first commit: write into `webhook/__init__.py` a `G LC_ALL=C tr -dc 'A-Za-z0-9' str: def rename_result_literal(copy_root: Path) -> str: - relative = "factory/skills/build/SKILL.md" + relative = "factory/phases/build/SKILL.md" target = copy_root / relative text = target.read_text(encoding="utf-8") target.write_text( @@ -126,6 +126,32 @@ def rename_result_literal(copy_root: Path) -> str: return relative +def remove_text(relative: str, text: str) -> Callable[[Path], str]: + """A mutation that deletes every occurrence of `text` from `relative`.""" + + def mutate(copy_root: Path) -> str: + target = copy_root / relative + original = target.read_text(encoding="utf-8") + if text not in original: + raise AssertionError(f"expected {text!r} in {relative}") + target.write_text(original.replace(text, ""), encoding="utf-8") + return relative + + return mutate + + +def replace_text(relative: str, old: str, new: str) -> Callable[[Path], str]: + def mutate(copy_root: Path) -> str: + target = copy_root / relative + original = target.read_text(encoding="utf-8") + if old not in original: + raise AssertionError(f"expected {old!r} in {relative}") + target.write_text(original.replace(old, new), encoding="utf-8") + return relative + + return mutate + + def break_f03_self_test(copy_root: Path) -> str: relative = "scripts/validate.sh" target = copy_root / relative @@ -153,8 +179,11 @@ def break_f03_self_test(copy_root: Path) -> str: "This seam is worth agreeing on with the user.", ) -BUILD_SKILL = "factory/skills/build/SKILL.md" -SHIP_SKILL = "factory/skills/ship/SKILL.md" +SCOPE_SKILL = "factory/phases/scope/SKILL.md" +LAUNCH = "factory/skills/run/references/launch.md" +NEVER_LAUNCH = "Never name or launch the next phase." +BUILD_SKILL = "factory/phases/build/SKILL.md" +SHIP_SKILL = "factory/phases/ship/SKILL.md" RUN_SKILL = "factory/skills/run/SKILL.md" # name -> (mutation or None, rebuild the Codex copies before validating) @@ -204,6 +233,15 @@ def break_f03_self_test(copy_root: Path) -> str: True, ), "broken_self_test": (break_f03_self_test, False), + "scope_routed": (append(SCOPE_SKILL, "\n\nAsk the user which one to pick.\n"), True), + "sentence_removed": (remove_text(BUILD_SKILL, NEVER_LAUNCH), True), + "launch_result_renamed": (replace_text(LAUNCH, "factory.result/1", "factory.result/2"), True), + "launch_sentence_removed": (remove_text(LAUNCH, NEVER_LAUNCH), True), + "body_too_long": (append(SHIP_SKILL, "\nfiller line\n" * 210), True), + "no_invocation_policy": ( + remove_text(BUILD_SKILL, "disable-model-invocation: true\n"), + False, + ), "result_literal_renamed": (rename_result_literal, True), "phase_invocation": (append(BUILD_SKILL, "\n\nOn success, invoke $factory:ship.\n"), True), **{ @@ -340,6 +378,35 @@ def test_broken_self_test_regex_fails_f03(self) -> None: def test_missing_result_protocol_literal_fails_f02(self) -> None: self.assert_gate_fails("result_literal_renamed", "[F02]") + def test_person_routing_in_the_interactive_phase_is_outside_f03(self) -> None: + outcome = self.outcome("scope_routed") + self.assertEqual(outcome.returncode, 0, outcome.output) + self.assertNotIn("[F03]", outcome.stdout) + + def test_unmutated_moved_tree_scans_no_interactive_phase(self) -> None: + outcome = self.outcome("clean") + self.assertNotIn("factory/phases/scope", outcome.stdout) + + def test_phase_body_without_the_never_launch_sentence_fails_f02(self) -> None: + self.assert_gate_fails("sentence_removed", "[F02]") + + def test_launch_wrapper_without_the_result_schema_fails_f02(self) -> None: + self.assert_gate_fails("launch_result_renamed", "[F02]") + + def test_launch_wrapper_without_the_never_launch_sentence_fails_f02(self) -> None: + self.assert_gate_fails("launch_sentence_removed", "[F02]") + + def test_phase_body_over_the_length_limit_fails_naming_it(self) -> None: + outcome = self.assert_gate_fails("body_too_long", "[F02]") + self.assertIn("Body is", outcome.stdout) + + def test_phase_body_without_the_invocation_policy_fails_naming_it(self) -> None: + outcome = self.outcome("no_invocation_policy") + self.assertEqual(outcome.returncode, 1, outcome.output) + self.assertIn("[F02]", outcome.stdout) + self.assertIn(BUILD_SKILL, outcome.stdout) + self.assertIn("disable-model-invocation", outcome.stdout) + def test_invoking_another_phase_fails_f02(self) -> None: self.assert_gate_fails("phase_invocation", "[F02]") @@ -375,6 +442,15 @@ def test_build_codex_plugin_check_reports_up_to_date(self) -> None: self.assertEqual(result.returncode, 0, result.stdout + result.stderr) self.assertIn("up to date", result.stdout) + def test_no_claude_only_frontmatter_is_left_under_generated_phases(self) -> None: + rebuild(self.copy_root) + bodies = sorted((self.copy_root / "plugins" / "factory" / "phases").glob("*/SKILL.md")) + self.assertEqual(len(bodies), 4) + for body in bodies: + self.assertNotIn("disable-model-invocation", body.read_text(encoding="utf-8"), body) + skills = sorted((self.copy_root / "plugins" / "factory" / "skills").iterdir()) + self.assertEqual([path.name for path in skills], ["run"]) + def test_architecture_check_is_clean_on_the_updated_overview(self) -> None: result = subprocess.run( [sys.executable, "dev/scripts/architecture-check.py", "docs/architecture.md", "--root", "."], @@ -397,7 +473,7 @@ def test_phase_copies_stay_within_five_lines_of_dev(self) -> None: .splitlines() ) factory_lines = len( - (REPO_ROOT / "factory" / "skills" / skill / "SKILL.md") + (REPO_ROOT / "factory" / "phases" / skill / "SKILL.md") .read_text(encoding="utf-8") .splitlines() ) diff --git a/factory/evals/tests/test_protocol_and_copies.py b/factory/evals/tests/test_protocol_and_copies.py index 6ae61e5..58c4218 100644 --- a/factory/evals/tests/test_protocol_and_copies.py +++ b/factory/evals/tests/test_protocol_and_copies.py @@ -13,10 +13,10 @@ FACTORY = REPO_ROOT / "factory" SKILL_ROOTS = { - "{scope-skill-root}": FACTORY / "skills" / "scope", - "{scope-review-skill-root}": FACTORY / "skills" / "scope-review", - "{build-skill-root}": FACTORY / "skills" / "build", - "{ship-skill-root}": FACTORY / "skills" / "ship", + "{scope-skill-root}": FACTORY / "phases" / "scope", + "{scope-review-skill-root}": FACTORY / "phases" / "scope-review", + "{build-skill-root}": FACTORY / "phases" / "build", + "{ship-skill-root}": FACTORY / "phases" / "ship", "{run-skill-root}": FACTORY / "skills" / "run", } diff --git a/factory/skills/build/SKILL.md b/factory/phases/build/SKILL.md similarity index 100% rename from factory/skills/build/SKILL.md rename to factory/phases/build/SKILL.md diff --git a/factory/skills/build/references/e2e-report.md b/factory/phases/build/references/e2e-report.md similarity index 100% rename from factory/skills/build/references/e2e-report.md rename to factory/phases/build/references/e2e-report.md diff --git a/factory/skills/build/references/layers.md b/factory/phases/build/references/layers.md similarity index 100% rename from factory/skills/build/references/layers.md rename to factory/phases/build/references/layers.md diff --git a/factory/skills/build/references/mocking.md b/factory/phases/build/references/mocking.md similarity index 100% rename from factory/skills/build/references/mocking.md rename to factory/phases/build/references/mocking.md diff --git a/factory/skills/build/references/parallel.md b/factory/phases/build/references/parallel.md similarity index 100% rename from factory/skills/build/references/parallel.md rename to factory/phases/build/references/parallel.md diff --git a/factory/skills/build/references/tests.md b/factory/phases/build/references/tests.md similarity index 100% rename from factory/skills/build/references/tests.md rename to factory/phases/build/references/tests.md diff --git a/factory/skills/build/scripts/check-tests.py b/factory/phases/build/scripts/check-tests.py similarity index 100% rename from factory/skills/build/scripts/check-tests.py rename to factory/phases/build/scripts/check-tests.py diff --git a/factory/skills/build/templates/e2e-report.html b/factory/phases/build/templates/e2e-report.html similarity index 100% rename from factory/skills/build/templates/e2e-report.html rename to factory/phases/build/templates/e2e-report.html diff --git a/factory/skills/scope-review/SKILL.md b/factory/phases/scope-review/SKILL.md similarity index 100% rename from factory/skills/scope-review/SKILL.md rename to factory/phases/scope-review/SKILL.md diff --git a/factory/skills/scope-review/references/lenses.md b/factory/phases/scope-review/references/lenses.md similarity index 100% rename from factory/skills/scope-review/references/lenses.md rename to factory/phases/scope-review/references/lenses.md diff --git a/factory/skills/scope/SKILL.md b/factory/phases/scope/SKILL.md similarity index 100% rename from factory/skills/scope/SKILL.md rename to factory/phases/scope/SKILL.md diff --git a/factory/skills/scope/references/bootstrap.md b/factory/phases/scope/references/bootstrap.md similarity index 100% rename from factory/skills/scope/references/bootstrap.md rename to factory/phases/scope/references/bootstrap.md diff --git a/factory/skills/scope/references/data-schema.md b/factory/phases/scope/references/data-schema.md similarity index 100% rename from factory/skills/scope/references/data-schema.md rename to factory/phases/scope/references/data-schema.md diff --git a/factory/skills/scope/references/reverse-mode.md b/factory/phases/scope/references/reverse-mode.md similarity index 100% rename from factory/skills/scope/references/reverse-mode.md rename to factory/phases/scope/references/reverse-mode.md diff --git a/factory/skills/scope/scripts/lint-spec.py b/factory/phases/scope/scripts/lint-spec.py similarity index 100% rename from factory/skills/scope/scripts/lint-spec.py rename to factory/phases/scope/scripts/lint-spec.py diff --git a/factory/skills/scope/templates/spec.html b/factory/phases/scope/templates/spec.html similarity index 100% rename from factory/skills/scope/templates/spec.html rename to factory/phases/scope/templates/spec.html diff --git a/factory/skills/ship/SKILL.md b/factory/phases/ship/SKILL.md similarity index 100% rename from factory/skills/ship/SKILL.md rename to factory/phases/ship/SKILL.md diff --git a/factory/skills/ship/references/data-schema.md b/factory/phases/ship/references/data-schema.md similarity index 100% rename from factory/skills/ship/references/data-schema.md rename to factory/phases/ship/references/data-schema.md diff --git a/factory/skills/ship/references/gauntlet.md b/factory/phases/ship/references/gauntlet.md similarity index 100% rename from factory/skills/ship/references/gauntlet.md rename to factory/phases/ship/references/gauntlet.md diff --git a/factory/skills/ship/references/lenses.md b/factory/phases/ship/references/lenses.md similarity index 100% rename from factory/skills/ship/references/lenses.md rename to factory/phases/ship/references/lenses.md diff --git a/factory/skills/ship/references/orchestration-heavy.md b/factory/phases/ship/references/orchestration-heavy.md similarity index 100% rename from factory/skills/ship/references/orchestration-heavy.md rename to factory/phases/ship/references/orchestration-heavy.md diff --git a/factory/skills/ship/references/orchestration.md b/factory/phases/ship/references/orchestration.md similarity index 100% rename from factory/skills/ship/references/orchestration.md rename to factory/phases/ship/references/orchestration.md diff --git a/factory/skills/ship/references/pull-request.md b/factory/phases/ship/references/pull-request.md similarity index 100% rename from factory/skills/ship/references/pull-request.md rename to factory/phases/ship/references/pull-request.md diff --git a/factory/skills/ship/references/remediation.md b/factory/phases/ship/references/remediation.md similarity index 100% rename from factory/skills/ship/references/remediation.md rename to factory/phases/ship/references/remediation.md diff --git a/factory/skills/ship/references/report-format.md b/factory/phases/ship/references/report-format.md similarity index 100% rename from factory/skills/ship/references/report-format.md rename to factory/phases/ship/references/report-format.md diff --git a/factory/skills/ship/references/tools.md b/factory/phases/ship/references/tools.md similarity index 100% rename from factory/skills/ship/references/tools.md rename to factory/phases/ship/references/tools.md diff --git a/factory/skills/ship/scripts/aggregate-findings.py b/factory/phases/ship/scripts/aggregate-findings.py similarity index 100% rename from factory/skills/ship/scripts/aggregate-findings.py rename to factory/phases/ship/scripts/aggregate-findings.py diff --git a/factory/skills/ship/scripts/pr-evidence.py b/factory/phases/ship/scripts/pr-evidence.py similarity index 100% rename from factory/skills/ship/scripts/pr-evidence.py rename to factory/phases/ship/scripts/pr-evidence.py diff --git a/factory/skills/ship/scripts/run-pi-agents.sh b/factory/phases/ship/scripts/run-pi-agents.sh similarity index 100% rename from factory/skills/ship/scripts/run-pi-agents.sh rename to factory/phases/ship/scripts/run-pi-agents.sh diff --git a/factory/skills/ship/templates/review.html b/factory/phases/ship/templates/review.html similarity index 100% rename from factory/skills/ship/templates/review.html rename to factory/phases/ship/templates/review.html diff --git a/factory/references/factory-run.md b/factory/references/factory-run.md index 43cf943..517da29 100644 --- a/factory/references/factory-run.md +++ b/factory/references/factory-run.md @@ -21,7 +21,7 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las { "schema": "factory.result/1", "phase": "build", - "skill_path": "/abs/path/to/factory/skills/build/SKILL.md", + "skill_path": "/abs/path/to/factory/phases/build/SKILL.md", "status": "done", "reason": "", "artifacts": ["path/or/description"], diff --git a/factory/references/pipeline-config.md b/factory/references/pipeline-config.md index 2b75602..b6809e9 100644 --- a/factory/references/pipeline-config.md +++ b/factory/references/pipeline-config.md @@ -85,4 +85,4 @@ A list-valued key takes repeated `--item` flags in place of a positional value. ## The built-in default pipeline -Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/skills/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. +Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/phases/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. diff --git a/factory/scripts/factory-config.py b/factory/scripts/factory-config.py index a8caed2..cf7f568 100755 --- a/factory/scripts/factory-config.py +++ b/factory/scripts/factory-config.py @@ -56,7 +56,7 @@ "interactive": True, "requires": "the go was recorded (the interview happened inline in this session, " "so the orchestrator already knows)", - "checks": [["${plugin_root}/skills/scope/scripts/lint-spec.py", "${plan_dir}/spec.md"]], + "checks": [["${plugin_root}/phases/scope/scripts/lint-spec.py", "${plan_dir}/spec.md"]], "seals": ["${plan_dir}/spec.md"], "attempts": 3, }, @@ -71,7 +71,7 @@ "interactive": False, "requires": "the spec's Validation block is green, and commits since handoff " "match the change plan's numbering", - "checks": [["${plugin_root}/skills/build/scripts/check-tests.py", "${plan_dir}"]], + "checks": [["${plugin_root}/phases/build/scripts/check-tests.py", "${plan_dir}"]], "attempts": 3, }, "ship": { @@ -79,15 +79,15 @@ "requires": "review_N.md carries a verdict for HEAD and gh pr view shows the PR " "open with head equal to HEAD, or (without a GitHub remote) git ls-remote origin " "shows the branch at HEAD and pr.md is written", - "checks": [["${plugin_root}/skills/ship/scripts/pr-evidence.py", "check", "${plan_dir}/pr.md"]], + "checks": [["${plugin_root}/phases/ship/scripts/pr-evidence.py", "check", "${plan_dir}/pr.md"]], "attempts": 3, }, } BUILTIN_PHASES = [ - {"id": "scope", "type": "interview", "skill": "${plugin_root}/skills/scope/SKILL.md"}, - {"id": "scope-review", "type": "review", "skill": "${plugin_root}/skills/scope-review/SKILL.md"}, - {"id": "build", "type": "implement", "skill": "${plugin_root}/skills/build/SKILL.md"}, - {"id": "ship", "type": "ship", "skill": "${plugin_root}/skills/ship/SKILL.md"}, + {"id": "scope", "type": "interview", "skill": "${plugin_root}/phases/scope/SKILL.md"}, + {"id": "scope-review", "type": "review", "skill": "${plugin_root}/phases/scope-review/SKILL.md"}, + {"id": "build", "type": "implement", "skill": "${plugin_root}/phases/build/SKILL.md"}, + {"id": "ship", "type": "ship", "skill": "${plugin_root}/phases/ship/SKILL.md"}, ] BUILTIN_DEFAULTS = {"ceiling": 12} DEFAULT_ATTEMPTS = 3 diff --git a/plugins/factory/.codex-plugin/plugin.json b/plugins/factory/.codex-plugin/plugin.json index 1aa8b81..7842362 100644 --- a/plugins/factory/.codex-plugin/plugin.json +++ b/plugins/factory/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "factory", "version": "0.1.0", - "description": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", + "description": "An orchestrator skill that drives a pipeline of typed phases declared in .factory/config.yaml (the built-in scope, scope-review, build, and ship when there is none), running the interactive phases inline and the rest unattended, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", "author": { "name": "Tobrun" }, @@ -9,8 +9,8 @@ "skills": "./skills/", "interface": { "displayName": "Factory", - "shortDescription": "Run scope, scope-review, build, and ship unattended.", - "longDescription": "An orchestrator skill that takes a request through an interactive scope, then runs scope-review, build, and ship as unattended phases, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", + "shortDescription": "Drive a declared pipeline of phases unattended.", + "longDescription": "An orchestrator skill that drives a pipeline of typed phases declared in .factory/config.yaml (the built-in scope, scope-review, build, and ship when there is none), running the interactive phases inline and the rest unattended, judging each phase's completion itself and repairing or relaunching it on failure, ending in a pull request or a report. Use on Claude Code or Codex when you want a request carried through to a shipped change without a human at each phase.", "developerName": "Tobrun", "category": "Developer Tools", "capabilities": [ diff --git a/plugins/factory/skills/build/SKILL.md b/plugins/factory/phases/build/SKILL.md similarity index 100% rename from plugins/factory/skills/build/SKILL.md rename to plugins/factory/phases/build/SKILL.md diff --git a/plugins/factory/skills/build/references/e2e-report.md b/plugins/factory/phases/build/references/e2e-report.md similarity index 100% rename from plugins/factory/skills/build/references/e2e-report.md rename to plugins/factory/phases/build/references/e2e-report.md diff --git a/plugins/factory/skills/build/references/layers.md b/plugins/factory/phases/build/references/layers.md similarity index 100% rename from plugins/factory/skills/build/references/layers.md rename to plugins/factory/phases/build/references/layers.md diff --git a/plugins/factory/skills/build/references/mocking.md b/plugins/factory/phases/build/references/mocking.md similarity index 100% rename from plugins/factory/skills/build/references/mocking.md rename to plugins/factory/phases/build/references/mocking.md diff --git a/plugins/factory/skills/build/references/parallel.md b/plugins/factory/phases/build/references/parallel.md similarity index 100% rename from plugins/factory/skills/build/references/parallel.md rename to plugins/factory/phases/build/references/parallel.md diff --git a/plugins/factory/skills/build/references/tests.md b/plugins/factory/phases/build/references/tests.md similarity index 100% rename from plugins/factory/skills/build/references/tests.md rename to plugins/factory/phases/build/references/tests.md diff --git a/plugins/factory/skills/build/scripts/check-tests.py b/plugins/factory/phases/build/scripts/check-tests.py similarity index 100% rename from plugins/factory/skills/build/scripts/check-tests.py rename to plugins/factory/phases/build/scripts/check-tests.py diff --git a/plugins/factory/skills/build/templates/e2e-report.html b/plugins/factory/phases/build/templates/e2e-report.html similarity index 100% rename from plugins/factory/skills/build/templates/e2e-report.html rename to plugins/factory/phases/build/templates/e2e-report.html diff --git a/plugins/factory/skills/scope-review/SKILL.md b/plugins/factory/phases/scope-review/SKILL.md similarity index 100% rename from plugins/factory/skills/scope-review/SKILL.md rename to plugins/factory/phases/scope-review/SKILL.md diff --git a/plugins/factory/skills/scope-review/references/lenses.md b/plugins/factory/phases/scope-review/references/lenses.md similarity index 100% rename from plugins/factory/skills/scope-review/references/lenses.md rename to plugins/factory/phases/scope-review/references/lenses.md diff --git a/plugins/factory/skills/scope/SKILL.md b/plugins/factory/phases/scope/SKILL.md similarity index 100% rename from plugins/factory/skills/scope/SKILL.md rename to plugins/factory/phases/scope/SKILL.md diff --git a/plugins/factory/skills/scope/references/bootstrap.md b/plugins/factory/phases/scope/references/bootstrap.md similarity index 100% rename from plugins/factory/skills/scope/references/bootstrap.md rename to plugins/factory/phases/scope/references/bootstrap.md diff --git a/plugins/factory/skills/scope/references/data-schema.md b/plugins/factory/phases/scope/references/data-schema.md similarity index 100% rename from plugins/factory/skills/scope/references/data-schema.md rename to plugins/factory/phases/scope/references/data-schema.md diff --git a/plugins/factory/skills/scope/references/reverse-mode.md b/plugins/factory/phases/scope/references/reverse-mode.md similarity index 100% rename from plugins/factory/skills/scope/references/reverse-mode.md rename to plugins/factory/phases/scope/references/reverse-mode.md diff --git a/plugins/factory/skills/scope/scripts/lint-spec.py b/plugins/factory/phases/scope/scripts/lint-spec.py similarity index 100% rename from plugins/factory/skills/scope/scripts/lint-spec.py rename to plugins/factory/phases/scope/scripts/lint-spec.py diff --git a/plugins/factory/skills/scope/templates/spec.html b/plugins/factory/phases/scope/templates/spec.html similarity index 100% rename from plugins/factory/skills/scope/templates/spec.html rename to plugins/factory/phases/scope/templates/spec.html diff --git a/plugins/factory/skills/ship/SKILL.md b/plugins/factory/phases/ship/SKILL.md similarity index 100% rename from plugins/factory/skills/ship/SKILL.md rename to plugins/factory/phases/ship/SKILL.md diff --git a/plugins/factory/skills/ship/references/data-schema.md b/plugins/factory/phases/ship/references/data-schema.md similarity index 100% rename from plugins/factory/skills/ship/references/data-schema.md rename to plugins/factory/phases/ship/references/data-schema.md diff --git a/plugins/factory/skills/ship/references/gauntlet.md b/plugins/factory/phases/ship/references/gauntlet.md similarity index 100% rename from plugins/factory/skills/ship/references/gauntlet.md rename to plugins/factory/phases/ship/references/gauntlet.md diff --git a/plugins/factory/skills/ship/references/lenses.md b/plugins/factory/phases/ship/references/lenses.md similarity index 100% rename from plugins/factory/skills/ship/references/lenses.md rename to plugins/factory/phases/ship/references/lenses.md diff --git a/plugins/factory/skills/ship/references/orchestration-heavy.md b/plugins/factory/phases/ship/references/orchestration-heavy.md similarity index 100% rename from plugins/factory/skills/ship/references/orchestration-heavy.md rename to plugins/factory/phases/ship/references/orchestration-heavy.md diff --git a/plugins/factory/skills/ship/references/orchestration.md b/plugins/factory/phases/ship/references/orchestration.md similarity index 100% rename from plugins/factory/skills/ship/references/orchestration.md rename to plugins/factory/phases/ship/references/orchestration.md diff --git a/plugins/factory/skills/ship/references/pull-request.md b/plugins/factory/phases/ship/references/pull-request.md similarity index 100% rename from plugins/factory/skills/ship/references/pull-request.md rename to plugins/factory/phases/ship/references/pull-request.md diff --git a/plugins/factory/skills/ship/references/remediation.md b/plugins/factory/phases/ship/references/remediation.md similarity index 100% rename from plugins/factory/skills/ship/references/remediation.md rename to plugins/factory/phases/ship/references/remediation.md diff --git a/plugins/factory/skills/ship/references/report-format.md b/plugins/factory/phases/ship/references/report-format.md similarity index 100% rename from plugins/factory/skills/ship/references/report-format.md rename to plugins/factory/phases/ship/references/report-format.md diff --git a/plugins/factory/skills/ship/references/tools.md b/plugins/factory/phases/ship/references/tools.md similarity index 100% rename from plugins/factory/skills/ship/references/tools.md rename to plugins/factory/phases/ship/references/tools.md diff --git a/plugins/factory/skills/ship/scripts/aggregate-findings.py b/plugins/factory/phases/ship/scripts/aggregate-findings.py similarity index 100% rename from plugins/factory/skills/ship/scripts/aggregate-findings.py rename to plugins/factory/phases/ship/scripts/aggregate-findings.py diff --git a/plugins/factory/skills/ship/scripts/pr-evidence.py b/plugins/factory/phases/ship/scripts/pr-evidence.py similarity index 100% rename from plugins/factory/skills/ship/scripts/pr-evidence.py rename to plugins/factory/phases/ship/scripts/pr-evidence.py diff --git a/plugins/factory/skills/ship/scripts/run-pi-agents.sh b/plugins/factory/phases/ship/scripts/run-pi-agents.sh similarity index 100% rename from plugins/factory/skills/ship/scripts/run-pi-agents.sh rename to plugins/factory/phases/ship/scripts/run-pi-agents.sh diff --git a/plugins/factory/skills/ship/templates/review.html b/plugins/factory/phases/ship/templates/review.html similarity index 100% rename from plugins/factory/skills/ship/templates/review.html rename to plugins/factory/phases/ship/templates/review.html diff --git a/plugins/factory/references/factory-run.md b/plugins/factory/references/factory-run.md index 43cf943..517da29 100644 --- a/plugins/factory/references/factory-run.md +++ b/plugins/factory/references/factory-run.md @@ -21,7 +21,7 @@ Every phase skill writes `.dev/{plan}/results/{phase}-{attempt}.json` as its las { "schema": "factory.result/1", "phase": "build", - "skill_path": "/abs/path/to/factory/skills/build/SKILL.md", + "skill_path": "/abs/path/to/factory/phases/build/SKILL.md", "status": "done", "reason": "", "artifacts": ["path/or/description"], diff --git a/plugins/factory/references/pipeline-config.md b/plugins/factory/references/pipeline-config.md index 2b75602..b6809e9 100644 --- a/plugins/factory/references/pipeline-config.md +++ b/plugins/factory/references/pipeline-config.md @@ -85,4 +85,4 @@ A list-valued key takes repeated `--item` flags in place of a positional value. ## The built-in default pipeline -Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/skills/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. +Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/phases/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. diff --git a/plugins/factory/scripts/factory-config.py b/plugins/factory/scripts/factory-config.py index a8caed2..cf7f568 100755 --- a/plugins/factory/scripts/factory-config.py +++ b/plugins/factory/scripts/factory-config.py @@ -56,7 +56,7 @@ "interactive": True, "requires": "the go was recorded (the interview happened inline in this session, " "so the orchestrator already knows)", - "checks": [["${plugin_root}/skills/scope/scripts/lint-spec.py", "${plan_dir}/spec.md"]], + "checks": [["${plugin_root}/phases/scope/scripts/lint-spec.py", "${plan_dir}/spec.md"]], "seals": ["${plan_dir}/spec.md"], "attempts": 3, }, @@ -71,7 +71,7 @@ "interactive": False, "requires": "the spec's Validation block is green, and commits since handoff " "match the change plan's numbering", - "checks": [["${plugin_root}/skills/build/scripts/check-tests.py", "${plan_dir}"]], + "checks": [["${plugin_root}/phases/build/scripts/check-tests.py", "${plan_dir}"]], "attempts": 3, }, "ship": { @@ -79,15 +79,15 @@ "requires": "review_N.md carries a verdict for HEAD and gh pr view shows the PR " "open with head equal to HEAD, or (without a GitHub remote) git ls-remote origin " "shows the branch at HEAD and pr.md is written", - "checks": [["${plugin_root}/skills/ship/scripts/pr-evidence.py", "check", "${plan_dir}/pr.md"]], + "checks": [["${plugin_root}/phases/ship/scripts/pr-evidence.py", "check", "${plan_dir}/pr.md"]], "attempts": 3, }, } BUILTIN_PHASES = [ - {"id": "scope", "type": "interview", "skill": "${plugin_root}/skills/scope/SKILL.md"}, - {"id": "scope-review", "type": "review", "skill": "${plugin_root}/skills/scope-review/SKILL.md"}, - {"id": "build", "type": "implement", "skill": "${plugin_root}/skills/build/SKILL.md"}, - {"id": "ship", "type": "ship", "skill": "${plugin_root}/skills/ship/SKILL.md"}, + {"id": "scope", "type": "interview", "skill": "${plugin_root}/phases/scope/SKILL.md"}, + {"id": "scope-review", "type": "review", "skill": "${plugin_root}/phases/scope-review/SKILL.md"}, + {"id": "build", "type": "implement", "skill": "${plugin_root}/phases/build/SKILL.md"}, + {"id": "ship", "type": "ship", "skill": "${plugin_root}/phases/ship/SKILL.md"}, ] BUILTIN_DEFAULTS = {"ceiling": 12} DEFAULT_ATTEMPTS = 3 diff --git a/plugins/factory/skills/build/agents/openai.yaml b/plugins/factory/skills/build/agents/openai.yaml deleted file mode 100644 index 2b1ae55..0000000 --- a/plugins/factory/skills/build/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Build" - short_description: "Execute a spec test-first through e2e" - default_prompt: "Use $factory:build to execute the current spec test-first and verify it end to end." -policy: - allow_implicit_invocation: false diff --git a/plugins/factory/skills/run/agents/openai.yaml b/plugins/factory/skills/run/agents/openai.yaml index 317e24b..784fa1a 100644 --- a/plugins/factory/skills/run/agents/openai.yaml +++ b/plugins/factory/skills/run/agents/openai.yaml @@ -1,6 +1,6 @@ interface: display_name: "Run" short_description: "Take a request from scope to a shipped change unattended" - default_prompt: "Use $factory:run to take a request through scope, then run scope-review, build, and ship unattended." + default_prompt: "Use $factory:run to drive the pipeline declared in .factory/config.yaml, or the built-in scope, scope-review, build, and ship, unattended." policy: allow_implicit_invocation: false diff --git a/plugins/factory/skills/scope-review/agents/openai.yaml b/plugins/factory/skills/scope-review/agents/openai.yaml deleted file mode 100644 index c2a624a..0000000 --- a/plugins/factory/skills/scope-review/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Scope Review" - short_description: "Review and auto-refine a settled spec" - default_prompt: "Use $factory:scope-review to review the settled spec with a verified agent panel and refine it in place before building." -policy: - allow_implicit_invocation: false diff --git a/plugins/factory/skills/scope/agents/openai.yaml b/plugins/factory/skills/scope/agents/openai.yaml deleted file mode 100644 index 2d9fa40..0000000 --- a/plugins/factory/skills/scope/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Scope" - short_description: "Spec a change by arguing its decisions" - default_prompt: "Use $factory:scope to spec this change with argued decisions and a change plan." -policy: - allow_implicit_invocation: false diff --git a/plugins/factory/skills/ship/agents/openai.yaml b/plugins/factory/skills/ship/agents/openai.yaml deleted file mode 100644 index 249ecff..0000000 --- a/plugins/factory/skills/ship/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Ship" - short_description: "Harden, review, then open a PR with proof" - default_prompt: "Use $factory:ship to run the quality gauntlet, the verified review, and open the pull request with evidence for this change." -policy: - allow_implicit_invocation: false diff --git a/scripts/build_codex_plugin.py b/scripts/build_codex_plugin.py index 8a0cb0a..1b1469f 100644 --- a/scripts/build_codex_plugin.py +++ b/scripts/build_codex_plugin.py @@ -86,36 +86,17 @@ def destination(self) -> Path: "factory": PluginConfig( name="factory", display_name="Factory", - short_description="Run scope, scope-review, build, and ship unattended.", + short_description="Drive a declared pipeline of phases unattended.", capabilities=("Interactive", "Write"), default_prompts=( "Take this request through scope, then run it unattended to a pull request.", ), + copied_dirs=("skills", "phases", "references", "scripts"), skill_ui={ "run": ( "Run", "Take a request from scope to a shipped change unattended", - "Use $factory:run to take a request through scope, then run scope-review, build, and ship unattended.", - ), - "scope": ( - "Scope", - "Spec a change by arguing its decisions", - "Use $factory:scope to spec this change with argued decisions and a change plan.", - ), - "scope-review": ( - "Scope Review", - "Review and auto-refine a settled spec", - "Use $factory:scope-review to review the settled spec with a verified agent panel and refine it in place before building.", - ), - "build": ( - "Build", - "Execute a spec test-first through e2e", - "Use $factory:build to execute the current spec test-first and verify it end to end.", - ), - "ship": ( - "Ship", - "Harden, review, then open a PR with proof", - "Use $factory:ship to run the quality gauntlet, the verified review, and open the pull request with evidence for this change.", + "Use $factory:run to drive the pipeline declared in .factory/config.yaml, or the built-in scope, scope-review, build, and ship, unattended.", ), }, ), @@ -160,6 +141,15 @@ def openai_yaml(config: PluginConfig, skill_name: str) -> str: ) +def strip_phase_policy(destination: Path) -> None: + """Phase bodies are read by path, not invoked, so they need no UI entry - only the strip.""" + for skill_md in sorted((destination / "phases").glob("*/SKILL.md")): + skill_md.write_text( + codex_skill(skill_md.read_text(encoding="utf-8"), skill_md.parent.name), + encoding="utf-8", + ) + + def build(config: PluginConfig, destination: Path) -> None: source = config.source claude_manifest = json.loads( @@ -195,6 +185,8 @@ def build(config: PluginConfig, destination: Path) -> None: encoding="utf-8", ) + strip_phase_policy(destination) + manifest = { "name": config.name, "version": claude_manifest["version"], diff --git a/scripts/validate.sh b/scripts/validate.sh index 9686f8f..5b34b18 100755 --- a/scripts/validate.sh +++ b/scripts/validate.sh @@ -617,9 +617,9 @@ check_pi() { # =========================================================================== # F03: unattended factory material never routes a decision to a person # -# Scoped to factory/skills/{scope-review,build,ship,run} (SKILL.md and their -# references) plus factory/references/factory-run.md only - scope keeps its -# interview, and ci-parity.md, contracts.md, and jira.md keep their dev +# Scoped to factory/phases/* minus the built-in pipeline's interactive phases, +# plus factory/skills/run (SKILL.md and their references) and +# factory/references/factory-run.md only - scope keeps its interview, and ci-parity.md, contracts.md, and jira.md keep their dev # wording because they are dev-shared references whose human-call branches are # overridden for a factory run by factory-run.md's unattended policy. # @@ -694,6 +694,10 @@ CLAUSE = re.compile(r"[.;:!?]|\s-\s") OPEN, CLOSE = "", "" # Only the orchestrator has pre-go lines that legitimately reach a person. MARKER_ALLOWED = ("factory/skills/run/SKILL.md",) +# The built-in pipeline's interactive phases keep their interview and sit outside +# the scan: D-go-placement puts every interactive phase before the go, so it has +# no post-go decision to route. Update this tuple with the built-in pipeline. +INTERACTIVE_PHASES = ("scope",) def routes(line): @@ -799,8 +803,11 @@ for sample in NON_SAMPLES: sys.exit(0) paths = [] -for phase in ("scope-review", "build", "ship", "run"): - root = pathlib.Path("factory/skills") / phase +phase_roots = [ + d for d in sorted(pathlib.Path("factory/phases").glob("*")) + if d.is_dir() and d.name not in INTERACTIVE_PHASES +] +for root in phase_roots + [pathlib.Path("factory/skills/run")]: if root.is_dir(): paths.extend(sorted(root.rglob("*.md"))) run_md = pathlib.Path("factory/references/factory-run.md") @@ -863,33 +870,83 @@ check_factory_protocol() { found=$(python3 - <<'PYEOF' import pathlib, re, sys -PHASES = ("scope", "scope-review", "build", "ship") +BUILT_IN = ("scope", "scope-review", "build", "ship") INVOKE = re.compile(r"\$factory:([a-z-]+)|/SKILL\.md\b") +RESULTS_PATH = "results/{phase}-{attempt}.json" +NEVER_LAUNCH = "Never name or launch the next phase." +MAX_LINES = 200 + + +def frontmatter(text): + """The frontmatter's key/value pairs, or None when the file has none.""" + lines = text.splitlines() + if not lines or lines[0].strip() != "---": + return None + fields = {} + for line in lines[1:]: + if line.strip() == "---": + return fields + key, sep, value = line.partition(":") + if sep: + fields[key.strip()] = value.strip() + return None + -for phase in PHASES: - sf = pathlib.Path("factory/skills") / phase / "SKILL.md" +def body_rules(sf, phase): + """The generic skill rules that reached these bodies while they lived under skills/.""" + text = sf.read_text(encoding="utf-8") + total = len(text.splitlines()) + if total > MAX_LINES: + print(f"{sf}: Body is {total} lines (recommend under 150; move detail to references/)") + fields = frontmatter(text) + if fields is None: + print(f"{sf}: Does not start with a complete --- frontmatter block") + return text + if not fields.get("name") or not fields.get("description"): + print(f"{sf}: Frontmatter 'name' or 'description' is empty or missing") + if fields.get("name") and fields["name"] != phase: + print(f"{sf}: Frontmatter name '{fields['name']}' != directory name '{phase}'") + if fields.get("disable-model-invocation") != "true": + print(f"{sf}: Frontmatter must set 'disable-model-invocation: true'") + return text + + +phases_dir = pathlib.Path("factory/phases") +names = sorted(d.name for d in phases_dir.glob("*") if d.is_dir()) if phases_dir.is_dir() else [] +for phase in BUILT_IN: + if phase not in names: + print(f"factory/phases/{phase}/SKILL.md: Factory phase skill not found") +for phase in names: + sf = phases_dir / phase / "SKILL.md" if not sf.is_file(): - print(f"{sf}: Factory phase skill not found") + print(f"{phases_dir / phase}: Missing SKILL.md") continue - text = sf.read_text(encoding="utf-8") + text = body_rules(sf, phase) if "factory-run.json" not in text: print(f"{sf}: Must mention factory-run.json") - if "results/{phase}-{attempt}.json" not in text: - print(f"{sf}: Must mention the literal results/{{phase}}-{{attempt}}.json") + if RESULTS_PATH not in text: + print(f"{sf}: Must mention the literal {RESULTS_PATH}") + if NEVER_LAUNCH not in text: + print(f"{sf}: Must carry the sentence \"{NEVER_LAUNCH}\"") for match in INVOKE.finditer(text): skill = match.group(1) if skill is None or skill != phase: print(f"{sf}: invokes another phase ({match.group(0)}); a phase skill never launches another") +launch_md = pathlib.Path("factory/skills/run/references/launch.md") +if not launch_md.is_file(): + print(f"{launch_md}: launch wrapper not found") +else: + launch_text = launch_md.read_text(encoding="utf-8") + for literal in (RESULTS_PATH, "factory.result/1", NEVER_LAUNCH): + if literal not in launch_text: + print(f"{launch_md}: Must carry the literal {literal}") + run_md = pathlib.Path("factory/skills/run/SKILL.md") if not run_md.is_file(): print(f"{run_md}: run orchestrator skill not found") sys.exit(0) -run_text = run_md.read_text(encoding="utf-8") -for phase in PHASES: - if phase not in run_text: - print(f"{run_md}: must name phase '{phase}'") -if "run-state.py" not in run_text: +if "run-state.py" not in run_md.read_text(encoding="utf-8"): print(f"{run_md}: must name run-state.py") PYEOF ) From c93ac9c37cb8bfcbd80267bf4f2b7cbd4e65d73a Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 09:30:16 +0200 Subject: [PATCH 23/30] feat(factory): inject phase bodies into a consuming repository Change set 6 of config-driven-factory/spec.md: factory-config.py inject copies named phases and what they reference into .factory/, records provenance and sha256s in .factory/.inject.json, refuses to overwrite edited copies without --force, and points the config at the copies. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- factory/README.md | 8 + factory/evals/tests/test_factory_checks.py | 28 +++ factory/evals/tests/test_factory_config.py | 164 +++++++++++++ factory/references/pipeline-config.md | 6 + factory/scripts/factory-config.py | 226 +++++++++++++++++- plugins/factory/references/pipeline-config.md | 6 + plugins/factory/scripts/factory-config.py | 226 +++++++++++++++++- 7 files changed, 662 insertions(+), 2 deletions(-) diff --git a/factory/README.md b/factory/README.md index c77121f..74aecaa 100644 --- a/factory/README.md +++ b/factory/README.md @@ -29,3 +29,11 @@ The run lives as long as the session; `.dev/{plan}/factory-run.json` is what sur `python3 factory/scripts/factory-config.py init` writes a starter `.factory/config.yaml`; `show --resolved`, `check`, `set` and `unset` read and edit it, and you normally never edit the file by hand. Each phase names a `type` declared under `types`, and a type declares exactly five things: `interactive`, `requires` (the outcome, in prose), `checks` (argv commands that verify it), `seals` (artifacts watched for drift) and `attempts`. The full schema is in [references/pipeline-config.md](references/pipeline-config.md). A phase is one of three classes. A built-in phase runs from the installed plugin. An injected phase is a copy under `.factory/skills/` whose hash still matches `.factory/.inject.json`. Anything else is foreign, and a foreign phase must declare `unattended_safe: true`: nothing can detect a phase that blocks on a person, so that declaration is the word of whoever read the skill. + +## Injecting our phases into your repository + +By default nothing of ours is copied into your repository: the default pipeline runs from the installed plugin, so a team with its own spec-driven method is never asked to carry ours. Run `python3 factory/scripts/factory-config.py inject [phase ...]` when you want editable copies of `scope`, `scope-review`, `build` or `ship` (all four when none is named). It copies each named phase plus the references, scripts and templates it links into `.factory/skills/`, `.factory/references/` and `.factory/scripts/`, records a sha256 per file in `.factory/.inject.json`, and writes or updates `.factory/config.yaml` so the phase points at its copy. + +The copies are ordinary files in your repository, so they enter the run branch's diff: `ship`'s gauntlet and review panel review them as part of the change, and `build`'s commit check sees an inject commit the change plan does not list. Commit them on their own first. + +An unedited copy takes a plugin upgrade the next time you run `inject`. An edited copy is refused, naming the file, until you pass `--force`, and a copy you edited counts as foreign in `check`: it then needs `unattended_safe: true` and is never hand-repaired by the orchestrator. diff --git a/factory/evals/tests/test_factory_checks.py b/factory/evals/tests/test_factory_checks.py index 1ae7118..71eb90d 100644 --- a/factory/evals/tests/test_factory_checks.py +++ b/factory/evals/tests/test_factory_checks.py @@ -24,6 +24,8 @@ from __future__ import annotations +import hashlib +import json import os import shutil import subprocess @@ -442,6 +444,32 @@ def test_build_codex_plugin_check_reports_up_to_date(self) -> None: self.assertEqual(result.returncode, 0, result.stdout + result.stderr) self.assertIn("up to date", result.stdout) + def test_inject_upgrades_an_unedited_copy_without_force(self) -> None: + """A plugin upgrade (a mutated body under the copy's plugin root) reaches an unedited copy.""" + script = self.copy_root / "factory" / "scripts" / "factory-config.py" + consumer = self.copy_root.parent / "consumer" + consumer.mkdir() + + def inject(*args: str) -> subprocess.CompletedProcess: + return subprocess.run( + [sys.executable, str(script), "--repo-root", str(consumer), "inject", *args], + check=False, + capture_output=True, + text=True, + ) + + self.assertEqual(inject("scope").returncode, 0) + body = self.copy_root / "factory" / "phases" / "scope" / "SKILL.md" + body.write_text(body.read_text(encoding="utf-8") + "\nAn upgraded line.\n", encoding="utf-8") + result = inject("scope") + self.assertEqual(result.returncode, 0, result.stderr) + copy = consumer / ".factory" / "skills" / "scope" / "SKILL.md" + self.assertIn("An upgraded line.", copy.read_text(encoding="utf-8")) + manifest = json.loads((consumer / ".factory" / ".inject.json").read_text(encoding="utf-8")) + self.assertEqual( + manifest["files"]["skills/scope/SKILL.md"], hashlib.sha256(copy.read_bytes()).hexdigest() + ) + def test_no_claude_only_frontmatter_is_left_under_generated_phases(self) -> None: rebuild(self.copy_root) bodies = sorted((self.copy_root / "plugins" / "factory" / "phases").glob("*/SKILL.md")) diff --git a/factory/evals/tests/test_factory_config.py b/factory/evals/tests/test_factory_config.py index 390426d..8cd29ea 100644 --- a/factory/evals/tests/test_factory_config.py +++ b/factory/evals/tests/test_factory_config.py @@ -8,6 +8,7 @@ import hashlib import json +import re import subprocess import sys import tempfile @@ -620,5 +621,168 @@ def test_show_resolved_json_shape(self) -> None: self.assertEqual([p["id"] for p in data["phases"]], ["scope", "scope-review", "build", "ship"]) +LINK_RE = re.compile(r"\]\(([^)#\s]+)(?:#[^)]*)?\)") +BACKTICKED = re.compile(r"`([^`\n]+)`") + + +class InjectTest(TempRepoTestCase): + def inject(self, *args: str) -> subprocess.CompletedProcess: + return run(self.repo, "inject", *args) + + def manifest(self) -> dict: + return json.loads((self.repo / ".factory" / ".inject.json").read_text()) + + def copy_of(self, rel: str) -> Path: + return self.repo / ".factory" / rel + + def test_inject_scope_into_a_repo_with_no_config(self) -> None: + result = self.inject("scope") + self.assertEqual(result.returncode, 0, result.stderr) + body = self.copy_of("skills/scope/SKILL.md") + self.assertTrue(body.is_file()) + self.assertIn("injected-from: factory@", body.read_text()) + self.assertTrue(self.copy_of("references/plan-layout.md").is_file()) + self.assertTrue(self.copy_of("scripts/skill-metrics.py").is_file()) + manifest = self.manifest() + self.assertEqual(manifest["plugin"], "factory") + copied = { + p.relative_to(self.repo / ".factory").as_posix() + for p in (self.repo / ".factory").rglob("*") + if p.is_file() and p.name not in (".inject.json", "config.yaml") + } + self.assertEqual(set(manifest["files"]), copied) + for rel, digest in manifest["files"].items(): + self.assertEqual(hashlib.sha256(self.copy_of(rel).read_bytes()).hexdigest(), digest, rel) + config = (self.repo / ".factory" / "config.yaml").read_text() + self.assertIn("${repo_root}/.factory/skills/scope/SKILL.md", config) + self.assertEqual(run(self.repo, "check").returncode, 0) + + def test_only_the_skill_root_placeholders_are_rewritten(self) -> None: + self.inject("scope") + text = self.copy_of("skills/scope/SKILL.md").read_text() + self.assertNotIn("{scope-skill-root}", text) + self.assertIn(".factory/skills/scope/scripts/lint-spec.py", text) + self.assertIn(".factory/skills/scope/../../scripts/skill-metrics.py", text) + + def test_cross_phase_references_pull_the_referenced_files(self) -> None: + self.inject("scope-review") + self.assertTrue(self.copy_of("skills/ship/scripts/aggregate-findings.py").is_file()) + self.assertTrue(self.copy_of("skills/scope/scripts/lint-spec.py").is_file()) + + def test_a_template_comes_along_without_a_header(self) -> None: + self.inject("build") + template = self.copy_of("skills/build/templates/e2e-report.html") + self.assertTrue(template.is_file()) + self.assertNotIn("injected-from", template.read_text()) + + def test_inject_into_an_existing_config_repoints_only_that_phase(self) -> None: + self.inject("build") + before = json.loads(run(self.repo, "show", "--resolved", "--json").stdout)["phases"] + self.assertEqual(self.inject("scope").returncode, 0) + after = json.loads(run(self.repo, "show", "--resolved", "--json").stdout)["phases"] + changed = [(a["id"], a["skill"]) for a, b in zip(after, before) if a != b] + self.assertEqual([phase for phase, _ in changed], ["scope"]) + self.assertTrue(changed[0][1].endswith(".factory/skills/scope/SKILL.md")) + + def test_check_after_inject_needs_no_unattended_safe(self) -> None: + self.inject("build") + self.assertEqual(run(self.repo, "check").returncode, 0) + self.assertNotIn("unattended_safe", (self.repo / ".factory" / "config.yaml").read_text()) + + def test_inject_twice_is_a_no_op(self) -> None: + self.inject("scope") + before = {p: p.read_bytes() for p in (self.repo / ".factory").rglob("*") if p.is_file()} + second = self.inject("scope") + self.assertEqual(second.returncode, 0) + self.assertIn("nothing to do", second.stdout) + after = {p: p.read_bytes() for p in (self.repo / ".factory").rglob("*") if p.is_file()} + self.assertEqual(before, after) + + def test_an_edited_copy_is_refused_naming_the_file(self) -> None: + self.inject("scope") + body = self.copy_of("skills/scope/SKILL.md") + body.write_text(body.read_text() + "\nlocal edit\n") + result = self.inject("scope") + self.assertEqual(result.returncode, 1) + self.assertIn("skills/scope/SKILL.md", result.stderr) + self.assertIn("local edit", body.read_text()) + + def test_force_overwrites_an_edited_copy_and_updates_the_manifest(self) -> None: + self.inject("scope") + body = self.copy_of("skills/scope/SKILL.md") + body.write_text(body.read_text() + "\nlocal edit\n") + result = self.inject("scope", "--force") + self.assertEqual(result.returncode, 0, result.stderr) + self.assertNotIn("local edit", body.read_text()) + self.assertEqual( + self.manifest()["files"]["skills/scope/SKILL.md"], + hashlib.sha256(body.read_bytes()).hexdigest(), + ) + + def test_check_after_editing_an_injected_copy_names_the_file_and_phase(self) -> None: + self.inject("build") + body = self.copy_of("skills/build/SKILL.md") + body.write_text(body.read_text() + "\nlocal edit\n") + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("skills/build/SKILL.md", result.stdout) + self.assertIn("'build'", result.stdout) + + def test_an_existing_copy_with_no_manifest_is_refused(self) -> None: + make_skill(self.repo, ".factory/skills/scope/SKILL.md", "mine\n") + result = self.inject("scope") + self.assertEqual(result.returncode, 1) + self.assertIn("skills/scope/SKILL.md", result.stderr) + self.assertIn(".inject.json", result.stderr) + self.assertEqual(self.copy_of("skills/scope/SKILL.md").read_text(), "mine\n") + self.assertFalse((self.repo / ".factory" / ".inject.json").exists()) + + def test_force_over_a_copy_with_no_manifest_writes_the_manifest(self) -> None: + make_skill(self.repo, ".factory/skills/scope/SKILL.md", "mine\n") + self.assertEqual(self.inject("scope", "--force").returncode, 0) + self.assertIn("injected-from", self.copy_of("skills/scope/SKILL.md").read_text()) + self.assertIn("skills/scope/SKILL.md", self.manifest()["files"]) + + def test_an_unknown_phase_is_a_bad_call(self) -> None: + result = self.inject("deploy") + self.assertEqual(result.returncode, 3) + self.assertIn("deploy", result.stderr) + + def test_inject_with_no_phase_installs_the_defaults(self) -> None: + self.assertEqual(self.inject().returncode, 0) + for phase in ("scope", "scope-review", "build", "ship"): + self.assertTrue(self.copy_of(f"skills/{phase}/SKILL.md").is_file(), phase) + self.assertEqual(run(self.repo, "check").returncode, 0) + + def test_every_relative_link_under_factory_resolves_to_a_listed_file(self) -> None: + self.inject() + listed = set(self.manifest()["files"]) + root = self.repo / ".factory" + problems = [] + for path in sorted(root.rglob("*.md")): + for target in LINK_RE.findall(path.read_text()): + if target.startswith(("http://", "https://", "mailto:", "/")): + continue + resolved = (path.parent / target).resolve() + rel = resolved.relative_to(root.resolve()).as_posix() if root.resolve() in resolved.parents else None + if rel not in listed: + problems.append(f"{path.relative_to(root)} -> {target}") + self.assertEqual(problems, []) + + def test_every_factory_skills_path_in_a_backticked_span_resolves_from_the_repo_root(self) -> None: + self.inject() + problems, checked = [], 0 + for path in sorted((self.repo / ".factory").rglob("*.md")): + for span in BACKTICKED.findall(path.read_text()): + for token in span.split(): + if not token.startswith(".factory/skills/"): + continue + checked += 1 + if not (self.repo / token).resolve().is_file(): + problems.append(f"{path.relative_to(self.repo)}: {token}") + self.assertGreater(checked, 0) + self.assertEqual(problems, []) + + if __name__ == "__main__": unittest.main() diff --git a/factory/references/pipeline-config.md b/factory/references/pipeline-config.md index b6809e9..eadef79 100644 --- a/factory/references/pipeline-config.md +++ b/factory/references/pipeline-config.md @@ -86,3 +86,9 @@ A list-valued key takes repeated `--item` flags in place of a positional value. ## The built-in default pipeline Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/phases/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. + +## `inject` + +`factory-config.py inject [phase ...] [--force]` copies the named phase bodies (all four when none is named) and the transitive closure of what they reference into `.factory/`, keeping the plugin's relative depth so no link needs rewriting: `phases/{phase}/` lands at `.factory/skills/{phase}/`, `references/` at `.factory/references/` and `scripts/` at `.factory/scripts/`. The closure follows three forms in Markdown files: relative links, `{x-skill-root}/...` paths and backticked `templates/*.html` paths. The only rewrite is each `{x-skill-root}` placeholder, which becomes the copy's own `.factory/skills/{x}` path relative to the repository root. Each copied `SKILL.md` gains an `injected-from: {plugin}@{version}` frontmatter key; no other file carries a header, and `.factory/.inject.json` is the provenance for the rest. + +A target that exists is refused, naming the file, unless `--force`, when its sha256 differs from its manifest entry, or when the manifest is missing, unparseable or has no entry for it. An unedited copy takes an upgrade without `--force`. `inject` also writes or updates `.factory/config.yaml` so each injected phase entry points at `${repo_root}/.factory/skills/{phase}/SKILL.md`, creating the config from the built-in default when there is none, and never sets `unattended_safe`, since a copy matching its manifest is injected rather than foreign. diff --git a/factory/scripts/factory-config.py b/factory/scripts/factory-config.py index cf7f568..49832cc 100755 --- a/factory/scripts/factory-config.py +++ b/factory/scripts/factory-config.py @@ -7,6 +7,7 @@ factory-config.py check factory-config.py set [--item ...] factory-config.py unset + factory-config.py inject [phase ...] [--force] Exit codes: init 0 written; 1 the file already exists @@ -16,6 +17,9 @@ 3 a malformed path or an unsupported document construct unset 0 recorded (a no-op if the key was already absent); 3 a malformed path + inject 0 copied, or nothing to do; 1 a target was edited since it was + injected or has no manifest entry (use --force); 3 an unknown phase + or an unusable config 3 always means the call itself could not be carried out - a malformed key path, an unparseable document, or a usage error - never a finding about the @@ -27,8 +31,11 @@ from __future__ import annotations import argparse +import copy import hashlib import json +import re +import shutil import sys from pathlib import Path from typing import NoReturn @@ -280,7 +287,7 @@ def _needs_quotes(text: str) -> bool: return True if text != text.strip(): return True - if ": " in text or text.endswith(":"): + if ": " in text or text.endswith(":") or " #" in text: return True if text[0] in "\"'[]{}&*|>#-?:@`": return True @@ -750,6 +757,222 @@ def render_json(resolved: dict) -> str: return json.dumps({"phases": phases}, indent=2) + "\n" +# --------------------------------------------------------------------------- +# inject: copy phase bodies and what they reference into .factory/ +# --------------------------------------------------------------------------- + +# A plugin directory and where its copy lands under .factory/. The plugin's +# relative depth is preserved, so no link in a copied file needs rewriting. +INJECT_MAP = {"phases": "skills", "references": "references", "scripts": "scripts"} +LINK_RE = re.compile(r"\]\(([^)#\s]+)(?:#[^)]*)?\)") +SKILL_ROOT_RE = re.compile(r"\{([a-z][a-z-]*)-skill-root\}((?:/[\w.-]+)*)") +TEMPLATE_RE = re.compile(r"`(templates/[\w./-]+\.html)`") +PROVENANCE_KEY = "injected-from" + + +def _owning_phase(plugin: Path, source: Path): + try: + parts = source.relative_to(plugin / "phases").parts + except ValueError: + return None + return parts[0] if len(parts) > 1 else None + + +def _referenced_paths(plugin: Path, source: Path) -> list: + """Files a Markdown file names by a relative link, a skill-root path or a template.""" + text = source.read_text(encoding="utf-8") + found = [] + for target in LINK_RE.findall(text): + if not target.startswith(("http://", "https://", "mailto:", "/")): + found.append(source.parent / target) + for phase, suffix in SKILL_ROOT_RE.findall(text): + found.append(plugin / "phases" / phase / suffix.rstrip(".,").lstrip("/")) + owner = _owning_phase(plugin, source) + if owner is not None: + found.extend(plugin / "phases" / owner / tpl for tpl in TEMPLATE_RE.findall(text)) + return found + + +def _injectable(plugin: Path, path: Path) -> bool: + try: + top = path.relative_to(plugin).parts[0] + except (ValueError, IndexError): + return False + return top in INJECT_MAP and path.is_file() + + +def inject_closure(plugin: Path, phases: list) -> list: + """Every file the named phase bodies reach, transitively, in a stable order.""" + seen: dict = {} + queue = [(plugin / "phases" / phase / "SKILL.md").resolve() for phase in phases] + while queue: + source = queue.pop(0) + if source in seen or not _injectable(plugin, source): + continue + seen[source] = None + if source.suffix == ".md": + queue.extend(p.resolve() for p in _referenced_paths(plugin, source)) + return sorted(seen) + + +def inject_destination(plugin: Path, source: Path) -> str: + """The path under .factory/ a plugin file lands at.""" + top, *rest = source.relative_to(plugin).parts + return "/".join([INJECT_MAP[top], *rest]) + + +def plugin_identity(plugin: Path) -> tuple: + try: + meta = json.loads((plugin / ".claude-plugin" / "plugin.json").read_text(encoding="utf-8")) + return str(meta.get("name", plugin.name)), str(meta.get("version", "unknown")) + except (OSError, json.JSONDecodeError): + return plugin.name, "unknown" + + +def add_provenance(text: str, origin: str) -> str: + """Add the provenance key to a SKILL.md's frontmatter, before its closing rule.""" + lines = text.split("\n") + if not lines or lines[0].strip() != "---": + return text + for index in range(1, len(lines)): + if lines[index].strip() == "---": + lines.insert(index, f"{PROVENANCE_KEY}: {origin}") + break + return "\n".join(lines) + + +def inject_content(plugin: Path, source: Path, phases: set, origin: str) -> bytes: + """A file's bytes as injected: skill-root placeholders resolved, provenance added.""" + if source.suffix != ".md": + return source.read_bytes() + text = source.read_text(encoding="utf-8") + text = SKILL_ROOT_RE.sub( + lambda m: f".factory/skills/{m.group(1)}{m.group(2)}" if m.group(1) in phases else m.group(0), + text, + ) + if source.name == "SKILL.md" and _owning_phase(plugin, source) is not None: + text = add_provenance(text, origin) + return text.encode("utf-8") + + +def load_manifest(repo_root: Path): + """Return (manifest dict, reason) - the dict is None with a reason it is unusable.""" + path = manifest_path(repo_root) + if not path.exists(): + return None, "is missing" + try: + manifest = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None, "is unparseable" + if not isinstance(manifest, dict) or not isinstance(manifest.get("files"), dict): + return None, "is unparseable" + return manifest, "" + + +def classify_target(target: Path, rel: str, content: bytes, manifest, reason: str): + """One of 'new', 'unchanged', 'upgrade' or a refusal message for a file to inject.""" + if not target.exists(): + return "new" + current = hashlib.sha256(target.read_bytes()).hexdigest() + entry = manifest["files"].get(rel) if manifest is not None else None + if manifest is None: + return f"{rel} exists and .factory/.inject.json {reason}" + if entry is None: + return f"{rel} exists and has no entry in .factory/.inject.json" + if current != entry: + return f"{rel} was edited since it was injected" + return "unchanged" if current == hashlib.sha256(content).hexdigest() else "upgrade" + + +def plan_inject(repo_root: Path, plugin: Path, phases: list, force: bool): + """Return (writes, unchanged, refusals): file plans keyed by destination path.""" + manifest, reason = load_manifest(repo_root) + origin = "%s@%s" % plugin_identity(plugin) + phase_names = {p.name for p in (plugin / "phases").iterdir() if p.is_dir()} + writes, unchanged, refusals = {}, {}, [] + for source in inject_closure(plugin, phases): + rel = inject_destination(plugin, source) + content = inject_content(plugin, source, phase_names, origin) + state = classify_target(repo_root / CONFIG_DIR / rel, rel, content, manifest, reason) + if state == "unchanged": + unchanged[rel] = (content, source) + elif state in ("new", "upgrade"): + writes[rel] = (content, source) + elif force: + if hashlib.sha256((repo_root / CONFIG_DIR / rel).read_bytes()).hexdigest() == hashlib.sha256(content).hexdigest(): + unchanged[rel] = (content, source) + else: + writes[rel] = (content, source) + else: + refusals.append(state) + return writes, unchanged, refusals + + +def injected_skill_path(phase: str) -> str: + return f"${{repo_root}}/{CONFIG_DIR}/skills/{phase}/SKILL.md" + + +def point_config_at_copies(repo_root: Path, phases: list) -> bool: + """Repoint (or add) each injected phase in the config; True when the file changed.""" + path = config_path(repo_root) + before = path.read_text(encoding="utf-8") if path.exists() else None + doc = read_doc_for_write(repo_root) if path.exists() else builtin_doc() + builtin_types = {p["id"]: p["type"] for p in BUILTIN_PHASES} + for phase in phases: + entry = next((p for p in doc["phases"] if p.get("id") == phase), None) + if entry is not None: + entry["skill"] = injected_skill_path(phase) + elif phase in builtin_types: + ptype = builtin_types[phase] + doc["types"].setdefault(ptype, copy.deepcopy(BUILTIN_TYPES[ptype])) + doc["phases"].append({"id": phase, "type": ptype, "skill": injected_skill_path(phase)}) + write_doc(repo_root, doc) + return path.read_text(encoding="utf-8") != before + + +def write_injected(repo_root: Path, plugin: Path, planned: dict, unchanged: dict, force: bool) -> None: + manifest, _ = load_manifest(repo_root) + files = dict(manifest["files"]) if manifest is not None else {} + for rel, (content, source) in {**unchanged, **planned}.items(): + if rel in planned: + target = repo_root / CONFIG_DIR / rel + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(content) + shutil.copymode(source, target) + files[rel] = hashlib.sha256(content).hexdigest() + name, version = plugin_identity(plugin) + text = json.dumps({"plugin": name, "version": version, "files": files}, indent=2, sort_keys=True) + manifest_path(repo_root).parent.mkdir(parents=True, exist_ok=True) + manifest_path(repo_root).write_text(text + "\n", encoding="utf-8") + + +def cmd_inject(args) -> int: + plugin = Path(args.plugin_root).resolve() + available = sorted(p.name for p in (plugin / "phases").iterdir() if (p / "SKILL.md").is_file()) + phases = list(args.phases) or available + unknown = [p for p in phases if p not in available] + if unknown: + print(f"unknown phase(s) {unknown}; available: {available}", file=sys.stderr) + return BAD_CALL + try: + writes, unchanged, refusals = plan_inject(args.repo_root, plugin, phases, args.force) + if refusals: + for message in refusals: + print(f"refusing: {message}; pass --force to overwrite", file=sys.stderr) + return 1 + before_manifest = manifest_path(args.repo_root).read_text() if manifest_path(args.repo_root).exists() else None + config_changed = point_config_at_copies(args.repo_root, phases) + except ConfigError as exc: + print(f"malformed document: {exc}", file=sys.stderr) + return BAD_CALL + if not writes and not config_changed and before_manifest is not None: + print("nothing to do: every copy is current and the config already names them") + return 0 + write_injected(args.repo_root, plugin, writes, unchanged, args.force) + print(f"injected {len(writes)} file(s), {len(unchanged)} already current; config names {phases}") + return 0 + + # --------------------------------------------------------------------------- # CLI # --------------------------------------------------------------------------- @@ -927,6 +1150,7 @@ def main(argv=None) -> int: "check": cmd_check, "set": cmd_set, "unset": cmd_unset, + "inject": cmd_inject, } handler = handlers.get(args.command) if handler is None: diff --git a/plugins/factory/references/pipeline-config.md b/plugins/factory/references/pipeline-config.md index b6809e9..eadef79 100644 --- a/plugins/factory/references/pipeline-config.md +++ b/plugins/factory/references/pipeline-config.md @@ -86,3 +86,9 @@ A list-valued key takes repeated `--item` flags in place of a positional value. ## The built-in default pipeline Equal to today's four phases: `scope` (interview, interactive), `scope-review` (review), `build` (implement), `ship` (ship), each declared with `${plugin_root}` so a plugin move only touches this one reference and `factory/scripts/factory-config.py`'s embedded copy. Today's criteria partition across the allowlist: `lint-spec.py ${plan_dir}/spec.md` (scope), `check-tests.py ${plan_dir}` (build) and `pr-evidence.py check ${plan_dir}/pr.md` (ship) become `checks` entries, each executable written as a `${plugin_root}/phases/{phase}/scripts/...` path; `gh pr view`, `git ls-remote` and the `Verdict: APPROVED` grep stay `requires` prose the orchestrator judges itself. `scope`'s `interview` type seals `${plan_dir}/spec.md`. Every built-in type declares `attempts: 3`; the default declares a ceiling of 12 attempts per run. + +## `inject` + +`factory-config.py inject [phase ...] [--force]` copies the named phase bodies (all four when none is named) and the transitive closure of what they reference into `.factory/`, keeping the plugin's relative depth so no link needs rewriting: `phases/{phase}/` lands at `.factory/skills/{phase}/`, `references/` at `.factory/references/` and `scripts/` at `.factory/scripts/`. The closure follows three forms in Markdown files: relative links, `{x-skill-root}/...` paths and backticked `templates/*.html` paths. The only rewrite is each `{x-skill-root}` placeholder, which becomes the copy's own `.factory/skills/{x}` path relative to the repository root. Each copied `SKILL.md` gains an `injected-from: {plugin}@{version}` frontmatter key; no other file carries a header, and `.factory/.inject.json` is the provenance for the rest. + +A target that exists is refused, naming the file, unless `--force`, when its sha256 differs from its manifest entry, or when the manifest is missing, unparseable or has no entry for it. An unedited copy takes an upgrade without `--force`. `inject` also writes or updates `.factory/config.yaml` so each injected phase entry points at `${repo_root}/.factory/skills/{phase}/SKILL.md`, creating the config from the built-in default when there is none, and never sets `unattended_safe`, since a copy matching its manifest is injected rather than foreign. diff --git a/plugins/factory/scripts/factory-config.py b/plugins/factory/scripts/factory-config.py index cf7f568..49832cc 100755 --- a/plugins/factory/scripts/factory-config.py +++ b/plugins/factory/scripts/factory-config.py @@ -7,6 +7,7 @@ factory-config.py check factory-config.py set [--item ...] factory-config.py unset + factory-config.py inject [phase ...] [--force] Exit codes: init 0 written; 1 the file already exists @@ -16,6 +17,9 @@ 3 a malformed path or an unsupported document construct unset 0 recorded (a no-op if the key was already absent); 3 a malformed path + inject 0 copied, or nothing to do; 1 a target was edited since it was + injected or has no manifest entry (use --force); 3 an unknown phase + or an unusable config 3 always means the call itself could not be carried out - a malformed key path, an unparseable document, or a usage error - never a finding about the @@ -27,8 +31,11 @@ from __future__ import annotations import argparse +import copy import hashlib import json +import re +import shutil import sys from pathlib import Path from typing import NoReturn @@ -280,7 +287,7 @@ def _needs_quotes(text: str) -> bool: return True if text != text.strip(): return True - if ": " in text or text.endswith(":"): + if ": " in text or text.endswith(":") or " #" in text: return True if text[0] in "\"'[]{}&*|>#-?:@`": return True @@ -750,6 +757,222 @@ def render_json(resolved: dict) -> str: return json.dumps({"phases": phases}, indent=2) + "\n" +# --------------------------------------------------------------------------- +# inject: copy phase bodies and what they reference into .factory/ +# --------------------------------------------------------------------------- + +# A plugin directory and where its copy lands under .factory/. The plugin's +# relative depth is preserved, so no link in a copied file needs rewriting. +INJECT_MAP = {"phases": "skills", "references": "references", "scripts": "scripts"} +LINK_RE = re.compile(r"\]\(([^)#\s]+)(?:#[^)]*)?\)") +SKILL_ROOT_RE = re.compile(r"\{([a-z][a-z-]*)-skill-root\}((?:/[\w.-]+)*)") +TEMPLATE_RE = re.compile(r"`(templates/[\w./-]+\.html)`") +PROVENANCE_KEY = "injected-from" + + +def _owning_phase(plugin: Path, source: Path): + try: + parts = source.relative_to(plugin / "phases").parts + except ValueError: + return None + return parts[0] if len(parts) > 1 else None + + +def _referenced_paths(plugin: Path, source: Path) -> list: + """Files a Markdown file names by a relative link, a skill-root path or a template.""" + text = source.read_text(encoding="utf-8") + found = [] + for target in LINK_RE.findall(text): + if not target.startswith(("http://", "https://", "mailto:", "/")): + found.append(source.parent / target) + for phase, suffix in SKILL_ROOT_RE.findall(text): + found.append(plugin / "phases" / phase / suffix.rstrip(".,").lstrip("/")) + owner = _owning_phase(plugin, source) + if owner is not None: + found.extend(plugin / "phases" / owner / tpl for tpl in TEMPLATE_RE.findall(text)) + return found + + +def _injectable(plugin: Path, path: Path) -> bool: + try: + top = path.relative_to(plugin).parts[0] + except (ValueError, IndexError): + return False + return top in INJECT_MAP and path.is_file() + + +def inject_closure(plugin: Path, phases: list) -> list: + """Every file the named phase bodies reach, transitively, in a stable order.""" + seen: dict = {} + queue = [(plugin / "phases" / phase / "SKILL.md").resolve() for phase in phases] + while queue: + source = queue.pop(0) + if source in seen or not _injectable(plugin, source): + continue + seen[source] = None + if source.suffix == ".md": + queue.extend(p.resolve() for p in _referenced_paths(plugin, source)) + return sorted(seen) + + +def inject_destination(plugin: Path, source: Path) -> str: + """The path under .factory/ a plugin file lands at.""" + top, *rest = source.relative_to(plugin).parts + return "/".join([INJECT_MAP[top], *rest]) + + +def plugin_identity(plugin: Path) -> tuple: + try: + meta = json.loads((plugin / ".claude-plugin" / "plugin.json").read_text(encoding="utf-8")) + return str(meta.get("name", plugin.name)), str(meta.get("version", "unknown")) + except (OSError, json.JSONDecodeError): + return plugin.name, "unknown" + + +def add_provenance(text: str, origin: str) -> str: + """Add the provenance key to a SKILL.md's frontmatter, before its closing rule.""" + lines = text.split("\n") + if not lines or lines[0].strip() != "---": + return text + for index in range(1, len(lines)): + if lines[index].strip() == "---": + lines.insert(index, f"{PROVENANCE_KEY}: {origin}") + break + return "\n".join(lines) + + +def inject_content(plugin: Path, source: Path, phases: set, origin: str) -> bytes: + """A file's bytes as injected: skill-root placeholders resolved, provenance added.""" + if source.suffix != ".md": + return source.read_bytes() + text = source.read_text(encoding="utf-8") + text = SKILL_ROOT_RE.sub( + lambda m: f".factory/skills/{m.group(1)}{m.group(2)}" if m.group(1) in phases else m.group(0), + text, + ) + if source.name == "SKILL.md" and _owning_phase(plugin, source) is not None: + text = add_provenance(text, origin) + return text.encode("utf-8") + + +def load_manifest(repo_root: Path): + """Return (manifest dict, reason) - the dict is None with a reason it is unusable.""" + path = manifest_path(repo_root) + if not path.exists(): + return None, "is missing" + try: + manifest = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None, "is unparseable" + if not isinstance(manifest, dict) or not isinstance(manifest.get("files"), dict): + return None, "is unparseable" + return manifest, "" + + +def classify_target(target: Path, rel: str, content: bytes, manifest, reason: str): + """One of 'new', 'unchanged', 'upgrade' or a refusal message for a file to inject.""" + if not target.exists(): + return "new" + current = hashlib.sha256(target.read_bytes()).hexdigest() + entry = manifest["files"].get(rel) if manifest is not None else None + if manifest is None: + return f"{rel} exists and .factory/.inject.json {reason}" + if entry is None: + return f"{rel} exists and has no entry in .factory/.inject.json" + if current != entry: + return f"{rel} was edited since it was injected" + return "unchanged" if current == hashlib.sha256(content).hexdigest() else "upgrade" + + +def plan_inject(repo_root: Path, plugin: Path, phases: list, force: bool): + """Return (writes, unchanged, refusals): file plans keyed by destination path.""" + manifest, reason = load_manifest(repo_root) + origin = "%s@%s" % plugin_identity(plugin) + phase_names = {p.name for p in (plugin / "phases").iterdir() if p.is_dir()} + writes, unchanged, refusals = {}, {}, [] + for source in inject_closure(plugin, phases): + rel = inject_destination(plugin, source) + content = inject_content(plugin, source, phase_names, origin) + state = classify_target(repo_root / CONFIG_DIR / rel, rel, content, manifest, reason) + if state == "unchanged": + unchanged[rel] = (content, source) + elif state in ("new", "upgrade"): + writes[rel] = (content, source) + elif force: + if hashlib.sha256((repo_root / CONFIG_DIR / rel).read_bytes()).hexdigest() == hashlib.sha256(content).hexdigest(): + unchanged[rel] = (content, source) + else: + writes[rel] = (content, source) + else: + refusals.append(state) + return writes, unchanged, refusals + + +def injected_skill_path(phase: str) -> str: + return f"${{repo_root}}/{CONFIG_DIR}/skills/{phase}/SKILL.md" + + +def point_config_at_copies(repo_root: Path, phases: list) -> bool: + """Repoint (or add) each injected phase in the config; True when the file changed.""" + path = config_path(repo_root) + before = path.read_text(encoding="utf-8") if path.exists() else None + doc = read_doc_for_write(repo_root) if path.exists() else builtin_doc() + builtin_types = {p["id"]: p["type"] for p in BUILTIN_PHASES} + for phase in phases: + entry = next((p for p in doc["phases"] if p.get("id") == phase), None) + if entry is not None: + entry["skill"] = injected_skill_path(phase) + elif phase in builtin_types: + ptype = builtin_types[phase] + doc["types"].setdefault(ptype, copy.deepcopy(BUILTIN_TYPES[ptype])) + doc["phases"].append({"id": phase, "type": ptype, "skill": injected_skill_path(phase)}) + write_doc(repo_root, doc) + return path.read_text(encoding="utf-8") != before + + +def write_injected(repo_root: Path, plugin: Path, planned: dict, unchanged: dict, force: bool) -> None: + manifest, _ = load_manifest(repo_root) + files = dict(manifest["files"]) if manifest is not None else {} + for rel, (content, source) in {**unchanged, **planned}.items(): + if rel in planned: + target = repo_root / CONFIG_DIR / rel + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(content) + shutil.copymode(source, target) + files[rel] = hashlib.sha256(content).hexdigest() + name, version = plugin_identity(plugin) + text = json.dumps({"plugin": name, "version": version, "files": files}, indent=2, sort_keys=True) + manifest_path(repo_root).parent.mkdir(parents=True, exist_ok=True) + manifest_path(repo_root).write_text(text + "\n", encoding="utf-8") + + +def cmd_inject(args) -> int: + plugin = Path(args.plugin_root).resolve() + available = sorted(p.name for p in (plugin / "phases").iterdir() if (p / "SKILL.md").is_file()) + phases = list(args.phases) or available + unknown = [p for p in phases if p not in available] + if unknown: + print(f"unknown phase(s) {unknown}; available: {available}", file=sys.stderr) + return BAD_CALL + try: + writes, unchanged, refusals = plan_inject(args.repo_root, plugin, phases, args.force) + if refusals: + for message in refusals: + print(f"refusing: {message}; pass --force to overwrite", file=sys.stderr) + return 1 + before_manifest = manifest_path(args.repo_root).read_text() if manifest_path(args.repo_root).exists() else None + config_changed = point_config_at_copies(args.repo_root, phases) + except ConfigError as exc: + print(f"malformed document: {exc}", file=sys.stderr) + return BAD_CALL + if not writes and not config_changed and before_manifest is not None: + print("nothing to do: every copy is current and the config already names them") + return 0 + write_injected(args.repo_root, plugin, writes, unchanged, args.force) + print(f"injected {len(writes)} file(s), {len(unchanged)} already current; config names {phases}") + return 0 + + # --------------------------------------------------------------------------- # CLI # --------------------------------------------------------------------------- @@ -927,6 +1150,7 @@ def main(argv=None) -> int: "check": cmd_check, "set": cmd_set, "unset": cmd_unset, + "inject": cmd_inject, } handler = handlers.get(args.command) if handler is None: From 92723cd08ad7ffcc661659e5aaabe05b9a4bfb14 Mon Sep 17 00:00:00 2001 From: tobrun Date: Mon, 21 Sep 2026 09:31:35 +0200 Subject: [PATCH 24/30] docs(factory): land the config-driven contracts and the per-host variants Change set 7 of config-driven-factory/spec.md: docs/contracts.md carries the three new factory contracts with their provenance resolved, and the evals README gains the custom-pipeline and foreign-phase manual variants. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G9Voi18VRysPejexYsV7g4 --- docs/contracts.md | 27 ++------- factory/evals/README.md | 12 +++- factory/evals/tests/test_contracts_ledger.py | 62 ++++++++++++++++++++ 3 files changed, 79 insertions(+), 22 deletions(-) create mode 100644 factory/evals/tests/test_contracts_ledger.py diff --git a/docs/contracts.md b/docs/contracts.md index 2dce7f4..94d7284 100644 --- a/docs/contracts.md +++ b/docs/contracts.md @@ -1,31 +1,16 @@ # Contracts -## Factory plugin - -C-factory-result: every factory phase skill writes the result path the launch prompt names (.dev/{plan}/results/{phase}-{attempt}.json) as its last action, with schema factory.result/1, and never launches the next phase - guaranteed by: the Factory context section of each phase SKILL.md, checked by validate.sh check_factory_protocol (scripts/validate.sh, factory-plugin/spec.md change set 5) - relied on by: the run skill's judgment loop, run-state.py check-result - (2026-09-20, D-result-envelope) -C-factory-unattended: after the go, no factory phase skill routes a decision to a person - scope is the one interactive phase and is outside the check's scan root, and run/SKILL.md's person-routing lines live in blocks - guaranteed by: the validate.sh check over factory/skills/{scope-review,build,ship,run} and factory/references/factory-run.md, whose regex self-tests against four sample phrases and fails validation on a failed self-test - relied on by: the orchestrator's judgment loop, which never waits on a person; a run out of options ends with a written report instead of a question - (2026-09-20, D-unattended-wording-check, D-human-touchpoints) -C-factory-plan-files: plan files under .dev/ are never committed by a factory run; the run branch carries code, tests, and docs/ only - guaranteed by: the run skill's preflight, which runs git check-ignore -q .dev/ in the consuming repository and appends the entry to .gitignore as a recorded repair when it is missing, so the rule has a carrier in any consuming repository and not only in the fixture ? verify: no free scenario reaches the missing-entry branch, because the fixture's setup.sh writes the entry itself (D-verification names this one of the six knowingly unproven behaviors) - relied on by: any consuming repository that has not ignored .dev/ itself; git log --name-only on the run branch is the proof the closing report cites - (2026-09-20, D-plan-files-ignored) - ## Config-driven factory C-factory-phase-contract: the launch prompt requires every phase, ours or foreign, to write .dev/{plan}/results/{phase}-{attempt}.json with schema factory.result/1 as its last action, and never to launch another phase. This imposes both, it does not guarantee them; a phase that writes nothing is read as a failed attempt - guaranteed by: the wrapper in run/references/launch.md, which validate.sh F02 checks for the results path literal, the factory.result/1 literal and the exact sentence "Never name or launch the next phase."; F02 over every built-in phase body for factory-run.json, the results path literal and that same sentence; and run-state.py check-result treating a missing or unparseable file as a failed attempt ? verify: not yet built; resolve when config-driven-factory/spec.md change set 7 lands + guaranteed by: the wrapper in run/references/launch.md, which validate.sh F02 checks for the results path literal, the factory.result/1 literal and the exact sentence "Never name or launch the next phase."; F02 over every built-in phase body for factory-run.json, the results path literal and that same sentence; and run-state.py check-result treating a missing or unparseable file as a failed attempt relied on by: the judgment loop, run-state.py check-result, and the one-invoker rule (D-invoker-invariant) - (2026-09-20, config-driven-factory/spec.md; supersedes C-factory-result once change set 7 lands) + (2026-09-20, config-driven-factory/spec.md) C-factory-unattended: after the go, no FACTORY-OWNED phase body routes a decision to a person. A phase body the factory did not write is covered by no check: its author declares it unattended_safe in the config, and a phase that blocks on a person is neither prevented nor detected - guaranteed by: the validate.sh scan over factory/phases and factory/skills/run, for factory-owned bodies only ? verify: not yet built; resolve when config-driven-factory/spec.md change set 7 lands + guaranteed by: the validate.sh scan over factory/phases and factory/skills/run, for factory-owned bodies only relied on by: the judgment loop, which never waits on a person - (2026-09-20, config-driven-factory/spec.md; supersedes the entry above once change set 7 lands) + (2026-09-20, config-driven-factory/spec.md) C-factory-plan-files: plan files under .dev/ are never committed by a factory run; the run branch carries code, tests, docs/ and, after an explicit inject, the .factory/ tree - guaranteed by: the run skill's preflight, which runs git check-ignore -q .dev/ in the consuming repository and appends the entry to .gitignore as a recorded repair when it is missing ? verify: not yet built; resolve when config-driven-factory/spec.md change set 7 lands + guaranteed by: the run skill's preflight, which runs git check-ignore -q .dev/ in the consuming repository and appends the entry to .gitignore as a recorded repair when it is missing relied on by: any consuming repository that has not ignored .dev/ itself; git log --name-only on the run branch is the proof the closing report cites, and ship reviews .factory/ files in that log as part of the change - (2026-09-20, config-driven-factory/spec.md; supersedes the entry above once change set 7 lands) + (2026-09-20, config-driven-factory/spec.md) diff --git a/factory/evals/README.md b/factory/evals/README.md index 8fe9b7e..f62bd24 100644 --- a/factory/evals/README.md +++ b/factory/evals/README.md @@ -3,7 +3,7 @@ Three levels of verification (D-verification): 1. **Structural, free**: `scripts/validate.sh`'s `check_factory_unattended`, `check_factory_protocol`, and `check_factory_script` - run on every `scripts/validate.sh` invocation. -2. **Unit tests, free**: `python3 -m unittest discover -s factory/evals/tests -t .` - `test_run_state.py`, `test_factory_checks.py`, `test_fixture_setup.py`. +2. **Unit tests, free**: `python3 -m unittest discover -s factory/evals/tests -t .` - `test_run_state.py`, `test_factory_config.py`, `test_stub_pipeline.py`, `test_factory_checks.py`, `test_fixture_setup.py`. The stub pipeline in `factory/evals/stubs/` proves the bookkeeping (attempts, timestamps, terminality, budgets, drift) and nothing about the orchestrator's prose loop. 3. **One real end-to-end run per host, paid and manual**: this file's procedure, recorded in `results.md`. ## The end-to-end procedure @@ -28,6 +28,16 @@ Per host (Claude Code, Codex), because these are paid host sessions build's mock Ship reaching `gh pr create` and failing there is part of the pass, not an exception to it: with `gh` installed and authenticated the command still refuses with "none of the git remotes configured for this repository point to a known GitHub host" before any network call, the orchestrator judges that under the third `gh` fault category in [`judgment.md`](../skills/run/references/judgment.md), and the run ends with a report naming the origin and the pull request as the one unfinished step. There is no run in which all four phases are `done` on this fixture; that ending leaves the pull-request step and the Codex GitHub prerequisites unproven. +### Custom-pipeline variant + +The stub fixture cannot prove the prose loop in `run/SKILL.md`, so a real run on a pipeline that is not ours is its own paid, manual check, one per host. In the fixture's destination, write a two-phase `.factory/config.yaml` with `factory-config.py init` and `set`: a first `interview` phase and a second unattended phase, both pointing at skills that are not ours (a minimal skill that writes one file is enough), then run `python3 factory/scripts/factory-config.py check` and expect exit 0. Invoke `/factory:run "a request"`. + +The variant passes when the run ends with `finished: yes` in `run-state.py show`, the closing report names each phase's type, skill path and timings from the attempt records, the interactive phase ran inline and the go followed it, and no phase was launched before the go. Record any of these that did not hold in `results.md`. + +### Foreign-phase variant + +One per host, on the same custom pipeline: make the unattended phase's skill deliberately fail its first attempt (write no result file, or a result whose `status` is `failed`). The variant passes when the attempt is closed failed with reason "no result file" or the phase's own reason, the orchestrator relaunches it or ends the run and never hand-repairs it (`show --resolved` classes it foreign), and the phase carried `unattended_safe: true`. Also run the pipeline once with that declaration removed and expect `check` to exit 1 naming the phase before any interview. + ### Planted-secret variant One per host, before build's first commit: write into `webhook/__init__.py` a `GITHUB_TOKEN = "ghp_{36 random alphanumerics}"`, generated at setup time as one contiguous literal: diff --git a/factory/evals/tests/test_contracts_ledger.py b/factory/evals/tests/test_contracts_ledger.py new file mode 100644 index 0000000..4fc501f --- /dev/null +++ b/factory/evals/tests/test_contracts_ledger.py @@ -0,0 +1,62 @@ +"""Change set 7: the contracts and ledger entries the config-driven factory lands. + +Reads the real docs/ tree only. +""" + +from __future__ import annotations + +import re +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +CONTRACTS = (REPO_ROOT / "docs" / "contracts.md").read_text(encoding="utf-8") +DECISIONS = (REPO_ROOT / "docs" / "decisions.md").read_text(encoding="utf-8") + +NEW_CONTRACTS = ("C-factory-phase-contract", "C-factory-unattended", "C-factory-plan-files") +SUPERSEDED = ( + "D-phase-skills", + "D-human-touchpoints", + "D-handoff-seal", + "D-failure-ladder", + "D-unattended-wording-check", +) + + +def entry_names(text: str) -> list[str]: + return re.findall(r"^(C-[a-z-]+):", text, flags=re.MULTILINE) + + +class ContractsLedgerTest(unittest.TestCase): + def test_each_new_contract_appears_exactly_once(self) -> None: + names = entry_names(CONTRACTS) + for name in NEW_CONTRACTS: + with self.subTest(name=name): + self.assertEqual(names.count(name), 1) + + def test_the_old_factory_plugin_entries_are_gone(self) -> None: + self.assertNotIn("C-factory-result", CONTRACTS) + self.assertNotIn("## Factory plugin", CONTRACTS) + self.assertEqual(sorted(entry_names(CONTRACTS)), sorted(NEW_CONTRACTS)) + + def test_no_provenance_line_still_waits_on_change_set_7(self) -> None: + self.assertNotIn("once change set 7 lands", CONTRACTS) + self.assertNotIn("? verify", CONTRACTS) + + def test_the_superseded_decisions_carry_their_marks(self) -> None: + for slug in SUPERSEDED: + with self.subTest(slug=slug): + line = next(l for l in DECISIONS.splitlines() if l.startswith(f"{slug}:")) + self.assertIn("superseded by D-", line) + + def test_the_guarantees_the_contracts_name_exist(self) -> None: + for path in ("scripts/validate.sh", "factory/skills/run/references/launch.md", + "factory/skills/run/scripts/run-state.py"): + self.assertTrue((REPO_ROOT / path).is_file(), path) + launch = (REPO_ROOT / "factory/skills/run/references/launch.md").read_text(encoding="utf-8") + for literal in ("factory.result/1", "Never name or launch the next phase."): + self.assertIn(literal, launch) + + +if __name__ == "__main__": + unittest.main() From 6f57000798fa8f211ff7cfe8e08694d5226bbac2 Mon Sep 17 00:00:00 2001 From: tobrun Date: Tue, 22 Sep 2026 15:49:39 +0200 Subject: [PATCH 25/30] fix(factory): stop the orchestrator from pausing between phases What: run/SKILL.md now says explicitly that once past the go, every remaining phase runs to completion in the same turn, one launch after another, with no interim progress report or check-in; ending the turn while run-state.py show still prints finished: no is a defect unless a real host constraint forces it, in which case say so and print the resume command. Step 7's advance also says to go straight back to step 1 for the next phase instead of stopping. Why: this design has no background runner anymore - run is one skill looping through phases inline in a single orchestrator turn - and nothing forbade ending that turn between phases. A run had stopped after scope-review instead of continuing into build, with nothing in the instructions to prevent the model from treating a routine phase completion as a natural stopping point. --- factory/skills/run/SKILL.md | 3 ++- plugins/factory/skills/run/SKILL.md | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index bf61910..4bb3e22 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -7,6 +7,7 @@ disable-model-invocation: true # Run You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase launch another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. +Once past the go, run every remaining phase to completion in this same turn, one launch after another, with no pause to report progress or ask whether to continue - the closing report in section 6 is the only status update, and it is written once, when the run is actually finished or ended. Ending your turn while `run-state.py show {plan}` still prints `finished: no`, without a genuine `end` decision or a host limit forcing it, is a defect: if a real constraint (context, length, a host timeout) leaves you unable to keep going, say so plainly and print the exact resume command (`/factory:run` with no request); never trail off after an attempt's routine status line as if that were a stopping point. The pipeline is data: an ordered list of typed phases in `.factory/config.yaml`, or the built-in default (`scope`, `scope-review`, `build`, `ship`) when that file is absent. You never infer what a phase means - everything you do with one comes from its type's declared axes, read via `factory-config.py show --resolved` ([references/pipeline-config.md](../../references/pipeline-config.md)). `${plugin_root}` is `{run-skill-root}/../..`, and `factory-config.py` resolves it from its own location, the same directory. @@ -59,7 +60,7 @@ For each phase after the prefix, in declared order: 4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. 5. Judge the phase per [references/judgment.md](references/judgment.md): run its `checks`, then weigh its `requires`. 6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. -7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run. Follow the phase's budget and the run's ceiling as printed by `run-state.py show {plan}`; `attempt` never refuses one, so honouring it is yours. +7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run. Follow the phase's budget and the run's ceiling as printed by `run-state.py show {plan}`; `attempt` never refuses one, so honouring it is yours. On `advance`, when this was not the last declared phase, go straight back to step 1 for the next phase in this same turn - advancing is not a place to stop and check in. 8. Before advancing, run `run-state.py diff-spec {plan}` and `run-state.py diff-config {plan} --pipeline .dev/{plan}/pipeline-now.json`; a dropped or reworded sealed scenario, `⊘` line or phase is evidence for the next judgment, not an automatic block. ## 6. Closing report diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index c752e20..93c92b2 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -6,6 +6,7 @@ description: Drive the pipeline declared in .factory/config.yaml, or the built-i # Run You are the orchestrator. You launch phase subagents, judge their results against evidence, repair or relaunch on failure, and end in a pull request or a report. You never implement a phase yourself and never let a phase launch another. `.dev/{plan}/factory-run.json` is yours alone to write; every other file is a phase's. +Once past the go, run every remaining phase to completion in this same turn, one launch after another, with no pause to report progress or ask whether to continue - the closing report in section 6 is the only status update, and it is written once, when the run is actually finished or ended. Ending your turn while `run-state.py show {plan}` still prints `finished: no`, without a genuine `end` decision or a host limit forcing it, is a defect: if a real constraint (context, length, a host timeout) leaves you unable to keep going, say so plainly and print the exact resume command (`/factory:run` with no request); never trail off after an attempt's routine status line as if that were a stopping point. The pipeline is data: an ordered list of typed phases in `.factory/config.yaml`, or the built-in default (`scope`, `scope-review`, `build`, `ship`) when that file is absent. You never infer what a phase means - everything you do with one comes from its type's declared axes, read via `factory-config.py show --resolved` ([references/pipeline-config.md](../../references/pipeline-config.md)). `${plugin_root}` is `{run-skill-root}/../..`, and `factory-config.py` resolves it from its own location, the same directory. @@ -58,7 +59,7 @@ For each phase after the prefix, in declared order: 4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. 5. Judge the phase per [references/judgment.md](references/judgment.md): run its `checks`, then weigh its `requires`. 6. A `stopped` result with kind `secret.found` or `action.destructive` ends the run immediately; no decision overrides it. The same two rules bind your own repairs: never rewrite history, force-push, or delete outside the run branch. -7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run. Follow the phase's budget and the run's ceiling as printed by `run-state.py show {plan}`; `attempt` never refuses one, so honouring it is yours. +7. Record the decision with `run-state.py record {plan}` before acting on it, then act per the failure ladder in [references/judgment.md](references/judgment.md): advance, repair it yourself and re-check, relaunch with guidance, or end the run. Follow the phase's budget and the run's ceiling as printed by `run-state.py show {plan}`; `attempt` never refuses one, so honouring it is yours. On `advance`, when this was not the last declared phase, go straight back to step 1 for the next phase in this same turn - advancing is not a place to stop and check in. 8. Before advancing, run `run-state.py diff-spec {plan}` and `run-state.py diff-config {plan} --pipeline .dev/{plan}/pipeline-now.json`; a dropped or reworded sealed scenario, `⊘` line or phase is evidence for the next judgment, not an automatic block. ## 6. Closing report From ef549f18cc4dafd921a6e35e0b93007e6985184f Mon Sep 17 00:00:00 2001 From: tobrun Date: Tue, 22 Sep 2026 15:58:29 +0200 Subject: [PATCH 26/30] feat(factory): add an optional per-type model axis, reopening D-model-selection What: a type (or a phase, overriding its type) may declare a `model` string, resolved by factory-config.py and printed by show --resolved. The orchestrator passes it to the host's subagent tool (launch.md's transport ladder) when that tool takes a model parameter, and it is silently unused when the host has none. check refuses model on an interactive type, since that phase runs inline and never reaches a subagent call to carry it to. The built-in pipeline declares no model for any type: the valid values differ by host (Claude Code's Agent tool vs. Codex's spawn_agent vs. opencode's task), so no single default survives being run from a different host than the one it was written for. Pin one per repository with factory-config.py set. Why: a real run inherited Codex/GPT-5 for every phase, and the user expected the removed runner's own per-phase defaults. D-model-selection in docs/decisions.md had deferred this with an explicit reopen condition - "a real run shows one phase needs a different model than the session" - which this is; superseded by D-model-per-phase, which also records why host-per-phase (mixing Claude Code and Codex in one run) cannot come back under the current one-session orchestrator. --- docs/decisions.md | 6 +- factory/README.md | 2 +- factory/evals/tests/test_factory_config.py | 74 +++++++++++++++++++ factory/references/pipeline-config.md | 11 ++- factory/scripts/factory-config.py | 13 +++- factory/skills/run/SKILL.md | 4 +- factory/skills/run/references/launch.md | 8 +- plugins/factory/references/pipeline-config.md | 11 ++- plugins/factory/scripts/factory-config.py | 13 +++- plugins/factory/skills/run/SKILL.md | 4 +- .../factory/skills/run/references/launch.md | 8 +- 11 files changed, 138 insertions(+), 16 deletions(-) diff --git a/docs/decisions.md b/docs/decisions.md index ccdb6a2..44e56b3 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -107,9 +107,13 @@ D-scope-review-deferrals: What does the unattended scope-review do with escalati D-metrics: How is a run measured? (2026-09-20, factory-plugin/spec.md) ✓ each phase copy keeps its `skill-metrics.py start/end` calls - the orchestrator records wall-clock per attempt in the run state (auto-applied at Confidence: 75%) ⚠ `skill-metrics.py` reads Claude transcripts, so Codex rows are zeros and subagent runs may anchor on the wrong transcript; the run state timings are the portable numbers and metrics rows are never judgment evidence ✗ a new metrics script for the orchestrator - more code for numbers nobody has asked for yet -D-model-selection: Can the orchestrator pick a model per phase? (2026-09-20, factory-plugin/spec.md) +D-model-selection: Can the orchestrator pick a model per phase? (2026-09-20, factory-plugin/spec.md) superseded by D-model-per-phase (2026-09-22, user request), which reopens it on the condition this entry names ✗ a per-phase model table - Claude's Agent tool takes a model, Codex's spawn does not document one; a table would be host-specific ⊘ not doing - phases inherit the host session's model; reopen when a real run shows one phase needs a different model than the session +D-model-per-phase: Reopening D-model-selection - a real run inherited Codex/GPT-5 for every phase, and the user expected the removed runner's own defaults (Claude/Fable for scope, Codex models for scope-review/build/ship) (2026-09-22, user request) + ✓ an optional `model` string per type (a phase's overrides its type's), passed to the host's subagent tool when that tool takes one - Claude Code's Agent tool does today; Codex's `spawn_agent` and opencode's `task` are asked for one and, absent support, the value is just unused and the subagent inherits the session's model ? verify: spawn_agent's and task's own model parameter, unchecked since 2026-09-20 - a single string, never a table, answers D-model-selection's stated objection directly; `factory-config.py check` refuses it on an `interactive` type, since that phase never reaches a subagent call to carry it to + ✗ restore the removed runner's per-phase host and model table verbatim (Claude for scope, Codex for the rest) - impossible under D-orchestrator-form/D-phase-transport: one session now drives the whole pipeline, so no phase can run on a different host than the one `run` was invoked in + ✗ bake host-specific model ids into the built-in pipeline's types - a value valid on the host it was written for (Claude's `fable`, say) is meaningless or an error on another (Codex, opencode); the built-in pipeline declares no `model`, and a repository pins one with `factory-config.py set` once it knows which host it runs in D-parallel-phases: Can phases overlap? (2026-09-20, factory-plugin/spec.md) ✗ start build while scope-review's panel runs - `spec.md` has one writer at a time (plan-layout.md) ⊘ not doing - phases run strictly in order; reopen if a run's wall clock is dominated by a phase that could safely overlap diff --git a/factory/README.md b/factory/README.md index 74aecaa..aae6d5d 100644 --- a/factory/README.md +++ b/factory/README.md @@ -26,7 +26,7 @@ The run lives as long as the session; `.dev/{plan}/factory-run.json` is what sur ## The pipeline config -`python3 factory/scripts/factory-config.py init` writes a starter `.factory/config.yaml`; `show --resolved`, `check`, `set` and `unset` read and edit it, and you normally never edit the file by hand. Each phase names a `type` declared under `types`, and a type declares exactly five things: `interactive`, `requires` (the outcome, in prose), `checks` (argv commands that verify it), `seals` (artifacts watched for drift) and `attempts`. The full schema is in [references/pipeline-config.md](references/pipeline-config.md). +`python3 factory/scripts/factory-config.py init` writes a starter `.factory/config.yaml`; `show --resolved`, `check`, `set` and `unset` read and edit it, and you normally never edit the file by hand. Each phase names a `type` declared under `types`, and a type declares up to six things: `interactive`, `requires` (the outcome, in prose), `checks` (argv commands that verify it), `seals` (artifacts watched for drift), `attempts`, and an optional `model` passed to the host's subagent tool when it takes one - the built-in pipeline sets none, since the valid values differ by host. The full schema is in [references/pipeline-config.md](references/pipeline-config.md). A phase is one of three classes. A built-in phase runs from the installed plugin. An injected phase is a copy under `.factory/skills/` whose hash still matches `.factory/.inject.json`. Anything else is foreign, and a foreign phase must declare `unattended_safe: true`: nothing can detect a phase that blocks on a person, so that declaration is the word of whoever read the skill. diff --git a/factory/evals/tests/test_factory_config.py b/factory/evals/tests/test_factory_config.py index 8cd29ea..2cd6e51 100644 --- a/factory/evals/tests/test_factory_config.py +++ b/factory/evals/tests/test_factory_config.py @@ -382,6 +382,35 @@ def test_builtin_default_go_after_scope(self) -> None: result = run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory") self.assertIn("go: after scope", result.stdout) + def test_model_on_an_interactive_type_is_a_finding(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"inter": {"interactive": True, "model": "some-model"}}, + "phases": [{"id": "a", "type": "inter", "skill": "${repo_root}/x/SKILL.md", + "unattended_safe": True}], + }, + ) + result = run(self.repo, "check") + self.assertEqual(result.returncode, 1) + self.assertIn("'a'", result.stdout) + self.assertIn("model", result.stdout) + self.assertIn("interactive", result.stdout) + + def test_model_on_an_unattended_phase_is_not_a_finding(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"interactive": False, "model": "some-model"}}, + "phases": [{"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}], + }, + ) + self.assertEqual(run(self.repo, "check").returncode, 0) + class InitTest(TempRepoTestCase): def test_init_twice_second_fails(self) -> None: @@ -419,6 +448,21 @@ def test_set_defaults_attempts_reflected_in_show_resolved(self) -> None: result = run(self.repo, "show", "--resolved") self.assertIn("attempts: 5", result.stdout) + def test_set_and_unset_a_phase_model_roundtrips(self) -> None: + self._write_phase() + run(self.repo, "set", "build.model", "sonnet") + result = run(self.repo, "show", "--resolved") + self.assertIn("model: sonnet", result.stdout) + run(self.repo, "unset", "build.model") + result = run(self.repo, "show", "--resolved") + self.assertNotIn("model:", result.stdout) + + def test_set_a_type_model_applies_to_its_phase(self) -> None: + self._write_phase() + run(self.repo, "set", "types.t.model", "opus") + result = run(self.repo, "show", "--resolved") + self.assertIn("model: opus", result.stdout) + def test_set_types_checks_with_items(self) -> None: make_skill(self.repo, "x/SKILL.md") make_skill(self.repo, "lint-spec.py") @@ -501,6 +545,36 @@ def test_no_config_file_shows_four_default_phases(self) -> None: for pid in ("scope", "scope-review", "build", "ship"): self.assertIn(pid, result.stdout) + def test_the_default_pipeline_declares_no_model(self) -> None: + result = run(self.repo, "show", "--resolved", "--json", plugin_root=REPO_ROOT / "factory") + for phase in json.loads(result.stdout)["phases"]: + self.assertNotIn("model", phase, phase) + self.assertNotIn("model:", run(self.repo, "show", "--resolved", plugin_root=REPO_ROOT / "factory").stdout) + + def test_a_phase_model_overrides_its_type_model(self) -> None: + make_skill(self.repo, "x/SKILL.md") + write_json_doc( + self.repo, + { + "version": 1, + "types": {"t": {"model": "type-model"}}, + "phases": [ + {"id": "a", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True}, + {"id": "b", "type": "t", "skill": "${repo_root}/x/SKILL.md", "unattended_safe": True, + "model": "phase-model"}, + ], + }, + ) + result = run(self.repo, "show", "--resolved") + self.assertEqual(result.returncode, 0, result.stdout) + blocks = re.split(r"^- ", result.stdout, flags=re.M)[1:] + block_a = next(b for b in blocks if b.startswith("a (")) + block_b = next(b for b in blocks if b.startswith("b (")) + self.assertIn("model: type-model", block_a) + self.assertNotIn("model: phase-model", block_a) + self.assertIn("model: phase-model", block_b) + self.assertNotIn("model: type-model", block_b) + def test_config_with_two_phases_shows_exactly_those_in_order(self) -> None: make_skill(self.repo, "x/SKILL.md") make_skill(self.repo, "y/SKILL.md") diff --git a/factory/references/pipeline-config.md b/factory/references/pipeline-config.md index eadef79..ad26f63 100644 --- a/factory/references/pipeline-config.md +++ b/factory/references/pipeline-config.md @@ -32,16 +32,25 @@ phases: `version` is always `1`. `defaults`, `types` and `phases` are the schema's only other top-level keys; an unknown key is a `check` finding. -### Types: exactly five axes, no more +### Types: six axes, no more - `interactive` (bool) - runs inline in the orchestrator's own session rather than launched. - `requires` (prose) - the outcome contract the orchestrator judges when the type's `checks` don't cover it, the same way `judgment.md` reads a phase today. - `checks` (a list of argv lists) - commands that verify the outcome. Each entry is an argv list, never a shell string. - `seals` (a list of paths) - artifacts whose drift `run-state.py handoff`/`diff-spec` watches. May use `${plan_dir}`. - `attempts` (int) - the per-type default retry budget. +- `model` (string, optional) - passed to the host's subagent tool when it takes one; see "Model selection" below. Judges nothing and verifies nothing, unlike the other five. The type set is open: any name declared under `types` is usable by a phase. A type the orchestrator has never seen is judged by its declared axes alone. +### Model selection + +`model` is a single string, per type or per phase (a phase's `model` overrides its type's), meant as-is for whichever host subagent tool the run is using. There is no per-host table: D-model-selection in `docs/decisions.md` rejected that as host-specific config, and this stays true - the value is whatever your chosen host's model catalog expects (Claude Code's Agent tool: `sonnet`, `opus`, `haiku`, `fable`; Codex's `spawn_agent` and opencode's `task`: their own model ids, or nothing, if the tool has no model parameter at all). + +The orchestrator passes `model` to the launch call when the host tool it is using accepts one; when it does not, the value is silently unused and the subagent inherits the session's model, same as today with no `model` declared. `run/SKILL.md`'s `launch.md` names the exact rule per host. + +`model` on an `interactive` type is a `check` finding: an interactive phase runs inline in the orchestrator's own session and never reaches a subagent call, so it has no launch to carry a model to. `scope` in the built-in pipeline can therefore never take one. The built-in pipeline declares no `model` for any type - the choice of host (Claude Code vs. Codex vs. opencode) decides which model ids are even valid, so there is no single default that survives being run from a different host than the one it was written for. Pin one for your own repository with `factory-config.py set types..model ` or `set .model ` once you know which host you run `/factory:run` in. + ### Phases: an ordered list Each entry is `{id, type, skill, checks, unattended_safe}`. `id`, `type` and `skill` are required; `checks` and `unattended_safe` are optional per-phase overrides. diff --git a/factory/scripts/factory-config.py b/factory/scripts/factory-config.py index 49832cc..07c325b 100755 --- a/factory/scripts/factory-config.py +++ b/factory/scripts/factory-config.py @@ -646,6 +646,14 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): resolved_seals = list(seals_raw) attempts = phase.get("attempts", type_def.get("attempts", defaults.get("attempts", DEFAULT_ATTEMPTS))) + interactive = bool(type_def.get("interactive", False)) + model = phase.get("model", type_def.get("model")) + if model is not None and interactive: + findings.append( + f"phase {pid!r} declares model {model!r} on interactive type {ptype!r}: an interactive phase " + "runs inline in the orchestrator's own session and never reaches a subagent call, so a model " + "here has no effect - remove it or move it to an unattended phase" + ) phase_class, class_findings = classify_phase(phase, roots, repo_root) findings.extend(class_findings) @@ -658,12 +666,13 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): "id": pid, "type": ptype, "class": phase_class, - "interactive": bool(type_def.get("interactive", False)), + "interactive": interactive, "skill": skill_resolved, "checks": resolved_checks, "requires": type_def.get("requires", ""), "seals": resolved_seals, "attempts": attempts, + "model": model, "unattended_safe": unattended_safe, } ) @@ -738,6 +747,8 @@ def render_human(resolved: dict) -> str: lines.append(f" interactive: {phase['interactive']}") lines.append(f" skill: {phase['skill']}") lines.append(f" attempts: {phase['attempts']}") + if phase["model"] is not None: + lines.append(f" model: {phase['model']}") if phase["requires"]: lines.append(f" requires: {phase['requires']}") for entry in phase["checks"]: diff --git a/factory/skills/run/SKILL.md b/factory/skills/run/SKILL.md index 4bb3e22..1259cef 100644 --- a/factory/skills/run/SKILL.md +++ b/factory/skills/run/SKILL.md @@ -24,7 +24,7 @@ If two unfinished state files exist under `.dev/` with no branch match, or a rec ## 2. Init or resume -A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it and create `.dev/{plan}/` with `request.md` holding the request. Write `python3 {run-skill-root}/../../scripts/factory-config.py show --resolved --json` to `.dev/{plan}/pipeline.json`, read `show --resolved` for each phase's type, `attempts`, `checks`, `requires`, `seals`, class and skill path plus the `ceiling` and the `go`, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch} --phases {ids in order, comma-separated} --attempts {phase}={n} ... --ceiling {ceiling}` (D-plan-slug, D-request-input). +A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it and create `.dev/{plan}/` with `request.md` holding the request. Write `python3 {run-skill-root}/../../scripts/factory-config.py show --resolved --json` to `.dev/{plan}/pipeline.json`, read `show --resolved` for each phase's type, `attempts`, `checks`, `requires`, `seals`, `model`, class and skill path plus the `ceiling` and the `go`, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch} --phases {ids in order, comma-separated} --attempts {phase}={n} ... --ceiling {ceiling}` (D-plan-slug, D-request-input). No request and a run branch checked out: resume that plan. No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when `run-state.py show {plan}` prints `finished: yes` - its `decisions` list ends in an `end` action, or in an `advance` on the last declared phase; nothing else marks a run finished, so every other state file is unfinished. @@ -55,7 +55,7 @@ At the go - or before the first launch when there is no prefix - create branch ` For each phase after the prefix, in declared order: 1. Take the attempt number first: `run-state.py attempt {plan} {phase} --type {type} --skill {skill path}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. -2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. +2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance; pass the type's resolved `model`, when it has one, to the host's subagent tool per launch.md's transport ladder. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. 3. Wait for the host's completion signal - a batch in flight is not over. 4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. 5. Judge the phase per [references/judgment.md](references/judgment.md): run its `checks`, then weigh its `requires`. diff --git a/factory/skills/run/references/launch.md b/factory/skills/run/references/launch.md index a9c8137..a5d291d 100644 --- a/factory/skills/run/references/launch.md +++ b/factory/skills/run/references/launch.md @@ -4,11 +4,13 @@ The host transport ladder for one phase agent, and the exact prompt template. Mi ## Transport ladder -1. **Claude Code**: the Agent tool. -2. **Codex**: `spawn_agent`. -3. **opencode**: the `task` tool. +1. **Claude Code**: the Agent tool. Pass the phase's resolved `model` (`show --resolved`) as the tool's `model` parameter when the type declares one; omit the parameter when it does not, which inherits the session's model exactly as before this axis existed. +2. **Codex**: `spawn_agent`. Pass `model` the same way if the tool accepts one; if it does not, the value is simply unused and the subagent inherits the session's model - never work around this by any other means (a subprocess, a nested `codex exec`), since that is the per-host argv the runner was removed for. +3. **opencode**: the `task` tool. Same rule as Codex: pass `model` if the tool takes one, otherwise it is unused. 4. **Unavailable**: stop before the interview - nobody pays for a scope that cannot run. +A type's `model` is never valid on more than one host's catalog at once - `show --resolved` never validates it against the host you happen to be running on, and an unrecognized value is the launch call's own error, not a config finding. `scope`'s `interview` type - and any other `interactive` type - never reaches this ladder at all: it runs inline in the orchestrator's own session, so it has no launch call to carry a model to, and `factory-config.py check` refuses a `model` declared there. + ## The wrapper The template below wraps any phase skill, ours or a team's own, and imposes the result shape on it: the result file at `results/{phase}-{attempt}.json`, schema `factory.result/1`, written as the phase's last action. It imposes the shape, it does not guarantee it - a skill with strong opinions about its own output can defeat it, so a phase that writes nothing is read as a failed attempt, and the phase's declared `checks` verify the outcome independently of what the result says. diff --git a/plugins/factory/references/pipeline-config.md b/plugins/factory/references/pipeline-config.md index eadef79..ad26f63 100644 --- a/plugins/factory/references/pipeline-config.md +++ b/plugins/factory/references/pipeline-config.md @@ -32,16 +32,25 @@ phases: `version` is always `1`. `defaults`, `types` and `phases` are the schema's only other top-level keys; an unknown key is a `check` finding. -### Types: exactly five axes, no more +### Types: six axes, no more - `interactive` (bool) - runs inline in the orchestrator's own session rather than launched. - `requires` (prose) - the outcome contract the orchestrator judges when the type's `checks` don't cover it, the same way `judgment.md` reads a phase today. - `checks` (a list of argv lists) - commands that verify the outcome. Each entry is an argv list, never a shell string. - `seals` (a list of paths) - artifacts whose drift `run-state.py handoff`/`diff-spec` watches. May use `${plan_dir}`. - `attempts` (int) - the per-type default retry budget. +- `model` (string, optional) - passed to the host's subagent tool when it takes one; see "Model selection" below. Judges nothing and verifies nothing, unlike the other five. The type set is open: any name declared under `types` is usable by a phase. A type the orchestrator has never seen is judged by its declared axes alone. +### Model selection + +`model` is a single string, per type or per phase (a phase's `model` overrides its type's), meant as-is for whichever host subagent tool the run is using. There is no per-host table: D-model-selection in `docs/decisions.md` rejected that as host-specific config, and this stays true - the value is whatever your chosen host's model catalog expects (Claude Code's Agent tool: `sonnet`, `opus`, `haiku`, `fable`; Codex's `spawn_agent` and opencode's `task`: their own model ids, or nothing, if the tool has no model parameter at all). + +The orchestrator passes `model` to the launch call when the host tool it is using accepts one; when it does not, the value is silently unused and the subagent inherits the session's model, same as today with no `model` declared. `run/SKILL.md`'s `launch.md` names the exact rule per host. + +`model` on an `interactive` type is a `check` finding: an interactive phase runs inline in the orchestrator's own session and never reaches a subagent call, so it has no launch to carry a model to. `scope` in the built-in pipeline can therefore never take one. The built-in pipeline declares no `model` for any type - the choice of host (Claude Code vs. Codex vs. opencode) decides which model ids are even valid, so there is no single default that survives being run from a different host than the one it was written for. Pin one for your own repository with `factory-config.py set types..model ` or `set .model ` once you know which host you run `/factory:run` in. + ### Phases: an ordered list Each entry is `{id, type, skill, checks, unattended_safe}`. `id`, `type` and `skill` are required; `checks` and `unattended_safe` are optional per-phase overrides. diff --git a/plugins/factory/scripts/factory-config.py b/plugins/factory/scripts/factory-config.py index 49832cc..07c325b 100755 --- a/plugins/factory/scripts/factory-config.py +++ b/plugins/factory/scripts/factory-config.py @@ -646,6 +646,14 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): resolved_seals = list(seals_raw) attempts = phase.get("attempts", type_def.get("attempts", defaults.get("attempts", DEFAULT_ATTEMPTS))) + interactive = bool(type_def.get("interactive", False)) + model = phase.get("model", type_def.get("model")) + if model is not None and interactive: + findings.append( + f"phase {pid!r} declares model {model!r} on interactive type {ptype!r}: an interactive phase " + "runs inline in the orchestrator's own session and never reaches a subagent call, so a model " + "here has no effect - remove it or move it to an unattended phase" + ) phase_class, class_findings = classify_phase(phase, roots, repo_root) findings.extend(class_findings) @@ -658,12 +666,13 @@ def resolve(repo_root: Path, plugin_root_path: Path = None): "id": pid, "type": ptype, "class": phase_class, - "interactive": bool(type_def.get("interactive", False)), + "interactive": interactive, "skill": skill_resolved, "checks": resolved_checks, "requires": type_def.get("requires", ""), "seals": resolved_seals, "attempts": attempts, + "model": model, "unattended_safe": unattended_safe, } ) @@ -738,6 +747,8 @@ def render_human(resolved: dict) -> str: lines.append(f" interactive: {phase['interactive']}") lines.append(f" skill: {phase['skill']}") lines.append(f" attempts: {phase['attempts']}") + if phase["model"] is not None: + lines.append(f" model: {phase['model']}") if phase["requires"]: lines.append(f" requires: {phase['requires']}") for entry in phase["checks"]: diff --git a/plugins/factory/skills/run/SKILL.md b/plugins/factory/skills/run/SKILL.md index 93c92b2..3df9f99 100644 --- a/plugins/factory/skills/run/SKILL.md +++ b/plugins/factory/skills/run/SKILL.md @@ -23,7 +23,7 @@ If two unfinished state files exist under `.dev/` with no branch match, or a rec ## 2. Init or resume -A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it and create `.dev/{plan}/` with `request.md` holding the request. Write `python3 {run-skill-root}/../../scripts/factory-config.py show --resolved --json` to `.dev/{plan}/pipeline.json`, read `show --resolved` for each phase's type, `attempts`, `checks`, `requires`, `seals`, class and skill path plus the `ceiling` and the `go`, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch} --phases {ids in order, comma-separated} --attempts {phase}={n} ... --ceiling {ceiling}` (D-plan-slug, D-request-input). +A request given (quoted text or a Markdown file path): derive a kebab-case plan slug from it and create `.dev/{plan}/` with `request.md` holding the request. Write `python3 {run-skill-root}/../../scripts/factory-config.py show --resolved --json` to `.dev/{plan}/pipeline.json`, read `show --resolved` for each phase's type, `attempts`, `checks`, `requires`, `seals`, `model`, class and skill path plus the `ceiling` and the `go`, then run `python3 {run-skill-root}/scripts/run-state.py init {plan} --request {request} --base {default-branch} --phases {ids in order, comma-separated} --attempts {phase}={n} ... --ceiling {ceiling}` (D-plan-slug, D-request-input). No request and a run branch checked out: resume that plan. No request and no branch: resume the single unfinished state file under `.dev/`. A run is finished when `run-state.py show {plan}` prints `finished: yes` - its `decisions` list ends in an `end` action, or in an `advance` on the last declared phase; nothing else marks a run finished, so every other state file is unfinished. @@ -54,7 +54,7 @@ At the go - or before the first launch when there is no prefix - create branch ` For each phase after the prefix, in declared order: 1. Take the attempt number first: `run-state.py attempt {plan} {phase} --type {type} --skill {skill path}` appends a `launched` entry and prints `{phase} attempt {N}: launched`. That `N` is the attempt number for the launch prompt, the result path, and the close below. The state file owns it alone - never count attempts yourself and never invent one. -2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. +2. Launch one fresh-context subagent per [references/launch.md](references/launch.md)'s prompt template, naming the phase's skill path, the plan, attempt `N`, and (on a relaunch) the previous reason with explicit guidance; pass the type's resolved `model`, when it has one, to the host's subagent tool per launch.md's transport ladder. The record is written before the launch on purpose: the launch call does not return until the phase is over, so an entry written after it would never exist for the crash it is there to expose. 3. Wait for the host's completion signal - a batch in flight is not over. 4. Run `run-state.py check-result .dev/{plan}/results/{phase}-{attempt}.json` with `{attempt}` = `N`; a missing or unparseable file is a failed attempt with that reason (exit 1 like any other failure, never a crash; exit 3 is a bad call, not a phase outcome). Close the attempt with `run-state.py attempt {plan} {phase} --status {done|failed|stopped} --result .dev/{plan}/results/{phase}-{attempt}.json`, the same path. 5. Judge the phase per [references/judgment.md](references/judgment.md): run its `checks`, then weigh its `requires`. diff --git a/plugins/factory/skills/run/references/launch.md b/plugins/factory/skills/run/references/launch.md index a9c8137..a5d291d 100644 --- a/plugins/factory/skills/run/references/launch.md +++ b/plugins/factory/skills/run/references/launch.md @@ -4,11 +4,13 @@ The host transport ladder for one phase agent, and the exact prompt template. Mi ## Transport ladder -1. **Claude Code**: the Agent tool. -2. **Codex**: `spawn_agent`. -3. **opencode**: the `task` tool. +1. **Claude Code**: the Agent tool. Pass the phase's resolved `model` (`show --resolved`) as the tool's `model` parameter when the type declares one; omit the parameter when it does not, which inherits the session's model exactly as before this axis existed. +2. **Codex**: `spawn_agent`. Pass `model` the same way if the tool accepts one; if it does not, the value is simply unused and the subagent inherits the session's model - never work around this by any other means (a subprocess, a nested `codex exec`), since that is the per-host argv the runner was removed for. +3. **opencode**: the `task` tool. Same rule as Codex: pass `model` if the tool takes one, otherwise it is unused. 4. **Unavailable**: stop before the interview - nobody pays for a scope that cannot run. +A type's `model` is never valid on more than one host's catalog at once - `show --resolved` never validates it against the host you happen to be running on, and an unrecognized value is the launch call's own error, not a config finding. `scope`'s `interview` type - and any other `interactive` type - never reaches this ladder at all: it runs inline in the orchestrator's own session, so it has no launch call to carry a model to, and `factory-config.py check` refuses a `model` declared there. + ## The wrapper The template below wraps any phase skill, ours or a team's own, and imposes the result shape on it: the result file at `results/{phase}-{attempt}.json`, schema `factory.result/1`, written as the phase's last action. It imposes the shape, it does not guarantee it - a skill with strong opinions about its own output can defeat it, so a phase that writes nothing is read as a failed attempt, and the phase's declared `checks` verify the outcome independently of what the result says. From c250409d0e0405244952cb1cc8eb324268ee2f35 Mon Sep 17 00:00:00 2001 From: tobrun Date: Tue, 22 Sep 2026 18:10:37 +0200 Subject: [PATCH 27/30] fix(ship): mutation testing is reported, not a shippability gate What: dropped "zero surviving mutants in scope" from the gauntlet's threshold defaults in both dev/skills/ship and factory/phases/ship (and their generated Codex copies). Fix agents still chase survivors under the existing two-round loop and equivalent-mutant carve-out, and a stubborn survivor is still recorded with its reason - it just no longer fails the run on its own. The ship report's killed/survived/ uncovered counts are the evidence a person reads; there is no score below which shipping is blocked. Recorded as D-mutation-threshold in docs/decisions.md, since no prior decision had actually set this default - it was unexamined. Why: a real run scored 43% (430/1003) mutants killed on a diff-scoped file and was ruled unshippable purely on that inherited coverage debt. The user's own words: "we aim to do our best but zero is a bit too harsh and unrealistic." --- dev/skills/ship/SKILL.md | 4 ++-- dev/skills/ship/references/gauntlet.md | 7 ++++++- dev/skills/ship/references/tools.md | 4 +++- docs/decisions.md | 6 ++++++ factory/phases/ship/SKILL.md | 4 ++-- factory/phases/ship/references/gauntlet.md | 7 ++++++- factory/phases/ship/references/tools.md | 4 +++- plugins/dev/skills/ship/SKILL.md | 4 ++-- plugins/dev/skills/ship/references/gauntlet.md | 7 ++++++- plugins/dev/skills/ship/references/tools.md | 4 +++- plugins/factory/phases/ship/SKILL.md | 4 ++-- plugins/factory/phases/ship/references/gauntlet.md | 7 ++++++- plugins/factory/phases/ship/references/tools.md | 4 +++- 13 files changed, 50 insertions(+), 16 deletions(-) diff --git a/dev/skills/ship/SKILL.md b/dev/skills/ship/SKILL.md index e88c861..720518d 100644 --- a/dev/skills/ship/SKILL.md +++ b/dev/skills/ship/SKILL.md @@ -1,6 +1,6 @@ --- name: ship -description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every check passes, phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. +description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every gating check passes - mutation testing is scored and reported, never a shippability gate - phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. disable-model-invocation: true --- @@ -42,7 +42,7 @@ Never weaken or skip a check because acquiring its tool is work. 5. **Dependency rules** against `docs/dependencies.md` ([../../references/dependency-rules.md](../../references/dependency-rules.md) owns the checker semantics; absent file: skip with a clear note, never invent rules). 6. **Coverage-weighted complexity** per in-scope function. 7. **Flakiness** - the tests the diff added or touched, repeated and shuffled until trusted. -8. **Mutation testing** over the in-scope source. +8. **Mutation testing** over the in-scope source - scored and reported in the wrap-up, never a reason on its own to block shipping (see gauntlet.md's "Mutation testing is reported, not a threshold"). Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. diff --git a/dev/skills/ship/references/gauntlet.md b/dev/skills/ship/references/gauntlet.md index 27f13f4..7f79b6f 100644 --- a/dev/skills/ship/references/gauntlet.md +++ b/dev/skills/ship/references/gauntlet.md @@ -50,7 +50,12 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve ## Thresholds are decisions -Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. +Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. When the user accepts a different threshold, record it in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. Never adjust a threshold silently to make a run pass. + +## Mutation testing is reported, not a threshold + +The other seven checks above judge whether this run may ship; mutation testing does not, and it is the one check with no default threshold at all. A codebase's baseline kill rate depends on how much test coverage the diff's area already had going in, which this run did not choose - failing the run over that inherited debt would make ordinary changes to weakly-tested code unshippable regardless of what the diff itself does. +Fix agents still chase survivors under the loop above and the fix vocabulary's "kill a mutant" entry, and equivalent mutants still get marked and left, exactly as before. When a survivor resists two fix rounds (item 4 of the loop), present it and move on rather than treating it as a blocker. The ship report's counts - killed, survived, uncovered, with the reason recorded against each stubborn survivor - are the evidence a person reads; there is no score below which the run fails. diff --git a/dev/skills/ship/references/tools.md b/dev/skills/ship/references/tools.md index 64985db..3397f95 100644 --- a/dev/skills/ship/references/tools.md +++ b/dev/skills/ship/references/tools.md @@ -62,6 +62,8 @@ Keep it affordable: only diff-touched tests, reuse build caches between repeats, ## 8. Mutation testing +Scored and reported, never a shippability gate - see gauntlet.md's "Mutation testing is reported, not a threshold" for why and how a stubborn survivor is left in place rather than blocking the run. + Prefer the ecosystem's mutation framework - Stryker (JS/TS), mutmut or cosmic-ray (Python), PIT (JVM), cargo-mutants (Rust), go-mutesting (Go). These handle mutant generation, test selection, and reporting far better than a hand-rolled loop; write only the thin config that scopes them. Hand-roll only when the ecosystem has nothing: an agent-written script that applies one mutation at a time (flip `<` to `<=`, `==` to `!=`, `+` to `-`, negate conditions, drop return values), runs the narrowest relevant test command, and records survivors. @@ -70,7 +72,7 @@ Keeping it affordable: - **Scope to the diff**: mutate only in-scope files, run only the tests that cover them (most frameworks do incremental or per-file runs; use that). - Set a per-mutant test timeout so an infinite-loop mutant cannot hang the run. -- Equivalent mutants (mutations that provably cannot change behavior) are the one legitimate survivor category: mark them as such in the report with the reasoning, don't chase them forever - two fix rounds, then escalate per the loop rule. +- Equivalent mutants (mutations that provably cannot change behavior) are one legitimate reason to leave a survivor in place, marked as such with the reasoning; a survivor that resists two fix rounds for any other reason is left in place too, per the loop rule - killing every mutant is the goal, not the bar for shipping. ## Fix-agent prompts diff --git a/docs/decisions.md b/docs/decisions.md index 44e56b3..eb210ab 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -12,6 +12,12 @@ D-complexity-threshold: Where does the ship gauntlet's coverage-weighted complex ✗ 6 on new functions - splits nearly every new function into helpers that each earn nothing on their own, against keeping indirection earned ✗ 6 on every touched function - a refactor of modules outside the change +D-mutation-threshold: Does the gauntlet's mutation-testing check gate shipping, and at what score? (2026-09-22) + ✓ no threshold at all - it is scored and reported (killed/survived/uncovered, with the reason recorded against each stubborn survivor) but never a `failed` result on its own; fix agents still chase survivors under the existing two-round loop and equivalent-mutant carve-out (user, 2026-09-22): a real run scored 430/1003 (43%) on a diff-scoped file and the zero-survivor default made it unshippable purely on that inherited coverage debt, which the diff itself did not create ⚠ this was an unexamined default - no prior decision set "zero surviving mutants in scope" before this entry + ✗ zero surviving mutants in scope (the prior unexamined default) - unrealistic for a first mutation-testing pass over code whose existing tests were never written against mutants; blocks ordinary changes to weakly-tested areas regardless of what the diff does + ✗ a percentage floor (e.g. 70%) - still an absolute line a run can fail on for coverage debt the diff did not create, just a more forgiving one; rejected for the same reason as zero, at a different number + ✗ zero-tolerance narrowed to only the diff's added lines - still a hard gate, and a diff that changes existing (not just new) lines still inherits their prior coverage debt + ## Removed D-remove-factory: The factory plugin and its runner are removed from this repository (2026-09-20) diff --git a/factory/phases/ship/SKILL.md b/factory/phases/ship/SKILL.md index a6f212d..b0f8a41 100644 --- a/factory/phases/ship/SKILL.md +++ b/factory/phases/ship/SKILL.md @@ -1,6 +1,6 @@ --- name: ship -description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every check passes, phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. +description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every gating check passes - mutation testing is scored and reported, never a shippability gate - phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. disable-model-invocation: true --- @@ -42,7 +42,7 @@ Never weaken or skip a check because acquiring its tool is work. 5. **Dependency rules** against `docs/dependencies.md` ([../../references/dependency-rules.md](../../references/dependency-rules.md) owns the checker semantics; absent file: skip with a clear note, never invent rules). 6. **Coverage-weighted complexity** per in-scope function. 7. **Flakiness** - the tests the diff added or touched, repeated and shuffled until trusted. -8. **Mutation testing** over the in-scope source. +8. **Mutation testing** over the in-scope source - scored and reported in the wrap-up, never a reason on its own to block shipping (see gauntlet.md's "Mutation testing is reported, not a threshold"). Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. diff --git a/factory/phases/ship/references/gauntlet.md b/factory/phases/ship/references/gauntlet.md index 91a1911..990f5d8 100644 --- a/factory/phases/ship/references/gauntlet.md +++ b/factory/phases/ship/references/gauntlet.md @@ -50,8 +50,13 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve ## Thresholds are decisions -Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. +Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. A run never moves a threshold it is being judged by: the defaults above, plus any `D-` entry already in `docs/decisions.md` (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), are fixed for the whole run, and no phase writes a threshold decision into `docs/decisions.md`. A threshold this run cannot meet is a `failed` result naming the finding, its score, and the threshold it missed; when the evidence argues the line itself sits wrong, record that argument in `auto_decided` as a proposal for review, and keep judging this run by the unchanged threshold. Never adjust a threshold silently to make a run pass. + +## Mutation testing is reported, not a threshold + +The other seven checks above judge whether this run may ship; mutation testing does not, and it is the one check with no default threshold at all. A codebase's baseline kill rate depends on how much test coverage the diff's area already had going in, which this run did not choose - failing the run over that inherited debt would make ordinary changes to weakly-tested code unshippable regardless of what the diff itself does. +Fix agents still chase survivors under the loop above and the fix vocabulary's "kill a mutant" entry, and equivalent mutants still get marked and left, exactly as before. When a survivor resists two fix rounds (item 4 of the loop), record it and move on rather than escalating it as a blocker. The ship report's counts - killed, survived, uncovered, with the reason recorded against each stubborn survivor - are the evidence a person reads; there is no score below which the run fails. diff --git a/factory/phases/ship/references/tools.md b/factory/phases/ship/references/tools.md index 2e4a157..d2ed205 100644 --- a/factory/phases/ship/references/tools.md +++ b/factory/phases/ship/references/tools.md @@ -62,6 +62,8 @@ Keep it affordable: only diff-touched tests, reuse build caches between repeats, ## 8. Mutation testing +Scored and reported, never a shippability gate - see gauntlet.md's "Mutation testing is reported, not a threshold" for why and how a stubborn survivor is left in place rather than blocking the run. + Prefer the ecosystem's mutation framework - Stryker (JS/TS), mutmut or cosmic-ray (Python), PIT (JVM), cargo-mutants (Rust), go-mutesting (Go). These handle mutant generation, test selection, and reporting far better than a hand-rolled loop; write only the thin config that scopes them. Hand-roll only when the ecosystem has nothing: an agent-written script that applies one mutation at a time (flip `<` to `<=`, `==` to `!=`, `+` to `-`, negate conditions, drop return values), runs the narrowest relevant test command, and records survivors. @@ -70,7 +72,7 @@ Keeping it affordable: - **Scope to the diff**: mutate only in-scope files, run only the tests that cover them (most frameworks do incremental or per-file runs; use that). - Set a per-mutant test timeout so an infinite-loop mutant cannot hang the run. -- Equivalent mutants (mutations that provably cannot change behavior) are the one legitimate survivor category: mark them as such in the report with the reasoning, don't chase them forever - two fix rounds, then escalate per the loop rule. +- Equivalent mutants (mutations that provably cannot change behavior) are one legitimate reason to leave a survivor in place, marked as such with the reasoning; a survivor that resists two fix rounds for any other reason is left in place too, per the loop rule - killing every mutant is the goal, not the bar for shipping. ## Fix-agent prompts diff --git a/plugins/dev/skills/ship/SKILL.md b/plugins/dev/skills/ship/SKILL.md index cee3dda..9d43057 100644 --- a/plugins/dev/skills/ship/SKILL.md +++ b/plugins/dev/skills/ship/SKILL.md @@ -1,6 +1,6 @@ --- name: ship -description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every check passes, phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. +description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every gating check passes - mutation testing is scored and reported, never a shippability gate - phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. --- # Ship @@ -41,7 +41,7 @@ Never weaken or skip a check because acquiring its tool is work. 5. **Dependency rules** against `docs/dependencies.md` ([../../references/dependency-rules.md](../../references/dependency-rules.md) owns the checker semantics; absent file: skip with a clear note, never invent rules). 6. **Coverage-weighted complexity** per in-scope function. 7. **Flakiness** - the tests the diff added or touched, repeated and shuffled until trusted. -8. **Mutation testing** over the in-scope source. +8. **Mutation testing** over the in-scope source - scored and reported in the wrap-up, never a reason on its own to block shipping (see gauntlet.md's "Mutation testing is reported, not a threshold"). Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. diff --git a/plugins/dev/skills/ship/references/gauntlet.md b/plugins/dev/skills/ship/references/gauntlet.md index 27f13f4..7f79b6f 100644 --- a/plugins/dev/skills/ship/references/gauntlet.md +++ b/plugins/dev/skills/ship/references/gauntlet.md @@ -50,7 +50,12 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve ## Thresholds are decisions -Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. +Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. When the user accepts a different threshold, record it in `docs/decisions.md` as a `D-` entry (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), dated and sourced to this run; the next run reads it from there instead of re-arguing. Never adjust a threshold silently to make a run pass. + +## Mutation testing is reported, not a threshold + +The other seven checks above judge whether this run may ship; mutation testing does not, and it is the one check with no default threshold at all. A codebase's baseline kill rate depends on how much test coverage the diff's area already had going in, which this run did not choose - failing the run over that inherited debt would make ordinary changes to weakly-tested code unshippable regardless of what the diff itself does. +Fix agents still chase survivors under the loop above and the fix vocabulary's "kill a mutant" entry, and equivalent mutants still get marked and left, exactly as before. When a survivor resists two fix rounds (item 4 of the loop), present it and move on rather than treating it as a blocker. The ship report's counts - killed, survived, uncovered, with the reason recorded against each stubborn survivor - are the evidence a person reads; there is no score below which the run fails. diff --git a/plugins/dev/skills/ship/references/tools.md b/plugins/dev/skills/ship/references/tools.md index 64985db..3397f95 100644 --- a/plugins/dev/skills/ship/references/tools.md +++ b/plugins/dev/skills/ship/references/tools.md @@ -62,6 +62,8 @@ Keep it affordable: only diff-touched tests, reuse build caches between repeats, ## 8. Mutation testing +Scored and reported, never a shippability gate - see gauntlet.md's "Mutation testing is reported, not a threshold" for why and how a stubborn survivor is left in place rather than blocking the run. + Prefer the ecosystem's mutation framework - Stryker (JS/TS), mutmut or cosmic-ray (Python), PIT (JVM), cargo-mutants (Rust), go-mutesting (Go). These handle mutant generation, test selection, and reporting far better than a hand-rolled loop; write only the thin config that scopes them. Hand-roll only when the ecosystem has nothing: an agent-written script that applies one mutation at a time (flip `<` to `<=`, `==` to `!=`, `+` to `-`, negate conditions, drop return values), runs the narrowest relevant test command, and records survivors. @@ -70,7 +72,7 @@ Keeping it affordable: - **Scope to the diff**: mutate only in-scope files, run only the tests that cover them (most frameworks do incremental or per-file runs; use that). - Set a per-mutant test timeout so an infinite-loop mutant cannot hang the run. -- Equivalent mutants (mutations that provably cannot change behavior) are the one legitimate survivor category: mark them as such in the report with the reasoning, don't chase them forever - two fix rounds, then escalate per the loop rule. +- Equivalent mutants (mutations that provably cannot change behavior) are one legitimate reason to leave a survivor in place, marked as such with the reasoning; a survivor that resists two fix rounds for any other reason is left in place too, per the loop rule - killing every mutant is the goal, not the bar for shipping. ## Fix-agent prompts diff --git a/plugins/factory/phases/ship/SKILL.md b/plugins/factory/phases/ship/SKILL.md index a004142..f37157e 100644 --- a/plugins/factory/phases/ship/SKILL.md +++ b/plugins/factory/phases/ship/SKILL.md @@ -1,6 +1,6 @@ --- name: ship -description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every check passes, phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. +description: Run the quality pass that ships a change in three phases - phase 1 is a deterministic harden gauntlet (the repo's own static analysis, security scan, dead code, duplication, dependency rules, coverage-weighted complexity, test flakiness, mutation testing) that loops fresh-context fix agents until every gating check passes - mutation testing is scored and reported, never a shippability gate - phase 2 is a read-only adversarially verified review panel over the post-fix diff, writing .dev/{plan-name}/review_N.md, a BLOCK verdict gets two autonomous fix-and-re-review rounds before it counts as a real blocker, and phase 3 pushes and opens the pull request with visual proof of the change (e2e screenshots for a frontend, before/after state or a red-then-green reproducing test otherwise) gated by a deterministic evidence check. Use after build to finish a change and open its PR, or on request for a single phase such as gauntlet only or review only. --- # Ship @@ -41,7 +41,7 @@ Never weaken or skip a check because acquiring its tool is work. 5. **Dependency rules** against `docs/dependencies.md` ([../../references/dependency-rules.md](../../references/dependency-rules.md) owns the checker semantics; absent file: skip with a clear note, never invent rules). 6. **Coverage-weighted complexity** per in-scope function. 7. **Flakiness** - the tests the diff added or touched, repeated and shuffled until trusted. -8. **Mutation testing** over the in-scope source. +8. **Mutation testing** over the in-scope source - scored and reported in the wrap-up, never a reason on its own to block shipping (see gauntlet.md's "Mutation testing is reported, not a threshold"). Run each check to completion per the loop in [references/gauntlet.md](references/gauntlet.md). When any check dispatched fixes, end the phase with the e2e refresh in the same reference: re-run the spec's `[e2e]` scenarios and overwrite the report, so phase 2 judges the post-fix code instead of stale evidence. diff --git a/plugins/factory/phases/ship/references/gauntlet.md b/plugins/factory/phases/ship/references/gauntlet.md index 91a1911..990f5d8 100644 --- a/plugins/factory/phases/ship/references/gauntlet.md +++ b/plugins/factory/phases/ship/references/gauntlet.md @@ -50,8 +50,13 @@ Fix agents never edit thresholds, rules files, or the tools themselves, and neve ## Thresholds are decisions -Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched; zero surviving mutants in scope. +Defaults: zero static analysis findings in scope; zero security findings; zero dead symbols in scope; no new clones over the 50-token threshold; zero dependency violations; complexity-coverage score at most 6 per function; zero flaky tests among those the diff touched. Agent-written code tolerates a higher complexity threshold than the human default of 4 - agents hold more paths in working memory - but where the line sits is a decision, not a config value. A run never moves a threshold it is being judged by: the defaults above, plus any `D-` entry already in `docs/decisions.md` (notation in [../../../references/decision-ledger.md](../../../references/decision-ledger.md)), are fixed for the whole run, and no phase writes a threshold decision into `docs/decisions.md`. A threshold this run cannot meet is a `failed` result naming the finding, its score, and the threshold it missed; when the evidence argues the line itself sits wrong, record that argument in `auto_decided` as a proposal for review, and keep judging this run by the unchanged threshold. Never adjust a threshold silently to make a run pass. + +## Mutation testing is reported, not a threshold + +The other seven checks above judge whether this run may ship; mutation testing does not, and it is the one check with no default threshold at all. A codebase's baseline kill rate depends on how much test coverage the diff's area already had going in, which this run did not choose - failing the run over that inherited debt would make ordinary changes to weakly-tested code unshippable regardless of what the diff itself does. +Fix agents still chase survivors under the loop above and the fix vocabulary's "kill a mutant" entry, and equivalent mutants still get marked and left, exactly as before. When a survivor resists two fix rounds (item 4 of the loop), record it and move on rather than escalating it as a blocker. The ship report's counts - killed, survived, uncovered, with the reason recorded against each stubborn survivor - are the evidence a person reads; there is no score below which the run fails. diff --git a/plugins/factory/phases/ship/references/tools.md b/plugins/factory/phases/ship/references/tools.md index 2e4a157..d2ed205 100644 --- a/plugins/factory/phases/ship/references/tools.md +++ b/plugins/factory/phases/ship/references/tools.md @@ -62,6 +62,8 @@ Keep it affordable: only diff-touched tests, reuse build caches between repeats, ## 8. Mutation testing +Scored and reported, never a shippability gate - see gauntlet.md's "Mutation testing is reported, not a threshold" for why and how a stubborn survivor is left in place rather than blocking the run. + Prefer the ecosystem's mutation framework - Stryker (JS/TS), mutmut or cosmic-ray (Python), PIT (JVM), cargo-mutants (Rust), go-mutesting (Go). These handle mutant generation, test selection, and reporting far better than a hand-rolled loop; write only the thin config that scopes them. Hand-roll only when the ecosystem has nothing: an agent-written script that applies one mutation at a time (flip `<` to `<=`, `==` to `!=`, `+` to `-`, negate conditions, drop return values), runs the narrowest relevant test command, and records survivors. @@ -70,7 +72,7 @@ Keeping it affordable: - **Scope to the diff**: mutate only in-scope files, run only the tests that cover them (most frameworks do incremental or per-file runs; use that). - Set a per-mutant test timeout so an infinite-loop mutant cannot hang the run. -- Equivalent mutants (mutations that provably cannot change behavior) are the one legitimate survivor category: mark them as such in the report with the reasoning, don't chase them forever - two fix rounds, then escalate per the loop rule. +- Equivalent mutants (mutations that provably cannot change behavior) are one legitimate reason to leave a survivor in place, marked as such with the reasoning; a survivor that resists two fix rounds for any other reason is left in place too, per the loop rule - killing every mutant is the goal, not the bar for shipping. ## Fix-agent prompts From d2eec7d8067b3c1d618adedf7abb304a98b29b4c Mon Sep 17 00:00:00 2001 From: tobrun Date: Fri, 25 Sep 2026 20:33:00 +0200 Subject: [PATCH 28/30] feat(bootstrap): add the agents-md skill that writes a repo's AGENTS.md What: a third plugin, bootstrap, with one skill, agents-md. It probes a consuming repository read-only (manifests, task runners, CI workflows, test layout, git history), maps what it finds onto a closed concept map distilled from the dev skills (test layers, mocking at boundaries, the e2e environment, CI parity, fix-at-source quality, commits, plans, PR evidence), interviews only for what the probe cannot settle, runs the commands it will list, merges and prunes an existing AGENTS.md and CLAUDE.md, and writes the root AGENTS.md after the user approves a diff. Gaps become open items; nothing is scaffolded, CLAUDE.md is never written, and nothing is committed. check-agents-md.py is the gate every draft loops against: required sections in order, the seven Commands slots filled or backed by an open item, a closed open-item slug vocabulary, stale backticked paths and links, commands whose package script, make target, just recipe, task, or file does not exist, and a default branch that disagrees with origin/HEAD. Checks it cannot settle statically print notes instead of violations, so a false positive never traps the loop. 60 unittest cases cover it and tie the concept map, the template, and the checker together; validate.sh runs them as B01. Wiring: marketplace entries for Claude Code and Codex, a generator entry that copies only skills/, the generated plugins/bootstrap tree, and a README row. Pi stays dev-only. Why: the dev workflow's lessons only reach repositories that install dev, and the repo facts its skills need (per-layer commands, the e2e launch and environment, the merge gate, the scope vocabulary) are re-probed on every run because no file declares them. A root AGENTS.md written from a probe carries both to any agent. Three functional runs on scratch copies of sibling repositories drove the checker fixes (`bun turbo`, test path filters, naming conventions) and are recorded in bootstrap/evals/results.md. Considered: a dev skill (grows dev's namespace and the Pi package for a one-time step); linking into dev/references (dangles in the generated Codex tree); a CLAUDE.md pointer file (Claude Code reads AGENTS.md directly). --- .agents/plugins/marketplace.json | 12 + .claude-plugin/marketplace.json | 5 + README.md | 9 +- bootstrap/.claude-plugin/plugin.json | 8 + bootstrap/README.md | 42 ++ bootstrap/evals/README.md | 15 + bootstrap/evals/agents-md.json | 67 +++ bootstrap/evals/results.md | 19 + bootstrap/evals/tests/__init__.py | 0 bootstrap/evals/tests/test_check_agents_md.py | 345 +++++++++++ bootstrap/evals/tests/test_concept_sources.py | 70 +++ bootstrap/skills/agents-md/SKILL.md | 76 +++ .../skills/agents-md/references/concepts.md | 38 ++ .../skills/agents-md/references/merge.md | 34 ++ .../skills/agents-md/references/probe.md | 59 ++ .../skills/agents-md/references/template.md | 77 +++ .../agents-md/scripts/check-agents-md.py | 552 ++++++++++++++++++ plugins/bootstrap/.codex-plugin/plugin.json | 24 + plugins/bootstrap/README.md | 3 + plugins/bootstrap/skills/agents-md/SKILL.md | 75 +++ .../skills/agents-md/agents/openai.yaml | 6 + .../skills/agents-md/references/concepts.md | 38 ++ .../skills/agents-md/references/merge.md | 34 ++ .../skills/agents-md/references/probe.md | 59 ++ .../skills/agents-md/references/template.md | 77 +++ .../agents-md/scripts/check-agents-md.py | 552 ++++++++++++++++++ scripts/build_codex_plugin.py | 15 + scripts/validate.sh | 22 + 28 files changed, 2330 insertions(+), 3 deletions(-) create mode 100644 bootstrap/.claude-plugin/plugin.json create mode 100644 bootstrap/README.md create mode 100644 bootstrap/evals/README.md create mode 100644 bootstrap/evals/agents-md.json create mode 100644 bootstrap/evals/results.md create mode 100644 bootstrap/evals/tests/__init__.py create mode 100644 bootstrap/evals/tests/test_check_agents_md.py create mode 100644 bootstrap/evals/tests/test_concept_sources.py create mode 100644 bootstrap/skills/agents-md/SKILL.md create mode 100644 bootstrap/skills/agents-md/references/concepts.md create mode 100644 bootstrap/skills/agents-md/references/merge.md create mode 100644 bootstrap/skills/agents-md/references/probe.md create mode 100644 bootstrap/skills/agents-md/references/template.md create mode 100755 bootstrap/skills/agents-md/scripts/check-agents-md.py create mode 100644 plugins/bootstrap/.codex-plugin/plugin.json create mode 100644 plugins/bootstrap/README.md create mode 100644 plugins/bootstrap/skills/agents-md/SKILL.md create mode 100644 plugins/bootstrap/skills/agents-md/agents/openai.yaml create mode 100644 plugins/bootstrap/skills/agents-md/references/concepts.md create mode 100644 plugins/bootstrap/skills/agents-md/references/merge.md create mode 100644 plugins/bootstrap/skills/agents-md/references/probe.md create mode 100644 plugins/bootstrap/skills/agents-md/references/template.md create mode 100755 plugins/bootstrap/skills/agents-md/scripts/check-agents-md.py diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index e1b760a..738da1a 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -27,6 +27,18 @@ "authentication": "ON_INSTALL" }, "category": "Developer Tools" + }, + { + "name": "bootstrap", + "source": { + "source": "local", + "path": "./plugins/bootstrap" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "Developer Tools" } ] } diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1b0725e..78662cf 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -16,6 +16,11 @@ "name": "factory", "source": "./factory", "description": "An orchestrator skill that drives a pipeline of typed phases declared in .factory/config.yaml (the built-in scope, scope-review, build, and ship by default), judging each phase's completion itself, ending in a pull request or a report." + }, + { + "name": "bootstrap", + "source": "./bootstrap", + "description": "Probe a repository's stack and write or refresh its root AGENTS.md, encoding the dev workflow's SDLC lessons (test layers, mocking at boundaries, CI parity, fix-at-source quality, commit and PR conventions) with the repo's exact commands and its gaps as open items." } ] } diff --git a/README.md b/README.md index 195027c..ce9f315 100644 --- a/README.md +++ b/README.md @@ -4,26 +4,29 @@ | ------ | -------- | ----- | | [dev](dev/) | A test-focused development workflow for Claude Code, Codex, opencode, and Pi. | `scope`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz` | | [factory](factory/) | Take a request from scope to a shipped pull request unattended, on Claude Code or Codex. | `run` | +| [bootstrap](bootstrap/) | Prepare any repository for agent work: probe its stack and write its root AGENTS.md from the workflow's SDLC lessons, on Claude Code or Codex. | `agents-md` | ## Claude Code ```bash /plugin marketplace add tobrun/workflow /plugin install dev@nurbot +/plugin install bootstrap@nurbot ``` -Invoke skills as `/scope`, `/build`, and so on, or namespaced as `/dev:scope`. +Invoke skills as `/scope`, `/build`, and so on, or namespaced as `/dev:scope`; bootstrap's skill is `/bootstrap:agents-md`. ## Codex ```bash codex plugin marketplace add tobrun/workflow codex plugin add dev@nurbot +codex plugin add bootstrap@nurbot ``` -Invoke skills as `$dev:scope`, `$dev:build`, and so on. +Invoke skills as `$dev:scope`, `$dev:build`, `$bootstrap:agents-md`, and so on. The distribution is explicit-invocation only. -The checked-in Codex package under `plugins/` is generated from `dev/`: +The checked-in Codex packages under `plugins/` are generated from each source plugin: ```bash python3 scripts/build_codex_plugin.py diff --git a/bootstrap/.claude-plugin/plugin.json b/bootstrap/.claude-plugin/plugin.json new file mode 100644 index 0000000..8af036f --- /dev/null +++ b/bootstrap/.claude-plugin/plugin.json @@ -0,0 +1,8 @@ +{ + "name": "bootstrap", + "version": "0.1.0", + "description": "Prepare a repository for agent work: probe its stack, map it onto the dev workflow's SDLC concepts (test layers, mocking at boundaries, CI parity, fix-at-source quality, commit and PR conventions), and write or refresh its root AGENTS.md with the repo's exact commands and its gaps as open items. Use on any repository, whether or not the dev plugin is installed.", + "author": { + "name": "Tobrun" + } +} diff --git a/bootstrap/README.md b/bootstrap/README.md new file mode 100644 index 0000000..2834fb5 --- /dev/null +++ b/bootstrap/README.md @@ -0,0 +1,42 @@ +# bootstrap + +Prepare a repository for agent work. +The `agents-md` skill probes the repository's stack, maps what it finds onto the SDLC concepts the [dev](../dev/) workflow is built on, and writes or refreshes the root `AGENTS.md`. + +The file it writes encodes those lessons as short outcome rules any agent can follow, with or without the dev plugin installed: +- tests on the cheapest layer that can fail for the right reason, and mocks only at boundaries the repo does not own; +- e2e against the built artifact in a stubbed, seeded environment, never production or shared staging; +- done means the merge gate passes locally, and findings are fixed at the source, never suppressed or retried away; +- conventional commits with a what and a why, plans kept out of git, and pull requests that carry evidence. + +It fills those rules in with the repository's exact commands (setup, build, check, unit, integration, e2e, and the merge gate), verified by running them, and records every missing concept as an open item instead of scaffolding it. + +## What it touches + +Only the root `AGENTS.md`. +An existing AGENTS.md or CLAUDE.md is merged and pruned: repo-specific gotchas are kept, stale facts are updated, and a rule that contradicts a lesson becomes a question rather than a silent overwrite. +CLAUDE.md is read but never written; Claude Code reads AGENTS.md directly, so the closing report recommends removing a now-redundant CLAUDE.md. +Nothing is committed; the diff is shown for approval before the file is written. + +Every draft loops against `skills/agents-md/scripts/check-agents-md.py` until it exits clean: required sections in order, every Commands slot filled or backed by an open item, no stale path, no command whose script, target, or file does not exist, and the stated default branch matching `origin/HEAD`. + +## Install + +### Claude Code + +```bash +/plugin marketplace add tobrun/workflow +/plugin install bootstrap@nurbot +``` + +Invoke it as `/bootstrap:agents-md`. + +### Codex + +```bash +codex plugin marketplace add tobrun/workflow +codex plugin add bootstrap@nurbot +``` + +Invoke it as `$bootstrap:agents-md`. +The distribution is explicit-invocation only. diff --git a/bootstrap/evals/README.md b/bootstrap/evals/README.md new file mode 100644 index 0000000..b25a376 --- /dev/null +++ b/bootstrap/evals/README.md @@ -0,0 +1,15 @@ +# Bootstrap evals + +Two levels of verification: + +1. **Unit tests, free**: `python3 -m unittest discover -s bootstrap/evals/tests -t .`, run on every `scripts/validate.sh` invocation as check B01. + `test_check_agents_md.py` proves the checker's CLI contract against scratch repositories. + `test_concept_sources.py` keeps the concept map, the template, and the checker agreeing, and fails when a dev source file the map was distilled from moves or is renamed; it cannot see a source whose content changed in place. +2. **Functional runs, paid and manual**: `agents-md.json` lists the fixtures and assertions, and `results.md` records the runs. + +## The functional procedure + +1. For each eval, make a scratch copy of a repository matching its fixture with `git clone --local`, so history and `origin/HEAD` come along and the live repository is never touched. +2. Install the plugin (`/plugin install bootstrap@nurbot`, or `codex plugin add bootstrap@nurbot`) and invoke `/bootstrap:agents-md` in the copy with the eval's prompt. +3. Answer the interview as the fixture's owner would. +4. Grade the run against the eval's assertions and the `every_eval` list, then record the result in `results.md`. diff --git a/bootstrap/evals/agents-md.json b/bootstrap/evals/agents-md.json new file mode 100644 index 0000000..b002040 --- /dev/null +++ b/bootstrap/evals/agents-md.json @@ -0,0 +1,67 @@ +{ + "skill": "agents-md", + "mode": "functional", + "note": "Run each eval on a scratch copy made with git clone --local (it keeps history and origin/HEAD), never on the live repository. results.md names the reference repository and commit behind each fixture.", + "evals": [ + { + "id": "merge-prune", + "prompt": "Write this repository's AGENTS.md.", + "fixture": "an Astro blog with both an AGENTS.md and a CLAUDE.md; AGENTS.md cites a config path that has since moved and contains em dashes; no test runner; one GitHub workflow that only deploys on push", + "assertions": [ + "The moved path is rewritten to the file that exists now, and the table cites the probe evidence", + "No em dash remains in the written file", + "Unit, Integration, and E2E are none, each with a matching open item, and there is an open item with slug ci because the deploy workflow is not a merge gate", + "CLAUDE.md is not modified, and the closing report recommends deleting or trimming it", + "The diff and the kept/updated/pruned table are shown before AGENTS.md is written" + ] + }, + { + "id": "claude-only-python", + "prompt": "Bootstrap an AGENTS.md here.", + "fixture": "a Python service with a Makefile whose check target runs lint, types, and tests, a pull-request workflow that runs make check, a robot framework suite that drives the built service, and a CLAUDE.md with two repo-specific gotchas but no AGENTS.md", + "assertions": [ + "The Merge gate slot is `make check`", + "The robot suite is recognized as the e2e layer, and its environment is either described in Tests or recorded as an e2e-env open item", + "Both CLAUDE.md gotchas appear in AGENTS.md in the section they concern", + "CLAUDE.md is not modified, and the closing report recommends deleting or trimming it" + ] + }, + { + "id": "rich-existing", + "prompt": "Refresh our AGENTS.md against the current code.", + "fixture": "a bun monorepo with a curated AGENTS.md that says tests must not run from the root, an e2e script that needs a live SSO session and a real access token, and conventional commits with scopes such as core, mapbox, server, and tui", + "assertions": [ + "The rule that tests must not run from the root is kept", + "The live e2e run is recorded as an e2e-env open item rather than presented as a mocked e2e layer", + "The scope list names the scopes the history actually uses", + "A second run on the written file produces a diff limited to facts that changed" + ] + }, + { + "id": "bare", + "prompt": "Write an AGENTS.md for this repo.", + "fixture": "a repository holding a single index.html game and a README, no package manifest, no CI, a handful of unconventional commits", + "assertions": [ + "The skill asks for commands it cannot find instead of guessing a runner", + "Every Commands slot the user cannot answer is none or n/a with a matching open item", + "The written file is under 60 lines" + ] + }, + { + "id": "conflict", + "prompt": "Update AGENTS.md.", + "fixture": "a JS repository whose AGENTS.md says to retry flaky e2e tests twice and whose playwright config sets retries: 2", + "assertions": [ + "The retry rule is raised as an interview question with a recommendation and a confidence score", + "The retry rule is neither silently kept nor silently removed", + "The written file does not state both the retry rule and the no-retries rule" + ] + } + ], + "every_eval": [ + "check-agents-md.py exits 0 on the written AGENTS.md", + "No file other than the root AGENTS.md changes in the repository", + "Nothing is committed", + "No question asks about a fact the probe table already resolved" + ] +} diff --git a/bootstrap/evals/results.md b/bootstrap/evals/results.md new file mode 100644 index 0000000..c584da5 --- /dev/null +++ b/bootstrap/evals/results.md @@ -0,0 +1,19 @@ +# Eval Results + +Status: three functional runs recorded on 2026-09-25, before the skill's first release. + +Every run executed the skill text directly in a Claude Code subagent (not through an installed plugin), on `git clone --local` scratch copies, with a simulated user who accepted every recommendation, skipped the closing newcomer question, and declined installs. + +| Eval | Reference repository and commit | Host | Result | Notes | +| ---- | ------------------------------- | ---- | ------ | ----- | +| merge-prune | `~/ws/blog` at 0a55308, plus its untracked AGENTS.md | Claude Code (subagent) | pass (5/5 and every_eval) | stale `src/content/config.ts` updated, em dashes gone, test layers and ci recorded as open items, CLAUDE.md untouched and flagged, checker clean on the first iteration | +| rich-existing | `~/ws/mapcode` at 0eb9c91 | Claude Code (subagent) | pass (3/4; the re-run assertion was exercised on the blog copy below) | root-test rule kept, live e2e recorded as `e2e-env`, real scopes used; also caught `.dev/` not being gitignored and a stale models-generator claim | +| re-run (merge-prune output) | the blog copy after the first run | Claude Code (subagent) | pass | 70 of 72 lines kept verbatim; the 2 changes corrected an unconditional analytics claim that the code makes conditional | + +## Fixes the runs drove + +- The checker rejected `bun turbo` (a dependency's binary) as a missing script; `bun`, `yarn`, and implicit `pnpm` calls now fall back to declared dependencies and print a note. +- The checker now checks path filters on `test` subcommands, knows framework and asset extensions, notes unknown runners in Commands, and exempts naming conventions such as `kebab-case.mdx`. +- SKILL.md: the skill root is defined, installs of any kind need a question, a command whose dependencies are missing "cannot be run" rather than fails, and defects noticed while probing go to the closing report. +- References: scope fallback for short or non-conventional histories, e2e harnesses that drive source, configured retries as open items, conflicts limited to rules an agent file states, and approval tables grouped by section. +- A refresh starts from the existing file, the template's rule lines are exempt from pruning, and the approval table counts kept lines instead of listing them. diff --git a/bootstrap/evals/tests/__init__.py b/bootstrap/evals/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/bootstrap/evals/tests/test_check_agents_md.py b/bootstrap/evals/tests/test_check_agents_md.py new file mode 100644 index 0000000..c211d98 --- /dev/null +++ b/bootstrap/evals/tests/test_check_agents_md.py @@ -0,0 +1,345 @@ +"""Unit tests for bootstrap/skills/agents-md/scripts/check-agents-md.py. + +Run from the repo root: python3 -m unittest discover -s bootstrap/evals/tests -t . +Each test builds a scratch repository and runs the checker as a subprocess, so it +proves the real CLI contract the agents-md skill loops against. +""" + +from __future__ import annotations + +import json +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +SCRIPT = REPO_ROOT / "bootstrap" / "skills" / "agents-md" / "scripts" / "check-agents-md.py" +EM_DASH = "\u2014" + +CLEAN = """# Fixture service + +A fixture service that exists to exercise the checker. +Default branch is `main`. + +## Commands + +- Setup: `npm install` +- Build: `npm run build` +- Check: `npm run check` +- Unit: `npm test` +- Integration: `npm run test:integration` +- E2E: none +- Merge gate: `make ci` + +## Tests + +Unit tests live next to the code in `src/`; integration tests live in `tests/integration/`. +Put each check on the cheapest layer that can fail for the right reason. + +## Quality + +Done means every merge-gate command passes locally. + +## Architecture + +The service reads `config/app.json` at startup. + +## Contributions + +Commit messages follow `type(scope): subject` with a What: and Why: body. + +## Open items + +- [ ] e2e: no layer drives the built service end to end. + +## Maintaining this file + +Keep only knowledge that almost every agent session needs. +""" + + +class CheckerTest(unittest.TestCase): + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + self.root = Path(self._tmp.name) + self.write("package.json", json.dumps({"scripts": { + "build": "tsc", "check": "tsc --noEmit", "test": "vitest run", "test:integration": "vitest run tests", + }})) + self.write("Makefile", "ci: check\n\tnpm run check\n\ncheck:\n\tnpm test\n") + self.write("src/index.ts", "export {};\n") + self.write("tests/integration/store.test.ts", "export {};\n") + self.write("config/app.json", "{}\n") + + def tearDown(self) -> None: + self._tmp.cleanup() + + def write(self, relative: str, content: str) -> Path: + path = self.root / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content, encoding="utf-8") + return path + + def git(self, *args: str) -> None: + subprocess.run(["git", "-C", str(self.root), *args], check=True, capture_output=True) + + def run_checker(self, text: str, *extra: str) -> subprocess.CompletedProcess: + self.write("AGENTS.md", text) + return subprocess.run( + [sys.executable, str(SCRIPT), str(self.root / "AGENTS.md"), "--root", str(self.root), *extra], + capture_output=True, text=True, + ) + + def assert_clean(self, text: str) -> subprocess.CompletedProcess: + result = self.run_checker(text) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + return result + + def assert_violation(self, text: str, fragment: str) -> None: + result = self.run_checker(text) + self.assertEqual(result.returncode, 1, result.stdout + result.stderr) + self.assertIn(fragment, result.stdout) + + +class ContractTest(CheckerTest): + def test_a_clean_file_passes(self) -> None: + result = self.assert_clean(CLEAN) + self.assertIn("check-agents-md: ok", result.stdout) + + def test_a_missing_file_exits_two(self) -> None: + result = subprocess.run( + [sys.executable, str(SCRIPT), str(self.root / "AGENTS.md"), "--root", str(self.root)], + capture_output=True, text=True, + ) + self.assertEqual(result.returncode, 2) + self.assertIn("missing:", result.stdout) + + def test_an_unknown_argument_exits_two(self) -> None: + self.write("AGENTS.md", CLEAN) + result = subprocess.run( + [sys.executable, str(SCRIPT), str(self.root / "AGENTS.md"), "--bogus"], capture_output=True, text=True, + ) + self.assertEqual(result.returncode, 2) + + +class StructureTest(CheckerTest): + def test_a_missing_section_is_a_violation(self) -> None: + text = CLEAN.replace("## Quality\n\nDone means every merge-gate command passes locally.\n\n", "") + self.assert_violation(text, "structure: missing '## Quality'") + + def test_sections_out_of_order_are_a_violation(self) -> None: + text = CLEAN.replace("## Tests", "## Placeholder").replace("## Quality", "## Tests").replace("## Placeholder", "## Quality") + self.assert_violation(text, "structure: sections out of order") + + def test_a_duplicated_section_is_a_violation(self) -> None: + text = CLEAN.replace("## Architecture\n", "## Architecture\n\nFirst half.\n\n## Architecture\n") + self.assert_violation(text, "'## Architecture' appears more than once") + + def test_maintaining_this_file_must_come_last(self) -> None: + self.assert_violation(CLEAN + "\n## Company knowledge\n\nLives in the wiki.\n", "must be the last section") + + def test_an_extra_section_before_maintaining_passes(self) -> None: + text = CLEAN.replace("## Open items", "## Company knowledge\n\nThe glossary lives in `config/app.json`.\n\n## Open items") + self.assert_clean(text) + + def test_two_titles_are_a_violation(self) -> None: + self.assert_violation("# Second title\n\n" + CLEAN, "expected exactly one '# ' title, found 2") + + def test_an_intro_without_the_default_branch_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("Default branch is `main`.\n", ""), "no 'Default branch is `x`' sentence") + + def test_headings_inside_a_code_fence_are_ignored(self) -> None: + text = CLEAN.replace("## Architecture\n", "## Architecture\n\n```\n## Tests\n# not a title\n```\n") + self.assert_clean(text) + + +class TextTest(CheckerTest): + def test_a_file_over_the_line_cap_is_a_violation(self) -> None: + result = self.run_checker(CLEAN, "--max-lines", "20") + self.assertEqual(result.returncode, 1) + self.assertIn("over the 20-line cap", result.stdout) + + def test_an_em_dash_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("fixture service that", f"fixture service {EM_DASH} that"), "em-dash: line 3") + + def test_a_leftover_fill_slot_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`make ci`", "<>"), "placeholder:") + + def test_a_repeated_line_is_a_violation(self) -> None: + line = "Put each check on the cheapest layer that can fail for the right reason.\n" + self.assert_violation(CLEAN.replace("## Quality\n", "## Quality\n\n" + line), "duplicate:") + + +class SlotTest(CheckerTest): + def test_a_missing_slot_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("- Build: `npm run build`\n", ""), "missing the '- Build:' slot") + + def test_none_without_its_open_item_is_a_violation(self) -> None: + text = CLEAN.replace("- [ ] e2e: no layer drives the built service end to end.", "- [ ] ci: nothing gates pull requests.") + self.assert_violation(text, "'E2E: none' (line 13) has no '- [ ] e2e:' open item") + + def test_na_with_a_reason_passes(self) -> None: + self.assert_clean(CLEAN.replace("- Build: `npm run build`", "- Build: n/a (a library published as source)")) + + def test_a_prose_slot_value_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`npm run build`", "run the build script"), "must be a backticked command") + + def test_merge_gate_none_needs_a_ci_open_item(self) -> None: + text = CLEAN.replace("- Merge gate: `make ci`", "- Merge gate: none") + self.assert_violation(text, "'Merge gate: none'") + self.assert_clean(text.replace("## Maintaining", "- [ ] ci: no workflow gates pull requests.\n\n## Maintaining").replace( + "end to end.\n\n- [ ] ci", "end to end.\n- [ ] ci")) + + +class OpenItemTest(CheckerTest): + def test_a_malformed_item_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("- [ ] e2e: no layer", "- e2e: no layer"), "is not '- [ ] slug: text'") + + def test_an_unknown_slug_is_a_violation(self) -> None: + text = CLEAN.replace("drives the built service end to end.", "drives the built service end to end.\n- [ ] vibes: unclear.") + self.assert_violation(text, "unknown slug 'vibes'") + + def test_an_empty_open_items_section_is_a_violation(self) -> None: + text = CLEAN.replace("- E2E: none", "- E2E: `npm run test:integration -- e2e`").replace( + "- [ ] e2e: no layer drives the built service end to end.\n", "") + self.assert_violation(text, "'## Open items' is empty") + + def test_a_path_inside_an_open_item_is_not_checked(self) -> None: + text = CLEAN.replace("end to end.", "end to end; add one under `e2e/journeys/`.") + self.assert_clean(text) + + +class PathTest(CheckerTest): + def test_a_stale_path_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("config/app.json", "config/settings.json"), "stale: `config/settings.json`") + + def test_a_missing_but_gitignored_path_passes(self) -> None: + self.git("init", "-q") + self.write(".gitignore", "dist/\n") + self.assert_clean(CLEAN.replace("Unit tests live", "Builds land in `dist/`.\nUnit tests live")) + + def test_a_url_and_a_git_ref_are_not_paths(self) -> None: + text = CLEAN.replace("Unit tests live", "Docs are at `https://example.com/a/b.md`; compare with `origin/main`.\nUnit tests live") + self.assert_clean(text) + + def test_a_bracketed_route_file_is_a_literal_path(self) -> None: + self.write("src/pages/[slug].astro", "---\n---\n") + self.assert_clean(CLEAN.replace("Unit tests live", "Routes such as `src/pages/[slug].astro` render posts.\nUnit tests live")) + + def test_a_bare_filename_found_anywhere_in_the_repo_passes(self) -> None: + self.write("tests/conftest.py", "\n") + self.assert_clean(CLEAN.replace("Unit tests live", "Shared fixtures sit in `conftest.py`.\nUnit tests live")) + + def test_a_naming_convention_is_not_a_path(self) -> None: + self.assert_clean(CLEAN.replace("Unit tests live", "Name posts `kebab-case.mdx` and components `PascalCase.tsx`.\nUnit tests live")) + + def test_a_framework_file_extension_is_checked(self) -> None: + self.assert_violation(CLEAN.replace("Unit tests live", "The header is `Heder.astro`.\nUnit tests live"), "stale: `Heder.astro`") + + def test_a_bare_filename_found_nowhere_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("Unit tests live", "See `ContactCard.tsx`.\nUnit tests live"), "stale: `ContactCard.tsx`") + + def test_a_stale_relative_link_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("Unit tests live", "See [the guide](docs/guide.md).\nUnit tests live"), "link 'docs/guide.md'") + + def test_the_plan_directory_is_exempt(self) -> None: + self.assert_clean(CLEAN.replace("with a What: and Why: body.", "with a What: and Why: body.\nPlans live under `.dev/`.")) + + +class CommandTest(CheckerTest): + def test_a_missing_npm_script_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("npm run build", "npm run bundle"), "`bundle` is not a script in package.json") + + def test_cd_resolves_against_the_nested_package(self) -> None: + self.write("packages/api/package.json", json.dumps({"scripts": {"test": "bun test"}})) + self.assert_clean(CLEAN.replace("`npm test`", "`cd packages/api && bun run test`")) + self.assert_violation(CLEAN.replace("`npm test`", "`cd packages/api && bun run lint`"), "not a script in packages/api/package.json") + + def test_cd_into_a_missing_directory_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`npm test`", "`cd packages/gone && npm test`"), "cds into missing `packages/gone`") + + def test_bun_runs_a_file_or_a_script(self) -> None: + self.write("e2e/run.ts", "\n") + self.assert_clean(CLEAN.replace("- E2E: none", "- E2E: `bun e2e/run.ts`").replace( + "- [ ] e2e: no layer drives the built service end to end.", "- [ ] e2e-env: the run needs a live token.")) + self.assert_violation(CLEAN.replace("`npm test`", "`bun dev`"), "`dev` is not a script in package.json") + + def test_bun_test_is_a_builtin(self) -> None: + self.assert_clean(CLEAN.replace("`npm test`", "`bun test`")) + + def test_a_missing_make_target_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`make ci`", "`make verify`"), "make target(s) verify not defined") + + def test_a_make_target_from_an_included_mk_file_passes(self) -> None: + self.write("build/rules.mk", "verify:\n\ttrue\n") + self.assert_clean(CLEAN.replace("`make ci`", "`make verify`")) + + def test_a_missing_script_file_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`npm run check`", "`python3 scripts/check.py`"), "`scripts/check.py` does not exist") + + def test_a_missing_wrapper_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`npm run check`", "`./gradlew check`"), "`./gradlew` does not exist") + + def test_a_just_recipe_is_resolved(self) -> None: + self.write("justfile", "check:\n npm run check\n") + self.assert_clean(CLEAN.replace("`npm run check`", "`just check`")) + self.assert_violation(CLEAN.replace("`npm run check`", "`just lint`"), "just recipe `lint` not defined") + + def test_a_task_is_resolved(self) -> None: + self.write("Taskfile.yml", "version: '3'\ntasks:\n check:\n cmds:\n - npm run check\n") + self.assert_clean(CLEAN.replace("`npm run check`", "`task check`")) + self.assert_violation(CLEAN.replace("`npm run check`", "`task lint`"), "task `lint` not defined") + + def test_an_unresolvable_runner_is_only_a_note(self) -> None: + result = self.assert_clean(CLEAN.replace("`npm run check`", "`npx tsc --noEmit`")) + self.assertIn("note: `npx tsc --noEmit`", result.stdout) + + def test_an_environment_prefix_is_skipped(self) -> None: + self.assert_clean(CLEAN.replace("`npm test`", "`CI=1 npm test`")) + + def test_bun_running_a_declared_dependency_binary_is_only_a_note(self) -> None: + self.write("package.json", json.dumps({ + "scripts": {"build": "tsc", "check": "tsc", "test": "vitest", "test:integration": "vitest"}, + "devDependencies": {"turbo": "2.0.0"}, + })) + result = self.assert_clean(CLEAN.replace("`make ci`", "`bun turbo test --filter=core`")) + self.assertIn("note: `bun turbo test --filter=core`", result.stdout) + + def test_bun_running_an_undeclared_name_is_a_violation(self) -> None: + self.assert_violation(CLEAN.replace("`make ci`", "`bun turbo test`"), "`turbo` is not a script in package.json") + + def test_a_test_path_filter_must_prefix_a_real_path(self) -> None: + self.assert_clean(CLEAN.replace("`npm run test:integration`", "`bun test tests/integ`")) + self.assert_violation(CLEAN.replace("`npm run test:integration`", "`bun test tests/gone`"), "`tests/gone` matches nothing in .") + + def test_an_unknown_runner_in_commands_is_only_a_note(self) -> None: + result = self.assert_clean(CLEAN.replace("`npm run check`", "`astro check`")) + self.assertIn("note: `astro check`", result.stdout) + + +class BranchTest(CheckerTest): + def setUp(self) -> None: + super().setUp() + self.git("init", "-q", "-b", "main") + self.git("-c", "user.name=t", "-c", "user.email=t@example.com", "commit", "-q", "--allow-empty", "-m", "init") + + def point_origin_head(self, branch: str) -> None: + self.git("update-ref", f"refs/remotes/origin/{branch}", "HEAD") + self.git("symbolic-ref", "refs/remotes/origin/HEAD", f"refs/remotes/origin/{branch}") + + def test_a_matching_default_branch_passes(self) -> None: + self.point_origin_head("main") + self.assert_clean(CLEAN) + + def test_a_mismatched_default_branch_is_a_violation(self) -> None: + self.point_origin_head("develop") + self.assert_violation(CLEAN, "branch: the file says `main` but origin/HEAD is `develop`") + + def test_an_unresolvable_origin_head_is_a_note(self) -> None: + result = self.assert_clean(CLEAN) + self.assertIn("note: origin/HEAD does not resolve locally", result.stdout) + + +if __name__ == "__main__": + unittest.main() diff --git a/bootstrap/evals/tests/test_concept_sources.py b/bootstrap/evals/tests/test_concept_sources.py new file mode 100644 index 0000000..a8eb091 --- /dev/null +++ b/bootstrap/evals/tests/test_concept_sources.py @@ -0,0 +1,70 @@ +"""Tripwires that keep the agents-md references, the checker, and the dev sources aligned. + +Run from the repo root: python3 -m unittest discover -s bootstrap/evals/tests -t . +The concept map is distilled from dev skill files; these tests catch a renamed or +moved source and a slug or section that drifted between the map, the template, +and the checker. They cannot catch a source whose content changed in place. +""" + +from __future__ import annotations + +import importlib.util +import re +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +SKILL = REPO_ROOT / "bootstrap" / "skills" / "agents-md" +ROW = re.compile(r"^\| `([a-z0-9-]+)` \|.*\| `([^`]+)` \|\s*$") + + +def load_checker(): + spec = importlib.util.spec_from_file_location("check_agents_md", SKILL / "scripts" / "check-agents-md.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def concept_rows() -> dict[str, str]: + text = (SKILL / "references" / "concepts.md").read_text(encoding="utf-8") + return {m.group(1): m.group(2) for line in text.splitlines() if (m := ROW.match(line))} + + +def template_skeleton() -> str: + text = (SKILL / "references" / "template.md").read_text(encoding="utf-8") + match = re.search(r"```markdown\n(.*?)\n```\n", text, re.DOTALL) + assert match, "template.md has no ```markdown skeleton" + return match.group(1) + + +class ConceptSourcesTest(unittest.TestCase): + def test_every_checker_slug_has_a_concept_row_and_no_row_is_unknown(self) -> None: + self.assertEqual(set(concept_rows()), load_checker().SLUGS) + + def test_every_concept_source_exists_in_the_dev_plugin(self) -> None: + for slug, source in concept_rows().items(): + with self.subTest(slug=slug): + self.assertTrue(source.startswith("dev/"), source) + self.assertTrue((REPO_ROOT / source).is_file(), f"{slug}: {source} no longer exists") + + def test_every_commands_slot_maps_to_a_concept_slug(self) -> None: + checker = load_checker() + self.assertLessEqual(set(checker.SLOTS.values()), checker.SLUGS) + + +class TemplateAgreesWithCheckerTest(unittest.TestCase): + def test_the_template_sections_are_the_required_sections_in_order(self) -> None: + titles = re.findall(r"^## (.+)$", template_skeleton(), re.MULTILINE) + self.assertEqual([t for t in titles if t != "Open items"], load_checker().SECTIONS) + self.assertEqual(titles[-2:], ["Open items", "Maintaining this file"]) + + def test_the_template_has_exactly_the_checker_slots(self) -> None: + labels = re.findall(r"^- ([A-Za-z0-9 ]+?): < None: + self.assertRegex(template_skeleton(), load_checker().DEFAULT_BRANCH.pattern.replace("`([^`]+)`", "`< 0`, a rerun plugin) contradicts the de-flake rule; it is an open item under its layer, not a fact to copy. diff --git a/bootstrap/skills/agents-md/references/merge.md b/bootstrap/skills/agents-md/references/merge.md new file mode 100644 index 0000000..8477df9 --- /dev/null +++ b/bootstrap/skills/agents-md/references/merge.md @@ -0,0 +1,34 @@ +# Merge and Prune + +The existing root AGENTS.md and CLAUDE.md are input, not output. +The template's own rule lines are not classified; they are the lessons, not generic advice to prune. +Classify each bullet or paragraph before drafting, splitting it only when its sentences land in different classes, and keep the classification: the approval table is built from it, grouped by section. + +| Class | When | What happens | +| ----- | ---- | ------------ | +| keep | repo-specific, and true or plausibly true and unverifiable ("requires `MAPBOX_ACCESS_TOKEN`", "tests must not run from the root") | moves into the section it concerns, wording kept, em dashes replaced with a plain dash | +| update | repo-specific but stale against the probe (a renamed script, a moved path, a changed command) | rewritten to the current fact, with the probe evidence cited in the table | +| prune | what the filesystem already shows, generic agent advice, a duplicate, or stale with no current equivalent | dropped, with the reason in the table | +| conflict | a rule the file states that contradicts a lesson ("mock the database in unit tests", "retry flaky tests", "use a different commit format") | never overwritten silently; it becomes an interview question, because it may be a deliberate decision | + +## Conflicts + +A conflict is an instruction in an agent file; a repo fact that falls short of a lesson (an e2e suite that needs a live credential, a retry in a config file) goes through the concept map as an open item instead. + +Ask one conflict at a time, with the lesson, the existing rule, and a recommendation. +When the person keeps their rule, it stays in its section and the conflicting lesson line is reworded to agree, never both. +When they drop it, the lesson line stands and the old rule is pruned. + +## Length + +When kept gotchas push the draft past 100 lines, condense them first, without joining sentences onto one line to hit the number. +Past the 150-line cap, ask which area-specific gotchas to drop; never drop a true, user-written gotcha silently. + +## CLAUDE.md + +This skill never writes or edits CLAUDE.md; Claude Code reads AGENTS.md directly. +After the merge, CLAUDE.md content that moved into AGENTS.md is duplicated, so the closing report recommends deleting CLAUDE.md, or trimming it to what only Claude uses (such as `@` imports). + +## Re-runs + +Re-running on a file this skill wrote should produce a minimal diff: keep the existing wording of every line whose fact did not change. diff --git a/bootstrap/skills/agents-md/references/probe.md b/bootstrap/skills/agents-md/references/probe.md new file mode 100644 index 0000000..716da78 --- /dev/null +++ b/bootstrap/skills/agents-md/references/probe.md @@ -0,0 +1,59 @@ +# Probe + +Read-only. +Every fact lands in the probe table with the file and line, or the command and its output, that proves it. +A fact you could only get by guessing is unresolved, not a low-confidence guess. +In a workspace or monorepo, probe the root and every member; a large repository gets one read-only subagent per group below, each returning its rows. + +## A. Git + +- Default branch, first that answers: `git symbolic-ref --short refs/remotes/origin/HEAD`; `gh repo view --json defaultBranchRef` when `gh` is authenticated; the branch filter on the pull-request trigger in CI; the only one of `origin/main`, `origin/master`, `origin/develop` that exists. +- Remotes, and whether this is a fork (an `upstream` remote, an upstream note in the README). +- Scope vocabulary: parse `git log --format=%s -400` with `^(\w+)(\(([^)]+)\))?!?:`, split comma-separated scopes, keep those used at least three times (twice in a history under 100 commits), and map each to the folder its commits touch. + With no conventional history, take the areas commits actually touch: workspace packages, else the source subfolders that change most, never a lone package name or `src`. +- Whether history already carries `Co-Authored-By` trailers, and whether `.gitignore` covers `.dev/`. + +## B. Stack and runners + +- JS/TS: the package manager from the lockfile or `packageManager`; `scripts`; workspaces, turbo, or nx; tsconfig; the lint and format config; vitest, jest, playwright, and cypress configs. +- Python: `pyproject.toml` tool sections (pytest markers, ruff, mypy, pyright, coverage), tox, nox, uv, poetry, `conftest.py` fixtures. +- Go: `go.mod`, `//go:build` tags that split test layers, `testdata/`, golangci config. +- Rust: the Cargo workspace, `tests/` integration crates, nextest, clippy config. +- JVM and Android: Gradle source sets (`integrationTest`, `androidTest`), detekt, ktlint, spotless, Maven profiles. +- Swift and Flutter: `Package.swift`, XCUITest targets, swiftlint; `pubspec.yaml`, `integration_test/`. +- Task runners: Makefile and included `.mk` files, justfile, Taskfile, a `scripts/` folder; prefer one aggregate target that covers lint, types, and tests when it exists. +- Toolchain pins (`.tool-versions`, `.nvmrc`, `.python-version`, mise, devcontainer, nix) and pre-commit, husky, or lefthook hooks. + +## C. Merge gate + +- Read every workflow whose trigger includes pull requests, collect each job's `run:` steps, and follow them into the Makefile targets and scripts they call. +- A job that reads `secrets.*` or needs hosted infrastructure is `remote-only`; record the reason. +- Required checks come from branch protection when `gh api` can read it; otherwise treat every pull-request job as required and say so. +- Read GitLab, CircleCI, Jenkins, Azure, Bitbucket, and Buildkite config the same way. +- A workflow that only deploys on push is not a gate. + +## D. Test layers + +- Unit: file patterns, where they live, the command, and whether they must run from a package directory rather than the root. +- Integration: a separate folder, a marker, a build tag, a source set, testcontainers, a compose file for tests, CI `services:`, or a `test:integration` script. + With none of those, grep the unit suite for database or network use and note integration tests hiding there. +- E2E: the tools in the concept map, an `e2e/` folder, a `test:e2e` script, or a harness that runs the built binary as a subprocess. + Find its launch command (playwright `webServer.command`, a start or preview script, a compose file, a run skill under `.claude/skills`) and whether it drives the built artifact or a dev server or source. + A harness that drives source or a dev server is still the e2e layer; the gap becomes an `e2e` open item, not a question. +- E2E environment: stub servers (msw, nock, wiremock), recorded fixtures (VCR, polly), localstack, seed scripts, test env files, fake timers. + Flag a base URL on an https staging or production host, and any real credential the run needs. +- Browser automation: whichever tool is installed, and whether the repo ships a frontend at all. +- Retries: `retries > 0`, jest retries, a rerun plugin; configured retries become an open item under the layer they belong to, since this skill never edits config. + +## E. Quality tools + +A tool counts only when a hook, script, or CI step runs it; one that is merely available (`npm audit`) does not. +Map each configured tool onto lint and typecheck, secret scanning, dependency audit, SAST, dead code, duplication, dependency rules, coverage and complexity, flakiness, and mutation. +Only a security gap, neither a secret scan nor a dependency audit, becomes an open item; the other categories are the ship gauntlet's concern, not this file's. + +## F. Durable docs and merge input + +- `docs/architecture.md`, `docs/decisions.md`, `docs/contracts.md`, `docs/dependencies.md`, an ADR folder, `ARCHITECTURE.md`, `CONTRIBUTING.md`. +- Dependency direction between workspace packages, only when it is clear from the manifests and not obvious from the names. +- The existing root AGENTS.md and CLAUDE.md are read in SKILL.md step 2; test each behavioral claim they make (a gitignore rule, a generator's output, a CI step) against the code, not only their paths and commands. + Nested agent files, `.cursorrules`, and `.github/copilot-instructions.md` are evidence only; `CLAUDE.local.md` is private and never read. diff --git a/bootstrap/skills/agents-md/references/template.md b/bootstrap/skills/agents-md/references/template.md new file mode 100644 index 0000000..d1ce208 --- /dev/null +++ b/bootstrap/skills/agents-md/references/template.md @@ -0,0 +1,77 @@ +# AGENTS.md Template + +The skeleton below is the output shape `check-agents-md.py` enforces. +Fill every `<>` slot from the probe; the checker fails while one remains. +Rule lines outside the slots are the distilled lessons; keep their meaning, adapt their wording to the repo's vocabulary, and drop a line only when the repo makes it untrue (a repo with no tests yet still keeps the test rules, because they describe the tests to add). + +```markdown +# <> + +<> +Default branch is `<>`. + +## Commands + +- Setup: <> +- Build: <> +- Check: <> +- Unit: <> +- Integration: <> +- E2E: <> +- Merge gate: <> + +## Tests + +<> +Put each check on the cheapest layer that can fail for the right reason: unit for one rule with no I/O, integration for owned components working together against a real store, e2e for a whole journey through the built artifact. +Mock only at boundaries this repo does not own (third-party network, the clock, randomness, slow infrastructure), prefer a small fake to a stub, and never mock code this repo owns. +Inject the clock and randomness instead of patching globals. +E2E drives the built artifact against stubbed third parties and a seeded store, never production or shared staging. +Every new test is seen failing before it passes, and a bug fix starts with a test that reproduces the bug. +Assert outcomes through the public interface with literal expected values, and name tests after the capability they prove. + +## Quality + +A change is done when every merge-gate command passes locally on the final checkout. +A failing check is work to fix, including one called flaky or pre-existing; prove "pre-existing" by running the same command on the merge base. +Fix findings at their source: never suppress them inline, loosen a tool's config, or add retries, sleeps, or looser assertions. +A secret found in the tree is escalated to a person immediately, not quietly removed. + +## Architecture + +<> +<> + +## Contributions + +Work in the smallest slice that shows observable behavior, one reviewable idea per commit. +Commit subjects are `type(scope): subject` in the imperative, with a `What:` and a `Why:` body; scopes are <>. +Never add an AI co-author trailer; a person is accountable for every commit. +Plan and scratch files live under `.dev/` and are never committed. +Every pull request carries evidence from a real run, and a bug fix shows its test failing on the merge base and passing on the branch. + +## Open items + +- [ ] <>: <> + +## Maintaining this file + +Keep this file to knowledge almost every agent session in this repository needs. +Do not repeat what the code already shows; point to the authoritative file or command instead. +Prefer rewriting or pruning an entry over appending a new one, and keep each command runnable exactly as written. +``` + +## Rules for filling it + +- Every command is exact, backticked, and runnable from the repository root, or prefixed with `cd dir && ` when it must run elsewhere. +- Backtick only paths that exist; describe an example, a generated file, or a future location in plain words. + A missing path is allowed in backticks only when git ignores it (a build output) or inside an open item. +- A slot covering several packages lists each command, comma-separated; extra Commands bullets after the seven slots (a run or dev command, code generation) are allowed when they carry what the manifest does not show: a flag, a working directory, or a precondition. +- When the Merge gate is `none`, the Quality "done" line names the commands that stand in for it until one exists. +- Adapt a rule line's nouns to the repo (a static site has fixed content, not a seeded store), never its meaning. +- Omit the Open items section entirely when there are no gaps. +- Kept user sections (a knowledge base, a fork policy, an outage-causing gotcha) go between Contributions and Open items, under their own H2. + A gotcha about an existing section goes in that section instead. +- Say each thing once, and do not describe the folder layout, the dependency list, or the framework unless something about it would surprise a newcomer. +- State outcomes and failure modes, not tool mandates; naming a tool is fine when it is a fact of this repo. +- One sentence per line, no em dash, 50 to 100 lines, never over 150. diff --git a/bootstrap/skills/agents-md/scripts/check-agents-md.py b/bootstrap/skills/agents-md/scripts/check-agents-md.py new file mode 100755 index 0000000..183b043 --- /dev/null +++ b/bootstrap/skills/agents-md/scripts/check-agents-md.py @@ -0,0 +1,552 @@ +#!/usr/bin/env python3 +"""Check that a repository's root AGENTS.md is well-formed and matches the repo. + +Usage: + python3 check-agents-md.py AGENTS.md [--root DIR] [--max-lines N] + +Exit 2 when the file is missing or the arguments are bad, 1 with one violation +per line, 0 when clean. Lines starting with "note:" never change the exit code; +they name facts the script cannot settle statically (a command it does not +resolve, a default branch it cannot read), for the caller to confirm by hand. + +Violations: a required section missing, out of order, or duplicated; the file +over --max-lines; an em dash or a leftover < list[tuple[int, str]]: + for index, (start, title) in enumerate(self.h2): + if title == name: + end = self.h2[index + 1][0] if index + 1 < len(self.h2) else float("inf") + return [(n, line) for n, line in self.lines if start < n < end] + return [] + + def intro(self) -> list[tuple[int, str]]: + first = self.h2[0][0] if self.h2 else float("inf") + return [(n, line) for n, line in self.lines if n < first] + + +def git(root: Path, *args: str) -> subprocess.CompletedProcess: + return subprocess.run(["git", "-C", str(root), *args], capture_output=True, text=True) + + +def ignored(root: Path, relative: str) -> bool: + # A trailing slash lets a directory-only pattern such as `dist/` match a path that does not exist yet. + bare = relative.rstrip("/") + return any(git(root, "check-ignore", "-q", form).returncode == 0 for form in (bare, bare + "/")) + + +def exists(base: Path, raw: str) -> bool: + candidate = raw.strip().rstrip("/") + if (base / candidate).exists(): + return True + return any(ch in candidate for ch in "*?[") and bool(glob.glob(str(base / candidate), recursive=True)) + + +def tracked_basenames(root: Path) -> set[str]: + listed = git(root, "ls-files", "--cached", "--others", "--exclude-standard") + if listed.returncode == 0: + return {Path(line).name for line in listed.stdout.splitlines()} + return {path.name for path in root.rglob("*") if ".git" not in path.parts} + + +def path_candidate(root: Path, token: str) -> bool: + """True when a backticked token reads as a repository path rather than prose or a ref.""" + if any(ch.isspace() for ch in token) or "://" in token: + return False + if token.startswith(("/", "~", "@", "$", "-", ".dev/")) or token == ".dev": + return False + if any(ch in token for ch in "{}<>()=|;,'\""): + return False + last = token.rstrip("/").rsplit("/", 1)[-1] + if last.split(".", 1)[0].lower() in NAMING_CONVENTIONS: + return False + has_extension = "." in last and last.rsplit(".", 1)[-1].lower() in FILE_EXTENSIONS + if "/" not in token: + return has_extension + first = token.split("/", 1)[0] + return has_extension or token.endswith("/") or (root / first).exists() + + +def check_structure(doc: Doc, violations: list[str]) -> None: + if len(doc.h1) != 1: + violations.append(f"structure: expected exactly one '# ' title, found {len(doc.h1)}") + titles = [title for _, title in doc.h2] + for title in sorted({t for t in titles if titles.count(t) > 1}): + violations.append(f"structure: '## {title}' appears more than once") + for name in SECTIONS: + if name not in titles: + violations.append(f"structure: missing '## {name}'") + present = [t for t in dict.fromkeys(titles) if t in SECTIONS] + if present != [name for name in SECTIONS if name in present]: + violations.append(f"structure: sections out of order, expected {', '.join(SECTIONS)}") + if titles and "Maintaining this file" in titles and titles[-1] != "Maintaining this file": + violations.append("structure: '## Maintaining this file' must be the last section") + if not any(DEFAULT_BRANCH.search(line) for _, line in doc.intro()): + violations.append("structure: the intro has no 'Default branch is `x`' sentence") + + +def check_text(text: str, max_lines: int, violations: list[str]) -> None: + lines = text.splitlines() + if len(lines) > max_lines: + violations.append(f"length: {len(lines)} lines, over the {max_lines}-line cap") + for number, line in enumerate(lines, 1): + if EM_DASH in line: + violations.append(f"em-dash: line {number}") + if "< set[str]: + slugs: set[str] = set() + seen: set[tuple[str, str]] = set() + section = doc.section("Open items") + if any(title == "Open items" for _, title in doc.h2) and not any(line.strip() for _, line in section): + violations.append("open-items: '## Open items' is empty; remove it when there are none") + for number, line in section: + if not line.strip() or line.startswith((" ", "\t")): + continue + match = OPEN_ITEM.match(line) + if not match: + violations.append(f"open-items: line {number} is not '- [ ] slug: text'") + continue + slug, detail = match.group(1), match.group(2).strip() + if slug not in SLUGS: + violations.append(f"open-items: line {number} uses unknown slug '{slug}'") + if (slug, detail) in seen: + violations.append(f"open-items: line {number} repeats an earlier item") + seen.add((slug, detail)) + slugs.add(slug) + return slugs + + +def check_slots(doc: Doc, item_slugs: set[str], violations: list[str]) -> None: + found: dict[str, int] = {} + for number, line in doc.section("Commands"): + match = SLOT_LINE.match(line) + if not match or match.group(1) not in SLOTS: + continue + label, value = match.group(1), match.group(2).strip() + found[label] = found.get(label, 0) + 1 + if value.startswith("`"): + continue + if re.match(r"^n/a \(.+\)", value): + continue + if re.match(r"^none\b", value): + if SLOTS[label] not in item_slugs: + violations.append(f"commands: '{label}: none' (line {number}) has no '- [ ] {SLOTS[label]}:' open item") + continue + violations.append(f"commands: '{label}' (line {number}) must be a backticked command, 'none', or 'n/a (reason)'") + for label in SLOTS: + if found.get(label, 0) == 0: + violations.append(f"commands: missing the '- {label}:' slot") + elif found[label] > 1: + violations.append(f"commands: the '{label}' slot appears {found[label]} times") + + +def check_paths(root: Path, doc: Doc, violations: list[str]) -> None: + skipped = {n for n, _ in doc.section("Open items")} + basenames: set[str] | None = None + for number, line in doc.lines: + if number in skipped: + continue + for raw in BACKTICK.findall(line): + token = raw.strip() + if not path_candidate(root, token) or ignored(root, token): + continue + if "/" in token: + if not exists(root, token): + violations.append(f"stale: `{token}` (line {number}) does not exist under {root}") + else: + if basenames is None: + basenames = tracked_basenames(root) + if not exists(root, token) and not fnmatch.filter(basenames, token): + violations.append(f"stale: `{token}` (line {number}) matches no file in the repository") + for target in LINK.findall(line): + if target.startswith(("http://", "https://", "mailto:", "#")): + continue + relative = target.split("#", 1)[0] + if relative and not exists(root, relative) and not ignored(root, relative): + violations.append(f"stale: link '{target}' (line {number}) does not exist under {root}") + + +def package_scripts(directory: Path) -> dict[str, str] | None: + manifest = directory / "package.json" + if not manifest.exists(): + return None + try: + return json.loads(manifest.read_text(encoding="utf-8")).get("scripts", {}) or {} + except (json.JSONDecodeError, AttributeError): + return None + + +def make_targets(directory: Path) -> tuple[set[str], bool] | None: + files = [directory / name for name in ("GNUmakefile", "makefile", "Makefile") if (directory / name).exists()] + if not files: + return None + files += sorted(directory.glob("*.mk")) + sorted(directory.glob("*/*.mk")) + targets: set[str] = set() + catch_all = False + for path in files: + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + match = re.match(r"^([^\s:#=][^:#=]*?)\s*::?(?!=)", line) + if not match: + continue + for name in match.group(1).split(): + if "%" in name: + catch_all = True + targets.add(name) + return targets, catch_all + + +def just_recipes(directory: Path) -> set[str] | None: + for name in ("justfile", "Justfile", ".justfile"): + path = directory / name + if path.exists(): + text = path.read_text(encoding="utf-8", errors="replace") + return {m.group(1) for m in re.finditer(r"^@?([A-Za-z0-9_-]+)[^:\n=]*:(?!=)", text, re.MULTILINE)} + return None + + +def task_names(directory: Path) -> set[str] | None: + for name in ("Taskfile.yml", "Taskfile.yaml", "taskfile.yml", "taskfile.yaml"): + path = directory / name + if not path.exists(): + continue + names: set[str] = set() + inside = False + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if re.match(r"^tasks:\s*$", line): + inside = True + elif re.match(r"^\S", line): + inside = False + elif inside: + match = re.match(r"^ ([A-Za-z0-9_.:-]+):", line) + if match: + names.add(match.group(1)) + return names + return None + + +def segments(command: str) -> list[list[str]]: + parts = re.split(r"\s*(?:&&|\|\||;)\s*", command) + result = [] + for part in parts: + part = part.split("|", 1)[0].strip() + try: + words = shlex.split(part) + except ValueError: + words = part.split() + while words and re.match(r"^[A-Z_][A-Z0-9_]*=", words[0]): + words = words[1:] + if words: + result.append(words) + return result + + +def declared_dependencies(*directories: Path) -> set[str]: + names: set[str] = set() + for directory in directories: + manifest = directory / "package.json" + if not manifest.exists(): + continue + try: + data = json.loads(manifest.read_text(encoding="utf-8")) + except json.JSONDecodeError: + continue + for key in ("dependencies", "devDependencies", "optionalDependencies"): + names.update((data.get(key) or {}).keys()) + return names + + +def path_arguments_exist(args: list[str], cwd: Path, rel: str) -> str: + """Test runners take path prefixes as filters; a slashed argument must prefix something real.""" + for arg in args: + if "/" in arg and not arg.startswith(("-", "/")) and "*" not in arg: + if not glob.glob(glob.escape(str(cwd / arg)) + "*"): + return f"`{arg}` matches nothing in {rel}" + return "" + + +def resolve_package_script(runner: str, words: list[str], cwd: Path, rel: str, root: Path) -> str | None: + """Return a violation, "" when resolved, or None when not statically decidable.""" + args = words[1:] + if not args: + return "" + first = args[0] + if first.startswith("-") or first in {"workspace", "workspaces"}: + return None + if first in {"run", "run-script"}: + if len(args) < 2: + return "" + name = args[1] + if name.startswith("-"): + return None + elif runner != "bun" and first in RUNNER_SCRIPTS: + name = RUNNER_SCRIPTS[first] + elif first == "test": + return path_arguments_exist(args[1:], cwd, rel) + elif first in PM_BUILTINS[runner]: + return "" + elif runner == "npm": + return None + else: + name = first + if runner == "bun" and (cwd / name).is_file(): + return "" + manifest = "package.json" if rel == "." else f"{rel}/package.json" + scripts = package_scripts(cwd) + if scripts is not None and name in scripts: + return "" + # bun and yarn (and an implicit pnpm call) also run a dependency's binary, e.g. `bun turbo`. + runs_binaries = runner in {"bun", "yarn"} or (runner == "pnpm" and first not in {"run", "run-script"}) + if runs_binaries and ((cwd / "node_modules" / ".bin" / name).exists() or name in declared_dependencies(cwd, root)): + return None + if scripts is None: + return f"no readable {manifest} for `{name}`" + return f"`{name}` is not a script in {manifest}" + + +def resolve(words: list[str], cwd: Path, root: Path) -> str | None: + runner = words[0] + rel = str(cwd.relative_to(root)) if cwd != root else "." + if runner in PM_BUILTINS: + return resolve_package_script(runner, words, cwd, rel, root) + if runner == "make": + args = words[1:] + if "-C" in args: + index = args.index("-C") + if index + 1 < len(args): + cwd = cwd / args[index + 1] + rel = str(cwd.relative_to(root)) + args = args[:index] + args[index + 2:] + wanted = [a for a in args if not a.startswith("-") and "=" not in a] + found = make_targets(cwd) + if found is None: + return f"no Makefile in {rel}" + targets, catch_all = found + missing = [t for t in wanted if t not in targets] + if missing and catch_all: + return None + return f"make target(s) {', '.join(missing)} not defined in {rel}" if missing else "" + if runner == "just": + recipes = just_recipes(cwd) + if recipes is None: + return f"no justfile in {rel}" + wanted = [a for a in words[1:2] if not a.startswith("-")] + return f"just recipe `{wanted[0]}` not defined in {rel}" if wanted and wanted[0] not in recipes else "" + if runner == "task": + names = task_names(cwd) + if names is None: + return f"no Taskfile in {rel}" + wanted = [a for a in words[1:2] if not a.startswith("-")] + return f"task `{wanted[0]}` not defined in {rel}" if wanted and wanted[0] not in names else "" + if runner in FILE_RUNNERS: + args = words[1:] + if not args or args[0] in {"-m", "-c", "-e", "--eval"}: + return "" + target = next((a for a in args if not a.startswith("-")), None) + if target is None: + return "" + return "" if (cwd / target).exists() else f"`{target}` does not exist in {rel}" + if runner.startswith("./"): + return "" if (cwd / runner).exists() else f"`{runner}` does not exist in {rel}" + if runner in NOTE_RUNNERS: + return None + return "" + + +def check_commands(root: Path, doc: Doc, violations: list[str], notes: list[str]) -> int: + skipped = {n for n, _ in doc.section("Open items")} + in_commands = {n for n, _ in doc.section("Commands")} + resolved = 0 + known = set(PM_BUILTINS) | FILE_RUNNERS | NOTE_RUNNERS | {"make", "just", "task", "cd"} + for number, line in doc.lines: + if number in skipped: + continue + for raw in BACKTICK.findall(line): + command = raw.strip() + parts = segments(command) + if not parts: + continue + if not (parts[0][0] in known or parts[0][0].startswith("./")): + if number in in_commands and " " in command: + notes.append(f"note: `{command}` (line {number}) uses a runner this script does not know; confirm it runs") + continue + cwd = root + for words in parts: + if words[0] == "cd": + target = words[1] if len(words) > 1 else "." + if not (cwd / target).is_dir(): + violations.append(f"command: `{command}` (line {number}) cds into missing `{target}`") + break + cwd = (cwd / target).resolve() + continue + outcome = resolve(words, cwd, root) + if outcome is None: + notes.append(f"note: `{command}` (line {number}) is not resolved statically; confirm it runs") + elif outcome: + violations.append(f"command: `{command}` (line {number}): {outcome}") + else: + resolved += 1 + return resolved + + +def check_branch(root: Path, doc: Doc, violations: list[str], notes: list[str]) -> None: + stated = next((m.group(1) for _, line in doc.intro() if (m := DEFAULT_BRANCH.search(line))), None) + if stated is None: + return + head = git(root, "symbolic-ref", "--short", "refs/remotes/origin/HEAD") + if head.returncode != 0: + notes.append(f"note: origin/HEAD does not resolve locally; confirm the default branch `{stated}` another way") + return + actual = head.stdout.strip().split("/", 1)[-1] + if actual != stated: + violations.append(f"branch: the file says `{stated}` but origin/HEAD is `{actual}`") + + +def check_duplicates(doc: Doc, violations: list[str]) -> None: + seen: dict[str, int] = {} + for number, line in doc.lines: + if HEADING.match(line): + continue + normal = " ".join(re.sub(r"^\s*(?:[-*]|\d+\.)\s+", "", line).lower().split()) + if len(normal) < 12 or set(normal) <= set("|-: "): + continue + if normal in seen: + violations.append(f"duplicate: line {number} repeats line {seen[normal]}") + else: + seen[normal] = number + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("agents_md") + parser.add_argument("--root", default=".") + parser.add_argument("--max-lines", type=int, default=150) + try: + args = parser.parse_args() + except SystemExit: + return 2 + root = Path(args.root).resolve() + path = Path(args.agents_md) + if not path.is_file(): + print(f"missing: {path}") + return 2 + if not root.is_dir(): + print(f"missing: --root {root} is not a directory") + return 2 + + text = path.read_text(encoding="utf-8", errors="replace") + doc = Doc(text) + violations: list[str] = [] + notes: list[str] = [] + check_structure(doc, violations) + check_text(text, args.max_lines, violations) + item_slugs = open_items(doc, violations) + check_slots(doc, item_slugs, violations) + check_paths(root, doc, violations) + resolved = check_commands(root, doc, violations, notes) + check_branch(root, doc, violations, notes) + check_duplicates(doc, violations) + + for note in dict.fromkeys(notes): + print(note) + if violations: + print("\n".join(violations)) + return 1 + print(f"check-agents-md: ok ({len(text.splitlines())} lines, {resolved} commands resolved, {len(set(notes))} notes)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/bootstrap/.codex-plugin/plugin.json b/plugins/bootstrap/.codex-plugin/plugin.json new file mode 100644 index 0000000..d520605 --- /dev/null +++ b/plugins/bootstrap/.codex-plugin/plugin.json @@ -0,0 +1,24 @@ +{ + "name": "bootstrap", + "version": "0.1.0", + "description": "Prepare a repository for agent work: probe its stack, map it onto the dev workflow's SDLC concepts (test layers, mocking at boundaries, CI parity, fix-at-source quality, commit and PR conventions), and write or refresh its root AGENTS.md with the repo's exact commands and its gaps as open items. Use on any repository, whether or not the dev plugin is installed.", + "author": { + "name": "Tobrun" + }, + "repository": "https://github.com/tobrun/workflow", + "skills": "./skills/", + "interface": { + "displayName": "Bootstrap", + "shortDescription": "Write a repository's AGENTS.md from a probe of its stack.", + "longDescription": "Prepare a repository for agent work: probe its stack, map it onto the dev workflow's SDLC concepts (test layers, mocking at boundaries, CI parity, fix-at-source quality, commit and PR conventions), and write or refresh its root AGENTS.md with the repo's exact commands and its gaps as open items. Use on any repository, whether or not the dev plugin is installed.", + "developerName": "Tobrun", + "category": "Developer Tools", + "capabilities": [ + "Interactive", + "Write" + ], + "defaultPrompt": [ + "Write or refresh this repository's AGENTS.md." + ] + } +} diff --git a/plugins/bootstrap/README.md b/plugins/bootstrap/README.md new file mode 100644 index 0000000..e9fe289 --- /dev/null +++ b/plugins/bootstrap/README.md @@ -0,0 +1,3 @@ +# bootstrap for Codex + +Generated from `bootstrap/` by `scripts/build_codex_plugin.py`. Do not edit this directory directly. diff --git a/plugins/bootstrap/skills/agents-md/SKILL.md b/plugins/bootstrap/skills/agents-md/SKILL.md new file mode 100644 index 0000000..698cd0a --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/SKILL.md @@ -0,0 +1,75 @@ +--- +name: agents-md +description: Probe a repository's stack (manifests, CI, test layout, git history), map it onto the dev workflow's SDLC concepts (test layers, mocking at boundaries, CI parity, fix-at-source quality, commit and PR conventions), and write or refresh its root AGENTS.md with the repo's exact commands, merging an existing AGENTS.md or CLAUDE.md and recording gaps as open items. Use when preparing a repository for agent work, or when its AGENTS.md has drifted from the code. +--- + +# AGENTS.md + +The deliverable is one file: the repository's root `AGENTS.md`. +It encodes the lessons of the dev workflow as short outcome rules any agent can follow without that workflow installed, filled in with this repo's real commands, layout, and gaps. + +Limits: +- Write only the root `AGENTS.md`; never write CLAUDE.md, CLAUDE.local.md, nested agent files, docs, tests, CI, configs, or `.gitignore`. +- A missing concept (no integration layer, no mocked e2e environment, no merge gate) becomes an open item, never a scaffold. +- Never commit, and never invoke another skill; recommend the next step instead. + +`{agents-md-skill-root}` below is the directory holding this SKILL.md, and `{project-slug}` is the basename of the repository root. + +## 1. Probe + +Follow [references/probe.md](references/probe.md), read-only. +In a workspace or large repository, launch one background read-only subagent per probe group and merge their rows. +Collect every fact into one table shown to the user: `concept | value | evidence | confidence`, where evidence is a file and line, or a command and its output. + +## 2. Map + +Resolve every slug in [references/concepts.md](references/concepts.md) to **present**, **absent**, or **unresolved**, using its "Reading the probe onto a row" rules. +Absent rows take their "When absent" behavior without a question. +Read the existing root AGENTS.md and CLAUDE.md now and classify each line per [references/merge.md](references/merge.md); every **conflict** joins the unresolved list. + +## 3. Interview + +Ask only about unresolved rows and conflicts, one question at a time, using the host's structured user-input tool when available. +Each question carries a recommendation and `Confidence: NN%` with the one fact that would most change it. +At `Confidence: 75%` or above, apply the recommendation instead of asking, and list every auto-applied item once at the end of the interview so one reply can overturn any of them. +When several remain, ask in this order: the merge gate, the e2e launch command and environment, conflicts with the lessons, then the default branch. +Close with one skippable question: what do newcomers, human or agent, get wrong here that the code does not show? +Its answer is the most valuable content in the file; place each gotcha in the section it concerns. +Never guess a command: ask, or record `none` with an open item. + +## 4. Verify the commands + +Run each command bound for the Commands section once, in the background with a timeout, while drafting. +Ask before any install, local or global (the Setup command included), and before anything that deploys, publishes, needs credentials, or would run for more than about five minutes. +Keep runners from installing on their own when dependencies are missing (for example `bun --no-install`). +A command whose dependencies are not installed, or that was declined, cannot be run: it stays in the file with its probe evidence, which the approval table shows. +Only a command that ran and failed gets an open item quoting its first error line. + +## 5. Draft and merge + +Create `/tmp/{project-slug}/bootstrap/` and draft `AGENTS.md` there from [references/template.md](references/template.md), folding in the kept and updated lines from step 2. +Keep the rule lines' meaning intact and word them in the repo's vocabulary; the file is read by every future session, so every line must earn its place. +A refresh, where the existing AGENTS.md already passes the checker's structure, starts the draft from that file instead of the template, changes only lines whose facts changed, and repeats the install and newcomer questions only when a Commands slot or a gotcha is in play. + +## 6. Check loop + +Loop until it exits 0: + +```bash +python3 {agents-md-skill-root}/scripts/check-agents-md.py /tmp/{project-slug}/bootstrap/AGENTS.md --root . +``` + +Fix the draft, never the checker or the repository. +Resolve each `note:` line by running the command; a command that cannot be run (step 4) keeps its probe evidence instead. + +## 7. Approve and write + +Show the diff, `git diff --no-index` of the current AGENTS.md (or `/dev/null`) against the draft, then one table of updated and pruned lines with their evidence (kept lines as a count per section), then the open items. +Wait for approval; apply requested edits and re-run step 6. +Copy the draft to the repository root and run the checker once more against `AGENTS.md`. + +Close with: +- The open items, each with what closing it takes. +- Defects noticed while probing that fit no section or slug, marked unverified. +- When a CLAUDE.md exists, a recommendation to delete it, or trim it to Claude-only content such as `@` imports, since its content now lives in AGENTS.md. +- The next step: review and commit the file (`/dev:commit` if the dev plugin is installed); nothing is committed here. diff --git a/plugins/bootstrap/skills/agents-md/agents/openai.yaml b/plugins/bootstrap/skills/agents-md/agents/openai.yaml new file mode 100644 index 0000000..04ca221 --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "AGENTS.md" + short_description: "Probe the stack and write a root AGENTS.md" + default_prompt: "Use $bootstrap:agents-md to probe this repository and write or refresh its root AGENTS.md." +policy: + allow_implicit_invocation: false diff --git a/plugins/bootstrap/skills/agents-md/references/concepts.md b/plugins/bootstrap/skills/agents-md/references/concepts.md new file mode 100644 index 0000000..1151d38 --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/references/concepts.md @@ -0,0 +1,38 @@ +# Concept Map + +Each row is one SDLC concept the dev workflow depends on, mapped from what the probe finds to what AGENTS.md says. +The slug column is a closed vocabulary: open items use exactly these slugs, and `check-agents-md.py` rejects any other. +The source column names the file in the workflow repository the lesson was distilled from; it is provenance for maintainers, not something to read during a run. + +| Slug | Look for | When present, AGENTS.md says | When absent | Source | +| ---- | -------- | ---------------------------- | ----------- | ------ | +| `setup` | lockfile, package manager, toolchain pins | Setup slot: the exact install command, plus any toolchain pin a fresh checkout needs | ask; if still unknown, `none` and an open item | `dev/skills/scope/SKILL.md` | +| `build` | build script or target, artifact location | Build slot: the command, and where the artifact lands if the e2e layer drives it | `n/a (reason)` for source-only libraries and scripts | `dev/skills/scope/SKILL.md` | +| `check` | lint, typecheck, format, an aggregate target | Check slot: the aggregate target first, else the individual commands | `none` and an open item | `dev/skills/ship/references/tools.md` | +| `unit` | test file patterns, runner config | Unit slot: the command, and in Tests where unit tests live and whether they must run from a package directory | `none` and an open item | `dev/skills/build/references/layers.md` | +| `integration` | separate folder, markers, build tags, source sets, testcontainers, CI `services:` | Integration slot: the command, and in Tests the folder and any service it needs | `none` and an open item: no layer runs owned components together against a real store | `dev/skills/build/references/layers.md` | +| `e2e` | playwright, cypress, detox, maestro, XCUITest, espresso, robot, a harness running the built binary | E2E slot: one command that builds, launches, and drives the artifact | `none` and an open item | `dev/skills/build/references/layers.md` | +| `e2e-env` | stub servers, recorded fixtures, seed scripts, fake clocks, test env files | one Tests line naming how third parties are stubbed, the store is seeded, and the clock is fixed | an open item naming what reaches the real world (a live token, a staging URL, an unseeded store); with no e2e layer at all, only when the built artifact itself calls a real third party | `dev/skills/build/references/mocking.md` | +| `browser` | an installed browser automation tool | the tool, named as a fact of this repo | an open item only when the repo ships a frontend | `dev/skills/build/SKILL.md` | +| `ci` | pull-request workflows and the scripts they call | Merge gate slot: the project-owned commands, then `remote-only: job (reason)` for jobs that need secrets or hosted infrastructure | `none` and an open item: nothing gates pull requests | `dev/references/ci-parity.md` | +| `security` | secret scanning, dependency audit, SAST in hooks or CI | nothing extra; the Check or Merge gate slot already runs it | one open item when there is neither a secret scan nor a dependency audit | `dev/skills/ship/references/gauntlet.md` | +| `architecture` | `docs/architecture.md`, `ARCHITECTURE.md`, a workspace dependency graph | Architecture: a pointer to the overview, or one to three lines of dependency direction that the code does not make obvious | one sentence saying there is no written overview | `dev/references/architecture.md` | +| `decisions` | `docs/decisions.md`, an ADR folder | Architecture: where settled decisions live, and to read them after forming a view, not before | covered by the absence sentence | `dev/references/decision-ledger.md` | +| `contracts` | `docs/contracts.md` | Architecture: where cross-boundary guarantees live, and that a change must keep them or say which it breaks | covered by the absence sentence | `dev/references/contracts.md` | +| `dependencies` | `docs/dependencies.md`, dependency-cruiser, import-linter, ArchUnit config | Architecture: where the allowed module edges are declared and what enforces them | covered by the absence sentence | `dev/references/dependency-rules.md` | +| `commits` | conventional subjects in `git log`, existing trailers | Contributions: the commit format and the real scope list | the format is adopted from now on, and scopes are the areas commits touch (see the probe) | `dev/skills/commit/references/message-format.md` | +| `plans` | whether `.gitignore` covers `.dev/` | Contributions: plan and scratch files live under `.dev/` and are never committed | an open item: `.dev/` is not gitignored | `dev/references/plan-layout.md` | +| `pr` | PR templates, a GitHub remote | Contributions: every PR carries evidence from a real run | always stated | `dev/skills/ship/references/pull-request.md` | +| `branch` | `origin/HEAD`, CI branch filters | intro: "Default branch is `x`." | ask | `dev/references/plan-layout.md` | + +## Reading the probe onto a row + +A concept is **present** when the probe cites a file or command that proves it, **absent** when the probe looked where the row says and found nothing, and **unresolved** when the evidence conflicts or the only proof would be running something unsafe. +Unresolved rows go to the interview; absent rows take their "When absent" behavior without asking. + +Watch for evidence that looks present but is not: +- An e2e suite whose base URL is an https staging or production host, or that needs a real credential, is `e2e` present and `e2e-env` absent. +- Integration tests hiding inside the unit suite (a unit test that opens a database or a socket) are not an integration layer; say so in Tests. +- A workflow that only deploys on push is not a merge gate. +- An e2e harness that drives source or a dev server instead of the built artifact is `e2e` present with an `e2e` open item. +- A configured retry (`retries > 0`, a rerun plugin) contradicts the de-flake rule; it is an open item under its layer, not a fact to copy. diff --git a/plugins/bootstrap/skills/agents-md/references/merge.md b/plugins/bootstrap/skills/agents-md/references/merge.md new file mode 100644 index 0000000..8477df9 --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/references/merge.md @@ -0,0 +1,34 @@ +# Merge and Prune + +The existing root AGENTS.md and CLAUDE.md are input, not output. +The template's own rule lines are not classified; they are the lessons, not generic advice to prune. +Classify each bullet or paragraph before drafting, splitting it only when its sentences land in different classes, and keep the classification: the approval table is built from it, grouped by section. + +| Class | When | What happens | +| ----- | ---- | ------------ | +| keep | repo-specific, and true or plausibly true and unverifiable ("requires `MAPBOX_ACCESS_TOKEN`", "tests must not run from the root") | moves into the section it concerns, wording kept, em dashes replaced with a plain dash | +| update | repo-specific but stale against the probe (a renamed script, a moved path, a changed command) | rewritten to the current fact, with the probe evidence cited in the table | +| prune | what the filesystem already shows, generic agent advice, a duplicate, or stale with no current equivalent | dropped, with the reason in the table | +| conflict | a rule the file states that contradicts a lesson ("mock the database in unit tests", "retry flaky tests", "use a different commit format") | never overwritten silently; it becomes an interview question, because it may be a deliberate decision | + +## Conflicts + +A conflict is an instruction in an agent file; a repo fact that falls short of a lesson (an e2e suite that needs a live credential, a retry in a config file) goes through the concept map as an open item instead. + +Ask one conflict at a time, with the lesson, the existing rule, and a recommendation. +When the person keeps their rule, it stays in its section and the conflicting lesson line is reworded to agree, never both. +When they drop it, the lesson line stands and the old rule is pruned. + +## Length + +When kept gotchas push the draft past 100 lines, condense them first, without joining sentences onto one line to hit the number. +Past the 150-line cap, ask which area-specific gotchas to drop; never drop a true, user-written gotcha silently. + +## CLAUDE.md + +This skill never writes or edits CLAUDE.md; Claude Code reads AGENTS.md directly. +After the merge, CLAUDE.md content that moved into AGENTS.md is duplicated, so the closing report recommends deleting CLAUDE.md, or trimming it to what only Claude uses (such as `@` imports). + +## Re-runs + +Re-running on a file this skill wrote should produce a minimal diff: keep the existing wording of every line whose fact did not change. diff --git a/plugins/bootstrap/skills/agents-md/references/probe.md b/plugins/bootstrap/skills/agents-md/references/probe.md new file mode 100644 index 0000000..716da78 --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/references/probe.md @@ -0,0 +1,59 @@ +# Probe + +Read-only. +Every fact lands in the probe table with the file and line, or the command and its output, that proves it. +A fact you could only get by guessing is unresolved, not a low-confidence guess. +In a workspace or monorepo, probe the root and every member; a large repository gets one read-only subagent per group below, each returning its rows. + +## A. Git + +- Default branch, first that answers: `git symbolic-ref --short refs/remotes/origin/HEAD`; `gh repo view --json defaultBranchRef` when `gh` is authenticated; the branch filter on the pull-request trigger in CI; the only one of `origin/main`, `origin/master`, `origin/develop` that exists. +- Remotes, and whether this is a fork (an `upstream` remote, an upstream note in the README). +- Scope vocabulary: parse `git log --format=%s -400` with `^(\w+)(\(([^)]+)\))?!?:`, split comma-separated scopes, keep those used at least three times (twice in a history under 100 commits), and map each to the folder its commits touch. + With no conventional history, take the areas commits actually touch: workspace packages, else the source subfolders that change most, never a lone package name or `src`. +- Whether history already carries `Co-Authored-By` trailers, and whether `.gitignore` covers `.dev/`. + +## B. Stack and runners + +- JS/TS: the package manager from the lockfile or `packageManager`; `scripts`; workspaces, turbo, or nx; tsconfig; the lint and format config; vitest, jest, playwright, and cypress configs. +- Python: `pyproject.toml` tool sections (pytest markers, ruff, mypy, pyright, coverage), tox, nox, uv, poetry, `conftest.py` fixtures. +- Go: `go.mod`, `//go:build` tags that split test layers, `testdata/`, golangci config. +- Rust: the Cargo workspace, `tests/` integration crates, nextest, clippy config. +- JVM and Android: Gradle source sets (`integrationTest`, `androidTest`), detekt, ktlint, spotless, Maven profiles. +- Swift and Flutter: `Package.swift`, XCUITest targets, swiftlint; `pubspec.yaml`, `integration_test/`. +- Task runners: Makefile and included `.mk` files, justfile, Taskfile, a `scripts/` folder; prefer one aggregate target that covers lint, types, and tests when it exists. +- Toolchain pins (`.tool-versions`, `.nvmrc`, `.python-version`, mise, devcontainer, nix) and pre-commit, husky, or lefthook hooks. + +## C. Merge gate + +- Read every workflow whose trigger includes pull requests, collect each job's `run:` steps, and follow them into the Makefile targets and scripts they call. +- A job that reads `secrets.*` or needs hosted infrastructure is `remote-only`; record the reason. +- Required checks come from branch protection when `gh api` can read it; otherwise treat every pull-request job as required and say so. +- Read GitLab, CircleCI, Jenkins, Azure, Bitbucket, and Buildkite config the same way. +- A workflow that only deploys on push is not a gate. + +## D. Test layers + +- Unit: file patterns, where they live, the command, and whether they must run from a package directory rather than the root. +- Integration: a separate folder, a marker, a build tag, a source set, testcontainers, a compose file for tests, CI `services:`, or a `test:integration` script. + With none of those, grep the unit suite for database or network use and note integration tests hiding there. +- E2E: the tools in the concept map, an `e2e/` folder, a `test:e2e` script, or a harness that runs the built binary as a subprocess. + Find its launch command (playwright `webServer.command`, a start or preview script, a compose file, a run skill under `.claude/skills`) and whether it drives the built artifact or a dev server or source. + A harness that drives source or a dev server is still the e2e layer; the gap becomes an `e2e` open item, not a question. +- E2E environment: stub servers (msw, nock, wiremock), recorded fixtures (VCR, polly), localstack, seed scripts, test env files, fake timers. + Flag a base URL on an https staging or production host, and any real credential the run needs. +- Browser automation: whichever tool is installed, and whether the repo ships a frontend at all. +- Retries: `retries > 0`, jest retries, a rerun plugin; configured retries become an open item under the layer they belong to, since this skill never edits config. + +## E. Quality tools + +A tool counts only when a hook, script, or CI step runs it; one that is merely available (`npm audit`) does not. +Map each configured tool onto lint and typecheck, secret scanning, dependency audit, SAST, dead code, duplication, dependency rules, coverage and complexity, flakiness, and mutation. +Only a security gap, neither a secret scan nor a dependency audit, becomes an open item; the other categories are the ship gauntlet's concern, not this file's. + +## F. Durable docs and merge input + +- `docs/architecture.md`, `docs/decisions.md`, `docs/contracts.md`, `docs/dependencies.md`, an ADR folder, `ARCHITECTURE.md`, `CONTRIBUTING.md`. +- Dependency direction between workspace packages, only when it is clear from the manifests and not obvious from the names. +- The existing root AGENTS.md and CLAUDE.md are read in SKILL.md step 2; test each behavioral claim they make (a gitignore rule, a generator's output, a CI step) against the code, not only their paths and commands. + Nested agent files, `.cursorrules`, and `.github/copilot-instructions.md` are evidence only; `CLAUDE.local.md` is private and never read. diff --git a/plugins/bootstrap/skills/agents-md/references/template.md b/plugins/bootstrap/skills/agents-md/references/template.md new file mode 100644 index 0000000..d1ce208 --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/references/template.md @@ -0,0 +1,77 @@ +# AGENTS.md Template + +The skeleton below is the output shape `check-agents-md.py` enforces. +Fill every `<>` slot from the probe; the checker fails while one remains. +Rule lines outside the slots are the distilled lessons; keep their meaning, adapt their wording to the repo's vocabulary, and drop a line only when the repo makes it untrue (a repo with no tests yet still keeps the test rules, because they describe the tests to add). + +```markdown +# <> + +<> +Default branch is `<>`. + +## Commands + +- Setup: <> +- Build: <> +- Check: <> +- Unit: <> +- Integration: <> +- E2E: <> +- Merge gate: <> + +## Tests + +<> +Put each check on the cheapest layer that can fail for the right reason: unit for one rule with no I/O, integration for owned components working together against a real store, e2e for a whole journey through the built artifact. +Mock only at boundaries this repo does not own (third-party network, the clock, randomness, slow infrastructure), prefer a small fake to a stub, and never mock code this repo owns. +Inject the clock and randomness instead of patching globals. +E2E drives the built artifact against stubbed third parties and a seeded store, never production or shared staging. +Every new test is seen failing before it passes, and a bug fix starts with a test that reproduces the bug. +Assert outcomes through the public interface with literal expected values, and name tests after the capability they prove. + +## Quality + +A change is done when every merge-gate command passes locally on the final checkout. +A failing check is work to fix, including one called flaky or pre-existing; prove "pre-existing" by running the same command on the merge base. +Fix findings at their source: never suppress them inline, loosen a tool's config, or add retries, sleeps, or looser assertions. +A secret found in the tree is escalated to a person immediately, not quietly removed. + +## Architecture + +<> +<> + +## Contributions + +Work in the smallest slice that shows observable behavior, one reviewable idea per commit. +Commit subjects are `type(scope): subject` in the imperative, with a `What:` and a `Why:` body; scopes are <>. +Never add an AI co-author trailer; a person is accountable for every commit. +Plan and scratch files live under `.dev/` and are never committed. +Every pull request carries evidence from a real run, and a bug fix shows its test failing on the merge base and passing on the branch. + +## Open items + +- [ ] <>: <> + +## Maintaining this file + +Keep this file to knowledge almost every agent session in this repository needs. +Do not repeat what the code already shows; point to the authoritative file or command instead. +Prefer rewriting or pruning an entry over appending a new one, and keep each command runnable exactly as written. +``` + +## Rules for filling it + +- Every command is exact, backticked, and runnable from the repository root, or prefixed with `cd dir && ` when it must run elsewhere. +- Backtick only paths that exist; describe an example, a generated file, or a future location in plain words. + A missing path is allowed in backticks only when git ignores it (a build output) or inside an open item. +- A slot covering several packages lists each command, comma-separated; extra Commands bullets after the seven slots (a run or dev command, code generation) are allowed when they carry what the manifest does not show: a flag, a working directory, or a precondition. +- When the Merge gate is `none`, the Quality "done" line names the commands that stand in for it until one exists. +- Adapt a rule line's nouns to the repo (a static site has fixed content, not a seeded store), never its meaning. +- Omit the Open items section entirely when there are no gaps. +- Kept user sections (a knowledge base, a fork policy, an outage-causing gotcha) go between Contributions and Open items, under their own H2. + A gotcha about an existing section goes in that section instead. +- Say each thing once, and do not describe the folder layout, the dependency list, or the framework unless something about it would surprise a newcomer. +- State outcomes and failure modes, not tool mandates; naming a tool is fine when it is a fact of this repo. +- One sentence per line, no em dash, 50 to 100 lines, never over 150. diff --git a/plugins/bootstrap/skills/agents-md/scripts/check-agents-md.py b/plugins/bootstrap/skills/agents-md/scripts/check-agents-md.py new file mode 100755 index 0000000..183b043 --- /dev/null +++ b/plugins/bootstrap/skills/agents-md/scripts/check-agents-md.py @@ -0,0 +1,552 @@ +#!/usr/bin/env python3 +"""Check that a repository's root AGENTS.md is well-formed and matches the repo. + +Usage: + python3 check-agents-md.py AGENTS.md [--root DIR] [--max-lines N] + +Exit 2 when the file is missing or the arguments are bad, 1 with one violation +per line, 0 when clean. Lines starting with "note:" never change the exit code; +they name facts the script cannot settle statically (a command it does not +resolve, a default branch it cannot read), for the caller to confirm by hand. + +Violations: a required section missing, out of order, or duplicated; the file +over --max-lines; an em dash or a leftover < list[tuple[int, str]]: + for index, (start, title) in enumerate(self.h2): + if title == name: + end = self.h2[index + 1][0] if index + 1 < len(self.h2) else float("inf") + return [(n, line) for n, line in self.lines if start < n < end] + return [] + + def intro(self) -> list[tuple[int, str]]: + first = self.h2[0][0] if self.h2 else float("inf") + return [(n, line) for n, line in self.lines if n < first] + + +def git(root: Path, *args: str) -> subprocess.CompletedProcess: + return subprocess.run(["git", "-C", str(root), *args], capture_output=True, text=True) + + +def ignored(root: Path, relative: str) -> bool: + # A trailing slash lets a directory-only pattern such as `dist/` match a path that does not exist yet. + bare = relative.rstrip("/") + return any(git(root, "check-ignore", "-q", form).returncode == 0 for form in (bare, bare + "/")) + + +def exists(base: Path, raw: str) -> bool: + candidate = raw.strip().rstrip("/") + if (base / candidate).exists(): + return True + return any(ch in candidate for ch in "*?[") and bool(glob.glob(str(base / candidate), recursive=True)) + + +def tracked_basenames(root: Path) -> set[str]: + listed = git(root, "ls-files", "--cached", "--others", "--exclude-standard") + if listed.returncode == 0: + return {Path(line).name for line in listed.stdout.splitlines()} + return {path.name for path in root.rglob("*") if ".git" not in path.parts} + + +def path_candidate(root: Path, token: str) -> bool: + """True when a backticked token reads as a repository path rather than prose or a ref.""" + if any(ch.isspace() for ch in token) or "://" in token: + return False + if token.startswith(("/", "~", "@", "$", "-", ".dev/")) or token == ".dev": + return False + if any(ch in token for ch in "{}<>()=|;,'\""): + return False + last = token.rstrip("/").rsplit("/", 1)[-1] + if last.split(".", 1)[0].lower() in NAMING_CONVENTIONS: + return False + has_extension = "." in last and last.rsplit(".", 1)[-1].lower() in FILE_EXTENSIONS + if "/" not in token: + return has_extension + first = token.split("/", 1)[0] + return has_extension or token.endswith("/") or (root / first).exists() + + +def check_structure(doc: Doc, violations: list[str]) -> None: + if len(doc.h1) != 1: + violations.append(f"structure: expected exactly one '# ' title, found {len(doc.h1)}") + titles = [title for _, title in doc.h2] + for title in sorted({t for t in titles if titles.count(t) > 1}): + violations.append(f"structure: '## {title}' appears more than once") + for name in SECTIONS: + if name not in titles: + violations.append(f"structure: missing '## {name}'") + present = [t for t in dict.fromkeys(titles) if t in SECTIONS] + if present != [name for name in SECTIONS if name in present]: + violations.append(f"structure: sections out of order, expected {', '.join(SECTIONS)}") + if titles and "Maintaining this file" in titles and titles[-1] != "Maintaining this file": + violations.append("structure: '## Maintaining this file' must be the last section") + if not any(DEFAULT_BRANCH.search(line) for _, line in doc.intro()): + violations.append("structure: the intro has no 'Default branch is `x`' sentence") + + +def check_text(text: str, max_lines: int, violations: list[str]) -> None: + lines = text.splitlines() + if len(lines) > max_lines: + violations.append(f"length: {len(lines)} lines, over the {max_lines}-line cap") + for number, line in enumerate(lines, 1): + if EM_DASH in line: + violations.append(f"em-dash: line {number}") + if "< set[str]: + slugs: set[str] = set() + seen: set[tuple[str, str]] = set() + section = doc.section("Open items") + if any(title == "Open items" for _, title in doc.h2) and not any(line.strip() for _, line in section): + violations.append("open-items: '## Open items' is empty; remove it when there are none") + for number, line in section: + if not line.strip() or line.startswith((" ", "\t")): + continue + match = OPEN_ITEM.match(line) + if not match: + violations.append(f"open-items: line {number} is not '- [ ] slug: text'") + continue + slug, detail = match.group(1), match.group(2).strip() + if slug not in SLUGS: + violations.append(f"open-items: line {number} uses unknown slug '{slug}'") + if (slug, detail) in seen: + violations.append(f"open-items: line {number} repeats an earlier item") + seen.add((slug, detail)) + slugs.add(slug) + return slugs + + +def check_slots(doc: Doc, item_slugs: set[str], violations: list[str]) -> None: + found: dict[str, int] = {} + for number, line in doc.section("Commands"): + match = SLOT_LINE.match(line) + if not match or match.group(1) not in SLOTS: + continue + label, value = match.group(1), match.group(2).strip() + found[label] = found.get(label, 0) + 1 + if value.startswith("`"): + continue + if re.match(r"^n/a \(.+\)", value): + continue + if re.match(r"^none\b", value): + if SLOTS[label] not in item_slugs: + violations.append(f"commands: '{label}: none' (line {number}) has no '- [ ] {SLOTS[label]}:' open item") + continue + violations.append(f"commands: '{label}' (line {number}) must be a backticked command, 'none', or 'n/a (reason)'") + for label in SLOTS: + if found.get(label, 0) == 0: + violations.append(f"commands: missing the '- {label}:' slot") + elif found[label] > 1: + violations.append(f"commands: the '{label}' slot appears {found[label]} times") + + +def check_paths(root: Path, doc: Doc, violations: list[str]) -> None: + skipped = {n for n, _ in doc.section("Open items")} + basenames: set[str] | None = None + for number, line in doc.lines: + if number in skipped: + continue + for raw in BACKTICK.findall(line): + token = raw.strip() + if not path_candidate(root, token) or ignored(root, token): + continue + if "/" in token: + if not exists(root, token): + violations.append(f"stale: `{token}` (line {number}) does not exist under {root}") + else: + if basenames is None: + basenames = tracked_basenames(root) + if not exists(root, token) and not fnmatch.filter(basenames, token): + violations.append(f"stale: `{token}` (line {number}) matches no file in the repository") + for target in LINK.findall(line): + if target.startswith(("http://", "https://", "mailto:", "#")): + continue + relative = target.split("#", 1)[0] + if relative and not exists(root, relative) and not ignored(root, relative): + violations.append(f"stale: link '{target}' (line {number}) does not exist under {root}") + + +def package_scripts(directory: Path) -> dict[str, str] | None: + manifest = directory / "package.json" + if not manifest.exists(): + return None + try: + return json.loads(manifest.read_text(encoding="utf-8")).get("scripts", {}) or {} + except (json.JSONDecodeError, AttributeError): + return None + + +def make_targets(directory: Path) -> tuple[set[str], bool] | None: + files = [directory / name for name in ("GNUmakefile", "makefile", "Makefile") if (directory / name).exists()] + if not files: + return None + files += sorted(directory.glob("*.mk")) + sorted(directory.glob("*/*.mk")) + targets: set[str] = set() + catch_all = False + for path in files: + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + match = re.match(r"^([^\s:#=][^:#=]*?)\s*::?(?!=)", line) + if not match: + continue + for name in match.group(1).split(): + if "%" in name: + catch_all = True + targets.add(name) + return targets, catch_all + + +def just_recipes(directory: Path) -> set[str] | None: + for name in ("justfile", "Justfile", ".justfile"): + path = directory / name + if path.exists(): + text = path.read_text(encoding="utf-8", errors="replace") + return {m.group(1) for m in re.finditer(r"^@?([A-Za-z0-9_-]+)[^:\n=]*:(?!=)", text, re.MULTILINE)} + return None + + +def task_names(directory: Path) -> set[str] | None: + for name in ("Taskfile.yml", "Taskfile.yaml", "taskfile.yml", "taskfile.yaml"): + path = directory / name + if not path.exists(): + continue + names: set[str] = set() + inside = False + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if re.match(r"^tasks:\s*$", line): + inside = True + elif re.match(r"^\S", line): + inside = False + elif inside: + match = re.match(r"^ ([A-Za-z0-9_.:-]+):", line) + if match: + names.add(match.group(1)) + return names + return None + + +def segments(command: str) -> list[list[str]]: + parts = re.split(r"\s*(?:&&|\|\||;)\s*", command) + result = [] + for part in parts: + part = part.split("|", 1)[0].strip() + try: + words = shlex.split(part) + except ValueError: + words = part.split() + while words and re.match(r"^[A-Z_][A-Z0-9_]*=", words[0]): + words = words[1:] + if words: + result.append(words) + return result + + +def declared_dependencies(*directories: Path) -> set[str]: + names: set[str] = set() + for directory in directories: + manifest = directory / "package.json" + if not manifest.exists(): + continue + try: + data = json.loads(manifest.read_text(encoding="utf-8")) + except json.JSONDecodeError: + continue + for key in ("dependencies", "devDependencies", "optionalDependencies"): + names.update((data.get(key) or {}).keys()) + return names + + +def path_arguments_exist(args: list[str], cwd: Path, rel: str) -> str: + """Test runners take path prefixes as filters; a slashed argument must prefix something real.""" + for arg in args: + if "/" in arg and not arg.startswith(("-", "/")) and "*" not in arg: + if not glob.glob(glob.escape(str(cwd / arg)) + "*"): + return f"`{arg}` matches nothing in {rel}" + return "" + + +def resolve_package_script(runner: str, words: list[str], cwd: Path, rel: str, root: Path) -> str | None: + """Return a violation, "" when resolved, or None when not statically decidable.""" + args = words[1:] + if not args: + return "" + first = args[0] + if first.startswith("-") or first in {"workspace", "workspaces"}: + return None + if first in {"run", "run-script"}: + if len(args) < 2: + return "" + name = args[1] + if name.startswith("-"): + return None + elif runner != "bun" and first in RUNNER_SCRIPTS: + name = RUNNER_SCRIPTS[first] + elif first == "test": + return path_arguments_exist(args[1:], cwd, rel) + elif first in PM_BUILTINS[runner]: + return "" + elif runner == "npm": + return None + else: + name = first + if runner == "bun" and (cwd / name).is_file(): + return "" + manifest = "package.json" if rel == "." else f"{rel}/package.json" + scripts = package_scripts(cwd) + if scripts is not None and name in scripts: + return "" + # bun and yarn (and an implicit pnpm call) also run a dependency's binary, e.g. `bun turbo`. + runs_binaries = runner in {"bun", "yarn"} or (runner == "pnpm" and first not in {"run", "run-script"}) + if runs_binaries and ((cwd / "node_modules" / ".bin" / name).exists() or name in declared_dependencies(cwd, root)): + return None + if scripts is None: + return f"no readable {manifest} for `{name}`" + return f"`{name}` is not a script in {manifest}" + + +def resolve(words: list[str], cwd: Path, root: Path) -> str | None: + runner = words[0] + rel = str(cwd.relative_to(root)) if cwd != root else "." + if runner in PM_BUILTINS: + return resolve_package_script(runner, words, cwd, rel, root) + if runner == "make": + args = words[1:] + if "-C" in args: + index = args.index("-C") + if index + 1 < len(args): + cwd = cwd / args[index + 1] + rel = str(cwd.relative_to(root)) + args = args[:index] + args[index + 2:] + wanted = [a for a in args if not a.startswith("-") and "=" not in a] + found = make_targets(cwd) + if found is None: + return f"no Makefile in {rel}" + targets, catch_all = found + missing = [t for t in wanted if t not in targets] + if missing and catch_all: + return None + return f"make target(s) {', '.join(missing)} not defined in {rel}" if missing else "" + if runner == "just": + recipes = just_recipes(cwd) + if recipes is None: + return f"no justfile in {rel}" + wanted = [a for a in words[1:2] if not a.startswith("-")] + return f"just recipe `{wanted[0]}` not defined in {rel}" if wanted and wanted[0] not in recipes else "" + if runner == "task": + names = task_names(cwd) + if names is None: + return f"no Taskfile in {rel}" + wanted = [a for a in words[1:2] if not a.startswith("-")] + return f"task `{wanted[0]}` not defined in {rel}" if wanted and wanted[0] not in names else "" + if runner in FILE_RUNNERS: + args = words[1:] + if not args or args[0] in {"-m", "-c", "-e", "--eval"}: + return "" + target = next((a for a in args if not a.startswith("-")), None) + if target is None: + return "" + return "" if (cwd / target).exists() else f"`{target}` does not exist in {rel}" + if runner.startswith("./"): + return "" if (cwd / runner).exists() else f"`{runner}` does not exist in {rel}" + if runner in NOTE_RUNNERS: + return None + return "" + + +def check_commands(root: Path, doc: Doc, violations: list[str], notes: list[str]) -> int: + skipped = {n for n, _ in doc.section("Open items")} + in_commands = {n for n, _ in doc.section("Commands")} + resolved = 0 + known = set(PM_BUILTINS) | FILE_RUNNERS | NOTE_RUNNERS | {"make", "just", "task", "cd"} + for number, line in doc.lines: + if number in skipped: + continue + for raw in BACKTICK.findall(line): + command = raw.strip() + parts = segments(command) + if not parts: + continue + if not (parts[0][0] in known or parts[0][0].startswith("./")): + if number in in_commands and " " in command: + notes.append(f"note: `{command}` (line {number}) uses a runner this script does not know; confirm it runs") + continue + cwd = root + for words in parts: + if words[0] == "cd": + target = words[1] if len(words) > 1 else "." + if not (cwd / target).is_dir(): + violations.append(f"command: `{command}` (line {number}) cds into missing `{target}`") + break + cwd = (cwd / target).resolve() + continue + outcome = resolve(words, cwd, root) + if outcome is None: + notes.append(f"note: `{command}` (line {number}) is not resolved statically; confirm it runs") + elif outcome: + violations.append(f"command: `{command}` (line {number}): {outcome}") + else: + resolved += 1 + return resolved + + +def check_branch(root: Path, doc: Doc, violations: list[str], notes: list[str]) -> None: + stated = next((m.group(1) for _, line in doc.intro() if (m := DEFAULT_BRANCH.search(line))), None) + if stated is None: + return + head = git(root, "symbolic-ref", "--short", "refs/remotes/origin/HEAD") + if head.returncode != 0: + notes.append(f"note: origin/HEAD does not resolve locally; confirm the default branch `{stated}` another way") + return + actual = head.stdout.strip().split("/", 1)[-1] + if actual != stated: + violations.append(f"branch: the file says `{stated}` but origin/HEAD is `{actual}`") + + +def check_duplicates(doc: Doc, violations: list[str]) -> None: + seen: dict[str, int] = {} + for number, line in doc.lines: + if HEADING.match(line): + continue + normal = " ".join(re.sub(r"^\s*(?:[-*]|\d+\.)\s+", "", line).lower().split()) + if len(normal) < 12 or set(normal) <= set("|-: "): + continue + if normal in seen: + violations.append(f"duplicate: line {number} repeats line {seen[normal]}") + else: + seen[normal] = number + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("agents_md") + parser.add_argument("--root", default=".") + parser.add_argument("--max-lines", type=int, default=150) + try: + args = parser.parse_args() + except SystemExit: + return 2 + root = Path(args.root).resolve() + path = Path(args.agents_md) + if not path.is_file(): + print(f"missing: {path}") + return 2 + if not root.is_dir(): + print(f"missing: --root {root} is not a directory") + return 2 + + text = path.read_text(encoding="utf-8", errors="replace") + doc = Doc(text) + violations: list[str] = [] + notes: list[str] = [] + check_structure(doc, violations) + check_text(text, args.max_lines, violations) + item_slugs = open_items(doc, violations) + check_slots(doc, item_slugs, violations) + check_paths(root, doc, violations) + resolved = check_commands(root, doc, violations, notes) + check_branch(root, doc, violations, notes) + check_duplicates(doc, violations) + + for note in dict.fromkeys(notes): + print(note) + if violations: + print("\n".join(violations)) + return 1 + print(f"check-agents-md: ok ({len(text.splitlines())} lines, {resolved} commands resolved, {len(set(notes))} notes)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/build_codex_plugin.py b/scripts/build_codex_plugin.py index 1b1469f..52dd5ce 100644 --- a/scripts/build_codex_plugin.py +++ b/scripts/build_codex_plugin.py @@ -100,6 +100,21 @@ def destination(self) -> Path: ), }, ), + "bootstrap": PluginConfig( + name="bootstrap", + display_name="Bootstrap", + short_description="Write a repository's AGENTS.md from a probe of its stack.", + capabilities=("Interactive", "Write"), + default_prompts=("Write or refresh this repository's AGENTS.md.",), + copied_dirs=("skills",), + skill_ui={ + "agents-md": ( + "AGENTS.md", + "Probe the stack and write a root AGENTS.md", + "Use $bootstrap:agents-md to probe this repository and write or refresh its root AGENTS.md.", + ), + }, + ), } diff --git a/scripts/validate.sh b/scripts/validate.sh index 5b34b18..c9b8c05 100755 --- a/scripts/validate.sh +++ b/scripts/validate.sh @@ -995,6 +995,27 @@ PYEOF fi } +# =========================================================================== +# B01: the bootstrap plugin's checker tests and concept-map tripwires pass +# =========================================================================== +check_bootstrap_script() { + local dir="bootstrap/evals/tests" + [ -d "$dir" ] || return + if ! python3 - "$dir" >"$LOG_DIR/bootstrap-script-tests.log" 2>&1 <<'PYEOF' +import sys, unittest + +loader = unittest.TestLoader() +suite = loader.discover(start_dir=sys.argv[1], top_level_dir=".") +result = unittest.TextTestRunner(verbosity=2).run(suite) +sys.exit(0 if result.wasSuccessful() and result.testsRun > 0 else 1) +PYEOF + then + tail -60 "$LOG_DIR/bootstrap-script-tests.log" >&2 + fail "B01" "$dir" \ + "Bootstrap script tests failed (full output: $LOG_DIR/bootstrap-script-tests.log); run: python3 -m unittest discover -s bootstrap/evals/tests -t ." + fi +} + # =========================================================================== # Main # =========================================================================== @@ -1019,6 +1040,7 @@ check_pi check_factory_unattended check_factory_protocol check_factory_script +check_bootstrap_script if [ -s "$ERROR_FILE" ]; then echo "" From 22cbfcaecef06ab59f7b78fd4c0e4c365c3d489f Mon Sep 17 00:00:00 2001 From: tobrun Date: Fri, 25 Sep 2026 20:33:00 +0200 Subject: [PATCH 29/30] docs(bootstrap): record the bootstrap plugin in the overview and ledger What: docs/architecture.md now names three plugins, adds a bootstrap component row, a "Bootstrapping a repository" flow, B01 in the validation flow, the root AGENTS.md at the consuming-repository boundary, and the checker as an entry point. docs/decisions.md gains a Bootstrap plugin area: D-bootstrap-plugin, -skill-name, -reach, -merge, -distilled-reference, -checker, and the not-doing entries -metrics, -pi, and -dev-reads-agents-md. CLAUDE.md describes the third plugin and says the Codex generator runs after changing any source plugin. Why: the overview and ledger are what future specs in this repository start from; D-packaging had rejected a third plugin for a different product, so the ledger records why this one differs, and the deferred follow-up of scope and build reading AGENTS.md's Commands section. --- CLAUDE.md | 10 ++++++---- docs/architecture.md | 18 ++++++++++++------ docs/decisions.md | 33 +++++++++++++++++++++++++++++++++ 3 files changed, 51 insertions(+), 10 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index b405514..eb3099d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -8,8 +8,10 @@ This is a monorepo for Tobrun's Claude Code, Codex, and Pi skills. The root `.claude-plugin/marketplace.json` references the Claude source plugins. `.agents/plugins/marketplace.json` references generated Codex plugins under `plugins/`. The root `package.json` exposes the `dev` source skills as a Pi -package. There are two plugins: `dev`, the hand-invoked development workflow, -and `factory`, whose `run` skill is the one sanctioned exception to "skills +package. There are three plugins: `dev`, the hand-invoked development workflow; +`bootstrap`, whose `agents-md` skill probes a consuming repository and writes +its root AGENTS.md from the workflow's SDLC lessons; and `factory`, whose `run` +skill is the one sanctioned exception to "skills never invoke each other" - it drives the pipeline declared in the consuming repository's `.factory/config.yaml`, or the built-in `scope`, `scope-review`, `build`, and `ship` phase bodies under `factory/phases/` by path, and judges @@ -23,7 +25,7 @@ each phase's completion itself. `run` is the plugin's only invocable skill. │ └── marketplace.json # Marketplace manifest listing all plugins ├── .agents/plugins/ │ └── marketplace.json # Codex marketplace manifest -├── {plugin-name}/ # Individual plugin directory (dev) +├── {plugin-name}/ # Individual plugin directory (dev, factory, bootstrap) │ ├── .claude-plugin/ │ │ └── plugin.json # Plugin metadata (name, version, author) │ ├── README.md # Plugin documentation @@ -53,7 +55,7 @@ each phase's completion itself. `run` is the plugin's only invocable skill. - Every `SKILL.md` must set `disable-model-invocation: true`; all skills in this repo are human-triggered only, and skills recommend the next step instead of invoking each other - except the factory `run` skill, the one sanctioned invoker, which launches its phase bodies (`factory/phases/`, not skills) by path. - Plan files under `.dev/` are never committed. - Do not edit `plugins/` directly. Run `python3 scripts/build_codex_plugin.py` - after changing `dev/`; the generator builds every configured + after changing any source plugin; the generator builds every configured plugin, removes Claude-only frontmatter, and writes Codex `agents/openai.yaml` invocation policy. It never copies `evals/`. - Keep the root Pi package version equal to `dev/.claude-plugin/plugin.json`. diff --git a/docs/architecture.md b/docs/architecture.md index 754647c..df734a0 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1,11 +1,11 @@ # Architecture Purpose: this repository is a monorepo of agent skills and the tooling that ships them. -Two plugins live here: `dev`, the hand-invoked development workflow (scope, scope-review, build, ship, commit, and two presentation skills), and `factory`, whose `run` skill orchestrates copies of the same four phases unattended. +Three plugins live here: `dev`, the hand-invoked development workflow (scope, scope-review, build, ship, commit, and two presentation skills); `factory`, whose `run` skill orchestrates copies of the same four phases unattended; and `bootstrap`, whose `agents-md` skill writes a consuming repository's root AGENTS.md from the workflow's lessons. A person invokes each `dev` skill by hand; skills never invoke each other, and each one recommends the next step instead - except the factory `run` skill, the one sanctioned invoker, which launches its copied phase skills by path and judges their completion itself. The generated Codex distributions under `plugins/` are built from their source plugin and never edited by hand. -Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin returns as an orchestrator skill; its removed runner stays gone) +Captured: 2026-09-17 (full, scope) - Updated: 2026-09-25 (the bootstrap plugin writes a consuming repository's AGENTS.md) ## Components @@ -13,10 +13,11 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin ret | --------- | -------------- | -------- | -------- | | dev skills | the interactive workflow skills and the references and scripts they share | `dev/skills/`, `dev/references/`, `dev/scripts/` | nothing at runtime; read by Claude, Codex, opencode, and Pi hosts | | dev evals | comprehension evals for the skills | `dev/evals/` | nothing at runtime | -| generated plugins | the Codex distribution built from `dev/`, with invocation policy in agents/openai.yaml | `plugins/` | Codex plugin marketplace | +| generated plugins | the Codex distributions built from each source plugin, with invocation policy in agents/openai.yaml | `plugins/` | Codex plugin marketplace | | repo scripts | validation of the whole repository, the plugin generator, and the Pi transport self-test | `scripts/` | every other component | | harden tools | optional local analyses a ship gauntlet can draw on: added lines, coverage-weighted complexity, flaky reruns, mutation | `tools/harden/` | nothing; run by hand against a target repository | | research and todo | plans, findings, and reading notes | `research/`, `todo/` | nothing | +| bootstrap plugin | the `agents-md` skill, its references distilled from the dev skills, `bootstrap/skills/agents-md/scripts/check-agents-md.py` that every draft loops against, and the checker's tests | `bootstrap/skills/`, `bootstrap/evals/` | the consuming repository's root AGENTS.md; reads its manifests, CI config, git history, and CLAUDE.md | | factory plugin | the `run` orchestrator skill (its only invocable skill), the four built-in phase bodies it reads by path, `factory/scripts/factory-config.py` for the pipeline config, their references and scripts | `factory/skills/`, `factory/phases/`, `factory/references/`, `factory/scripts/`, `factory/evals/` | the host's subagent tool, the consuming repository's .dev/ and .factory/ | ## Flows @@ -28,7 +29,12 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin ret ### Building the Codex distribution 1. `scripts/build_codex_plugin.py` copies skills/, references/, and `scripts/` of the source plugin (and factory/phases/ for the factory) into plugins/{name}/, strips Claude-only frontmatter, and writes agents/openai.yaml for invocable skills. -2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), the Pi package and transport (P01), and the factory plugin's unattended wording and result protocol; `--check` mode of the generator fails when `plugins/` is stale. +2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), the Pi package and transport (P01), the factory plugin's unattended wording and result protocol, and the bootstrap checker's tests (B01); `--check` mode of the generator fails when `plugins/` is stale. + +### Bootstrapping a repository +1. A person invokes `/bootstrap:agents-md` (or `$bootstrap:agents-md`) in a consuming repository. +2. The skill probes the stack read-only, maps it onto its concept map, interviews only for what the probe cannot settle, and runs each command it will list. +3. It drafts outside the repository, loops `bootstrap/skills/agents-md/scripts/check-agents-md.py` until it exits 0, shows the diff for approval, then writes the root AGENTS.md; it never commits and never writes CLAUDE.md. ### A factory run 1. A person invokes `/factory:run "a request"` (or a path to one); preflight runs factory-config.py check, which classes each phase as built-in, injected or foreign and validates the pipeline from the consuming repository's .factory/config.yaml (the built-in default when there is none), and stops before the interview on any finding. @@ -43,7 +49,7 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-20 (the factory plugin ret | Claude Code, Codex, opencode, Pi | host | the skills | each reads the skills in its own format; the generated tree under `plugins/` serves Codex | | git and GitHub | external | the ship skill | branch state and `gh pr` calls made during a ship run | | Jira | external HTTP | dev skills | through `acli`, only when .dev/config.json enables it | -| consuming repository | store | the skills | `.dev/` plan files, `docs/` ledgers, and the .factory/ config and injected phase copies | +| consuming repository | store | the skills | `.dev/` plan files, `docs/` ledgers, the .factory/ config and injected phase copies, and the root AGENTS.md bootstrap writes | ## Cross-cutting @@ -52,4 +58,4 @@ Rules: no em dash anywhere, SKILL.md under roughly 150 lines, every skill `disab ## Entry points -`scripts/validate.sh`, `scripts/build_codex_plugin.py`, `scripts/test_pi_runner.sh`, `dev/scripts/skill-metrics.py`. +`scripts/validate.sh`, `scripts/build_codex_plugin.py`, `scripts/test_pi_runner.sh`, `dev/scripts/skill-metrics.py`, `bootstrap/skills/agents-md/scripts/check-agents-md.py`. diff --git a/docs/decisions.md b/docs/decisions.md index eb210ab..a933948 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -303,3 +303,36 @@ D-cross-harness: Does this slice also route phases to other harnesses and models D-foreign-metrics: Do foreign phases report skill metrics? (2026-09-20, config-driven-factory/spec.md) ✗ require the `skill-metrics.py` call in the launch prompt - imposes our tooling on a foreign skill ⊘ not doing - the orchestrator records its own per-attempt timestamps under D-phase-timings, so nothing is required of the skill; reopen if a custom pipeline needs per-skill metrics ? verify: the closing report shows per-phase timings on a custom pipeline + +## Bootstrap plugin + +D-bootstrap-plugin: Where does the skill that writes a repository's AGENTS.md live? (2026-09-25, user request) + ✓ a third plugin, `bootstrap` - its audience is any repository, including ones that never install `dev`, and the user named it as a new plugin (user, 2026-09-25) ⚠ D-packaging rejected "a third plugin to version and install", but for factory's phase bodies, a different product with a different audience + ✗ a `dev` skill - grows dev's namespace and the Pi package for a one-time setup step most dev sessions never need + ✗ a factory phase - bootstrapping happens once per repository, not once per request +D-bootstrap-skill-name: What is the skill called? (2026-09-25, user request) + ✓ `agents-md`, invoked as `/bootstrap:agents-md` - names its one deliverable (auto-applied at Confidence: 85%) + ✗ `bootstrap` - `/bootstrap:bootstrap` stutters, and "bootstrap" already names the ledger bootstrap in `dev/skills/scope/references/bootstrap.md` + ✗ `init` - collides with the host's built-in `/init` +D-bootstrap-reach: What does the skill write? (2026-09-25, user request) + ✓ the root AGENTS.md only, written as standalone outcome rules any agent follows without the dev plugin; a missing concept becomes an open item, never a scaffold (user, 2026-09-25) + ✗ a CLAUDE.md that imports `@AGENTS.md` - Claude Code reads AGENTS.md directly now (user, 2026-09-25); an existing CLAUDE.md is merge input, and the closing report recommends removing it + ✗ scaffolding `docs/architecture.md`, `docs/dependencies.md`, or a `.gitignore` entry - out of the reach the user chose + ✗ nested per-directory AGENTS.md files - root only (user, 2026-09-25) +D-bootstrap-merge: What happens to an existing AGENTS.md or CLAUDE.md? (2026-09-25, user request) + ✓ merge and prune: every line is classed keep, update, prune, or conflict; a conflict with a lesson is always a question; the diff is approved before writing (user, 2026-09-25) + ✗ overwrite - loses the repo-specific gotchas that are the file's most valuable content + ✗ audit only - leaves the drift in place +D-bootstrap-distilled-reference: Where do the lessons the file encodes come from at runtime? (2026-09-25, user request) + ✓ the skill's own `concepts.md` and `template.md`, distilled from the dev skills with each row naming its source - the generated Codex tree copies only the plugin's own directory, so a link into `dev/references` would dangle there ⚠ a dev lesson that changes in place is not caught; `test_concept_sources.py` catches only a moved or renamed source + ✗ link into `dev/references` - resolves in this repository and breaks in the Codex distribution +D-bootstrap-checker: How is the written file kept honest? (2026-09-25, user request) + ✓ `check-agents-md.py`, which the skill loops against, gated by B01 in `scripts/validate.sh` - per P-deterministic-guards-over-prose; only resolvers that can be definitely wrong raise violations, the rest print notes, so a false positive never traps the loop ⚠ the rule lines' wording stays a judgment call the checker does not read + ✗ prose instructions alone - the stale paths and missing scripts found in existing sibling AGENTS.md files show prose alone drifts +D-bootstrap-metrics: Does the skill report run metrics? (2026-09-25, user request) + ✗ copy `skill-metrics.py` - writes `.dev/metrics.jsonl` into a repository whose `.dev/` may not be gitignored yet, against the AGENTS.md-only reach + ⊘ not doing - reopen if per-run cost across repositories is wanted +D-bootstrap-pi: Does bootstrap ship to Pi? (2026-09-25, user request) + ⊘ not doing - P01 and D-pi keep the Pi package dev-only; reopen when Pi gains plugin namespaces +D-bootstrap-dev-reads-agents-md: Do scope and build read AGENTS.md's Commands section? (2026-09-25, user request) + ⊘ not doing yet - scope's Validation block and build's command discovery still probe manifests every run; reopen once bootstrap has run on real repositories and the Commands section has proven reliable ? verify: a functional run in `bootstrap/evals/results.md` From 0130c2f155f3e07e6f3d013907c78dc548f4f54e Mon Sep 17 00:00:00 2001 From: tobrun Date: Thu, 1 Oct 2026 14:23:32 +0200 Subject: [PATCH 30/30] perf(build): gate each wave once, brief each agent, cap change sets What: build now runs a spec's Validation block once per wave, as the gate before the wave's commits ("The wave gate" in parallel.md). A change-set implementer runs only the test files it adds or edits plus typecheck and lint. Commands marked `(end of build)` in the block, and the e2e suite, benchmarks, and suite-repeating analyses even when unmarked, run once on the final tree with the CI-parity gate. ci-parity.md no longer re-runs a command that is green on the same tree. The seen-red rule keeps its guarantee and loses its cost: free when the test is written first, otherwise one break per slice running one test file. change-set-brief.py (new, build/scripts) cuts spec.md down to one change set: every section but research and the change plan, the decisions the change set links, its own plan, and implementation-notes.md without its test and seam inventories. Lines are verbatim. parallel.md hands each agent its brief instead of the whole spec and notes. lint-spec.py fails a change set over 25 scenarios (change sets already logged in implementation-notes.md are exempt, since they never renumber) and prints, on a clean spec, the build waves the file lists allow and the shared files that make a change set wait. scope asks for a wide plan, one owner per shared file, and the `(end of build)` mark; the scope-review feasibility lens checks both. check-tests.py splits the `Tests added:` line only where a path::name follows the comma, so a test name may hold commas. The factory phase copies carry the same edits, plugins/ is regenerated, dev goes to 3.3.0, and validate.sh gains D01, which runs the 22 new unit tests under dev/evals/tests. docs/decisions.md gains a Build speed area: D-wave-gate, D-seen-red-cost, D-change-set-size, D-wave-report, D-change-set-brief. Why: feedback on the contexia searchable-kb build named five causes of a slow build. The Validation block (lint, typecheck, build, unit, integration, e2e, bench, and a complexity script that re-runs the suite with coverage) ran more than once per change set. One change set spent 47 break-and-rerun cycles proving 75 tests red. Change set 4 carried 38 scenarios over about 60 files and took 73 minutes. Change sets 3, 4, and 5 queued behind shared files. Every agent read about 130 KB of spec and notes before writing anything; the brief for change set 5 is 42 KB. The comma split was costing loops too: 75 of the 91 problems check-tests.py reported on that plan were test names cut in two. An old-versus-new run of the parallel-wave eval is recorded in dev/evals/results.md: the Validation block ran 2 times instead of 4 and no subagent ran it. Wall time did not improve on that fixture, whose suite costs 20 seconds; a real build has to show the saving. Considered: running the block once per build (commits in between would be unverified); a file-count limit (file lists are prose, the count would be a guess); failing the lint on a serial plan (no threshold separates a careless chain from a necessary one); worktrees for overlapping change sets (an overlap usually is a real dependency). Directive: dev/skills/{scope,build}/scripts and their copies under factory/phases must stay byte-identical; FactoryCopiesTest checks the three scripts this change touches. --- dev/.claude-plugin/plugin.json | 2 +- dev/README.md | 5 +- dev/evals/README.md | 1 + dev/evals/build.json | 7 +- dev/evals/results.md | 29 +- dev/evals/scope.json | 2 + dev/evals/tests/__init__.py | 0 dev/evals/tests/test_build_speed.py | 264 ++++++++++++++++++ dev/references/ci-parity.md | 2 + dev/skills/build/SKILL.md | 10 +- dev/skills/build/references/parallel.md | 25 +- dev/skills/build/scripts/change-set-brief.py | 169 +++++++++++ dev/skills/build/scripts/check-tests.py | 4 +- dev/skills/scope-review/references/lenses.md | 3 +- dev/skills/scope/SKILL.md | 6 +- dev/skills/scope/scripts/lint-spec.py | 107 ++++++- docs/architecture.md | 2 +- docs/decisions.md | 23 ++ factory/phases/build/SKILL.md | 10 +- factory/phases/build/references/parallel.md | 25 +- .../phases/build/scripts/change-set-brief.py | 169 +++++++++++ factory/phases/build/scripts/check-tests.py | 4 +- .../phases/scope-review/references/lenses.md | 3 +- factory/phases/scope/SKILL.md | 6 +- factory/phases/scope/scripts/lint-spec.py | 107 ++++++- factory/references/ci-parity.md | 2 + package.json | 2 +- plugins/dev/.codex-plugin/plugin.json | 2 +- plugins/dev/references/ci-parity.md | 2 + plugins/dev/skills/build/SKILL.md | 10 +- .../dev/skills/build/references/parallel.md | 25 +- .../skills/build/scripts/change-set-brief.py | 169 +++++++++++ .../dev/skills/build/scripts/check-tests.py | 4 +- .../skills/scope-review/references/lenses.md | 3 +- plugins/dev/skills/scope/SKILL.md | 6 +- plugins/dev/skills/scope/scripts/lint-spec.py | 107 ++++++- plugins/factory/phases/build/SKILL.md | 10 +- .../phases/build/references/parallel.md | 25 +- .../phases/build/scripts/change-set-brief.py | 169 +++++++++++ .../phases/build/scripts/check-tests.py | 4 +- .../phases/scope-review/references/lenses.md | 3 +- plugins/factory/phases/scope/SKILL.md | 6 +- .../factory/phases/scope/scripts/lint-spec.py | 107 ++++++- plugins/factory/references/ci-parity.md | 2 + scripts/validate.sh | 22 ++ 45 files changed, 1594 insertions(+), 71 deletions(-) create mode 100644 dev/evals/tests/__init__.py create mode 100644 dev/evals/tests/test_build_speed.py create mode 100755 dev/skills/build/scripts/change-set-brief.py create mode 100755 factory/phases/build/scripts/change-set-brief.py create mode 100755 plugins/dev/skills/build/scripts/change-set-brief.py create mode 100755 plugins/factory/phases/build/scripts/change-set-brief.py diff --git a/dev/.claude-plugin/plugin.json b/dev/.claude-plugin/plugin.json index ae8825b..e8403dc 100644 --- a/dev/.claude-plugin/plugin.json +++ b/dev/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dev", - "version": "3.2.0", + "version": "3.3.0", "description": "Development workflow skills: scope changes with argued decisions, build across unit/integration/e2e with every scenario proven by tests, ship with a deterministic quality gauntlet and an adversarially verified review, create structured commits that feed a decision ledger, and render pitches or comprehension quizzes.", "author": { "name": "Tobrun" diff --git a/dev/README.md b/dev/README.md index 336b89d..baad37d 100644 --- a/dev/README.md +++ b/dev/README.md @@ -14,7 +14,8 @@ Every skill is explicit-invocation only: Claude Code and Pi use `disable-model-i Specs a change by interviewing for the real problem behind the request, cataloging every design decision (with a subagent blind-spot pass on full-size changes), and arguing each one against alternatives in a `✓`/`✗`/`?`/`⚠`/`⊘` notation with evidence marks. Writes a self-contained spec at `.dev/{plan-name}/spec.md` - research decisions, scope with invariants and a Validation block of the repo's real commands, and a change plan of numbered change sets each ending in a layer-tagged `tests:` line - designed as a fresh-context handoff to `build`. -A checker (`scripts/lint-spec.py`) enforces the spec's mechanics - unique slugs, argued alternatives, echoes that match their decision, tagged test scenarios - so the prose stays about judgment. +A checker (`scripts/lint-spec.py`) enforces the spec's mechanics - unique slugs, argued alternatives, echoes that match their decision, tagged test scenarios, at most 25 scenarios per change set - so the prose stays about judgment. +On a clean spec it prints the build waves the file lists allow and the shared files that make change sets wait, so the plan is shaped for parallel work before build starts. Promotes durable decisions to `docs/decisions.md` and cross-boundary invariants to `docs/contracts.md`, renders an expandable-card spec view, and has a reverse mode that audits the implicit decisions already embedded in existing code. ### scope-review @@ -28,7 +29,9 @@ An APPROVED verdict means every finding was refined or answered: `build` can sta Executes a spec's change sets at the layer each `tests:` scenario is tagged with - unit for business logic, integration for real cross-component seams, e2e for driving the actual running application. Enforces outcomes rather than rituals: every test must have been seen to fail before its green counts, with strict failing-test-first reserved for bug fixes, where red is the proof the issue was actually reproduced. +The red is kept cheap: free when the test is written first, one break per slice and one test file otherwise. Runs independent change sets in parallel as waves of subagents batched by disjoint file lists in spec order, committing each change set and appending to a running `implementation-notes.md` that logs any deviations forced by an edge case. +Each subagent starts from a brief that `scripts/change-set-brief.py` cuts from the spec for its change set and runs only its own tests; the spec's Validation block runs once per wave as the gate before its commits, and e2e, benchmark, and coverage commands run once at the end. Once every change set is committed, drives the real app against a mocked environment, loops until every e2e scenario passes, then renders the e2e report: screenshots per scenario for frontend systems, Test Scenario and Data Model State tables for everything else. Build never pushes or opens a PR; `ship` does, once the change is hardened and reviewed. diff --git a/dev/evals/README.md b/dev/evals/README.md index 2f3b956..0dbef3c 100644 --- a/dev/evals/README.md +++ b/dev/evals/README.md @@ -6,6 +6,7 @@ Eval definitions for the `dev` plugin's skills: realistic prompts and objective - `{skill}.json` - one file per skill: the eval prompt(s), the fixture each expects, and the assertions to grade the output against. Covers all 7 skills: `scope`, `scope-review`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz`. - `results.md` - the record of the most recent full run: scores, methodology, and findings. +- `tests/` - unit tests for the deterministic scripts the skills loop against (`lint-spec.py`, `change-set-brief.py`), run by `scripts/validate.sh` as check D01. `build` runs in `"functional"` mode (a real fixture, a real subagent run, assertions checked against the actual output). `scope`, `scope-review`, `commit`, `ship`, `to-pitch`, and `to-quiz` run in `"comprehension"` mode instead - each depends on either an interactive question loop, a live codebase, or prior artifacts (a finished spec, implementation notes, an e2e report) that are too expensive to stage on every iteration, so these check policy comprehension of the skill text directly. diff --git a/dev/evals/build.json b/dev/evals/build.json index 93368cb..147d02d 100644 --- a/dev/evals/build.json +++ b/dev/evals/build.json @@ -22,6 +22,8 @@ "Tests are behavior-named (e.g. \"applies a discount\") rather than test1/test2, cover cases beyond the happy path, and no assertion recomputes the expected value the way the implementation does", "The user is not asked about running ship after change set 1 finishes - no review prompt appears until both are done", "Each change set gets its own commit on the current branch (at least 2 commits)", + "The spec's Validation block is run once per change set as the gate before its commit, not repeated on an unchanged tree", + "No test is proven red by its own break-and-rerun: tests are written first, or one break covers a slice and only the affected test file is run", ".dev/{plan-name}/implementation-notes.md is created and has one entry per change set", "After both change sets are done, the user is asked to run ship rather than a review panel or gauntlet being launched automatically", "Each implementation-notes.md entry carries a Tests added: line naming real path::test name references", @@ -38,7 +40,9 @@ "The subagents do not commit, stage, or otherwise touch git state - the orchestrator makes the commits", "The subagents do not write to implementation-notes.md themselves; the orchestrator appends each entry", "The subagents are told not to launch the app or run e2e scenarios", - "The spec's Validation block commands are re-run by the orchestrator rather than the subagent's claim being taken at face value", + "scripts/change-set-brief.py is run for the wave and each subagent is handed its brief rather than told to read the whole spec.md and implementation-notes.md", + "The subagents are told to run only their own test files plus typecheck and lint, not the spec's Validation block", + "The spec's Validation block commands are run by the orchestrator, once for the wave, rather than the subagents' claims being taken at face value", "Three commits exist, one per change set, in spec order", "The orchestrator carries each subagent's testsAdded into that change set's Tests added: line rather than discarding it" ] @@ -77,6 +81,7 @@ "fixture": "a repo whose spec Validation block runs unit/lint/build, while .github/workflows/pr.yml also requires a project-owned screenshot-matrix command. The active-month screenshot fails deterministically on the first UTC day, although the feature's tagged e2e scenarios pass.", "assertions": [ "The pull-request workflow is inspected and the screenshot-matrix command is run before the user is offered a PR", + "A required command that already ran green on the unchanged final tree is recorded, not run a second time", "The screenshot failure is diagnosed and fixed rather than labeled pre-existing, flaky, unrelated, or an accepted deviation", "The fix removes the time-dependent assumption without adding a retry, sleep, timeout increase, or looser assertion", "The full screenshot-matrix command is rerun and green before the PR question", diff --git a/dev/evals/results.md b/dev/evals/results.md index 36fae3f..7cd9888 100644 --- a/dev/evals/results.md +++ b/dev/evals/results.md @@ -1,10 +1,37 @@ # Eval Results -Status: no run recorded for the current skill set. +Status: no full run recorded for the current skill set; one partial run below. The last recorded run (2026-07-24) covered the pre-pivot five-skill chain and was invalidated by the pivot to the scope pipeline; its scores were removed rather than left to invite false confidence. Run the harness below against the current 6 skills (`scope`, `commit`, `build`, `ship`, `to-pitch`, `to-quiz`) and replace this file with the dated results. +## Partial run, 2026-10-01: build `parallel-wave`, before and after the build-speed change + +One functional run per version, not the full harness. +Old is the build skill at commit 22cbfca, new is the working tree that adds the wave gate, the change-set brief, and the cheap-red rule. +The fixture was a Node library with a 20 second legacy test in its suite, and a spec with change sets 1 and 2 disjoint and change set 3 editing files of both. +Every run of a Validation command was written to a log by the fixture's package scripts, so the counts are measured, not reported. + +| Check | Old | New | +| ----- | --- | --- | +| Change sets 1 and 2 launched as subagents in one message | pass | pass | +| Change set 3 started only after 1 and 2 were committed | pass | pass | +| Subagents left git state and `implementation-notes.md` alone | pass | pass | +| Subagents told not to launch the app | pass | pass | +| Each subagent handed a brief from `change-set-brief.py` instead of the whole spec | fail (reads all of `spec.md`) | pass | +| Subagents run only their own test file plus syntax checks | fail (each ran the Validation block) | pass | +| Orchestrator runs the Validation block once per wave | pass | pass | +| Three change-set commits in spec order, `check-tests.py` clean | pass | pass | +| Runs of the full Validation block (log) | 4 | 2 | +| Extra break-and-rerun cycles to see a test red | 2 | 2 | +| Duration by `skill-metrics.py` | 7m 10s | 8m 55s | + +Findings: + +- The Validation block ran half as often, and no subagent ran it. +- Wall time did not improve on this fixture: the suite costs 20 seconds, so two saved runs are under a minute, less than the difference between two model runs. The saving grows with the cost of the block and the number of change sets; a real build is needed to measure it. +- The old run's notes named a test with a comma in it, which `check-tests.py` then split in two. The checker now splits only where a `path::name` follows the comma. + ## Re-running this harness 1. Pick a baseline commit (the last commit before the change under test) and the working tree as "new". diff --git a/dev/evals/scope.json b/dev/evals/scope.json index bb1f22a..8189bbd 100644 --- a/dev/evals/scope.json +++ b/dev/evals/scope.json @@ -12,6 +12,8 @@ "The change is sized small or full, the choice is told to the user with a reason, and the user can override", "The spec is written to .dev/{plan-name}/spec.md with a research section of D- slugged decisions using the marks (chosen, rejected, open, accepted downside, not doing)", "The scope section ends with a Validation block listing the repo's real commands, discovered rather than guessed", + "The Validation block lists each check once, and any e2e suite, benchmark, or coverage re-run in it is marked (end of build)", + "No change set carries more than 25 scenarios, and edits to a file several change sets need are gathered into one change set rather than repeated across them", "The spec does not implement code; no source files are modified", "The run ends by recommending build; it is never invoked directly", "Every change set ends with one tests: line whose scenarios carry [unit], [integration], or [e2e] tags, or tests: none with a reason", diff --git a/dev/evals/tests/__init__.py b/dev/evals/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/dev/evals/tests/test_build_speed.py b/dev/evals/tests/test_build_speed.py new file mode 100644 index 0000000..1d36bf1 --- /dev/null +++ b/dev/evals/tests/test_build_speed.py @@ -0,0 +1,264 @@ +"""Unit tests for the two checks that keep a build fast: lint-spec.py's change-set +size limit and wave report, and build's change-set-brief.py. + +Run from the repo root: python3 -m unittest discover -s dev/evals/tests -t . +Each test writes a scratch plan directory and runs the script as a subprocess, so +it proves the CLI contract the scope and build skills loop against. +""" + +from __future__ import annotations + +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[3] +LINT = REPO_ROOT / "dev" / "skills" / "scope" / "scripts" / "lint-spec.py" +BRIEF = REPO_ROOT / "dev" / "skills" / "build" / "scripts" / "change-set-brief.py" +CHECK = REPO_ROOT / "dev" / "skills" / "build" / "scripts" / "check-tests.py" + +HEAD = """# Coupons + +## Research + +D-coupon-store: Where do coupons live? + ✓ a table - one query reads them + ✗ a config file - a redeploy per coupon + +D-expiry-clock: Which clock decides expiry? + ✓ the server clock - one source of time + ✗ the client clock - a user can set it back + +## Scope + +Inputs: a coupon code. Outputs: a discounted total. + +### Validation + +- `npm run typecheck` +- `npm test` + +## Change plan + +""" + + +def scenarios(count: int) -> str: + return "; ".join(f"[unit] case {n} -> outcome {n}" for n in range(1, count + 1)) + + +def change_set(number: int, files: str, count: int = 1, decision: str = "") -> str: + link = f" - decisions: {decision}" if decision else "" + return f"{number}. Change set {number}\n a. {files} - edit{link}\n tests: {scenarios(count)}\n\n" + + +def run(script: Path, *args: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, str(script), *args], check=False, capture_output=True, text=True + ) + + +class PlanCase(unittest.TestCase): + def setUp(self) -> None: + scratch = tempfile.TemporaryDirectory() + self.addCleanup(scratch.cleanup) + self.plan = Path(scratch.name) / "coupons" + self.plan.mkdir() + + def write_spec(self, *change_sets: str) -> Path: + spec = self.plan / "spec.md" + spec.write_text(HEAD + "".join(change_sets), encoding="utf-8") + return spec + + def write_notes(self, text: str) -> None: + (self.plan / "implementation-notes.md").write_text(text, encoding="utf-8") + + +class ScenarioLimitTest(PlanCase): + def test_change_set_at_the_limit_is_clean(self) -> None: + spec = self.write_spec(change_set(1, "`src/a.ts`", 25)) + result = run(LINT, str(spec)) + self.assertEqual(result.returncode, 0, result.stdout) + + def test_change_set_over_the_limit_fails_naming_it_and_its_count(self) -> None: + spec = self.write_spec(change_set(1, "`src/a.ts`"), change_set(2, "`src/b.ts`", 26)) + result = run(LINT, str(spec)) + self.assertEqual(result.returncode, 1) + self.assertIn("change set 2 carries 26 scenarios", result.stdout) + self.assertNotIn("change set 1 carries", result.stdout) + + def test_oversized_change_set_already_built_is_not_resized(self) -> None: + spec = self.write_spec(change_set(1, "`src/a.ts`", 40), change_set(2, "`src/b.ts`")) + self.write_notes("## Change set 1: Change set 1\n- What was done: it\n") + result = run(LINT, str(spec)) + self.assertEqual(result.returncode, 0, result.stdout) + + +class WaveReportTest(PlanCase): + def waves(self, *change_sets: str) -> str: + result = run(LINT, str(self.write_spec(*change_sets))) + self.assertEqual(result.returncode, 0, result.stdout) + return result.stdout + + def test_disjoint_change_sets_share_one_wave(self) -> None: + out = self.waves(change_set(1, "`src/a.ts`"), change_set(2, "`src/b.ts`")) + self.assertIn("2 change sets in 1 waves - [1 2]", out) + self.assertNotIn("waits on", out) + + def test_shared_file_queues_change_sets_and_is_named(self) -> None: + out = self.waves( + change_set(1, "`src/a.ts`, `src/compose.ts`"), + change_set(2, "`src/b.ts`, `src/compose.ts`"), + change_set(3, "`src/c.ts`, `src/compose.ts`"), + ) + self.assertIn("3 change sets in 3 waves - [1] [2] [3]", out) + self.assertIn(" 2 waits on 1: src/compose.ts", out) + self.assertIn(" 3 waits on 2: src/compose.ts", out) + + def test_later_change_set_skips_ahead_only_past_change_sets_it_shares_nothing_with(self) -> None: + out = self.waves( + change_set(1, "`src/a.ts`"), + change_set(2, "`src/a.ts`, `src/b.ts`"), + change_set(3, "`src/b.ts`"), + change_set(4, "`src/d.ts`"), + ) + # 3 shares nothing with 1 but must not overtake 2, which is still held back. + self.assertIn("4 change sets in 3 waves - [1 4] [2] [3]", out) + + def test_bare_file_name_matches_the_path_that_ends_in_it(self) -> None: + out = self.waves(change_set(1, "`src/app/compose.ts`"), change_set(2, "`compose.ts`")) + self.assertIn("[1] [2]", out) + + def test_code_in_backticks_and_the_tests_line_are_not_files(self) -> None: + out = self.waves( + change_set(1, "`src/a.ts` - sets `embedder.model` and `job.stages`"), + change_set(2, "`src/b.ts` - reads `embedder.model` and `job.stages`"), + ) + self.assertIn("[1 2]", out) + + def test_built_change_sets_are_left_out_of_the_waves(self) -> None: + spec = self.write_spec( + change_set(1, "`src/a.ts`"), change_set(2, "`src/a.ts`"), change_set(3, "`src/c.ts`") + ) + self.write_notes("## Change set 1: Change set 1\n- What was done: it\n") + result = run(LINT, str(spec)) + self.assertIn("2 change sets in 1 waves - [2 3]", result.stdout) + + def test_single_change_set_prints_no_waves(self) -> None: + out = self.waves(change_set(1, "`src/a.ts`")) + self.assertNotIn("build waves", out) + + +class ChangeSetBriefTest(PlanCase): + def setUp(self) -> None: + super().setUp() + self.write_spec( + change_set(1, "`src/store.ts`", decision="D-coupon-store (✓ a table)"), + change_set(2, "`src/expiry.ts`", decision="D-expiry-clock (✓ the server clock)"), + ) + + def brief(self, number: int = 2) -> str: + result = run(BRIEF, str(self.plan), str(number)) + self.assertEqual(result.returncode, 0, result.stderr) + return result.stdout + + def test_brief_keeps_its_change_set_and_the_scope_section_verbatim(self) -> None: + out = self.brief() + self.assertIn("2. Change set 2\n a. `src/expiry.ts` - edit", out) + self.assertIn("Inputs: a coupon code. Outputs: a discounted total.", out) + self.assertIn("- `npm run typecheck`", out) + + def test_brief_keeps_only_the_decisions_its_change_set_links(self) -> None: + out = self.brief() + self.assertIn("D-expiry-clock: Which clock decides expiry?", out) + self.assertIn(" ✗ the client clock - a user can set it back", out) + self.assertNotIn("D-coupon-store: Where do coupons live?", out) + + def test_brief_names_the_other_change_sets_without_their_plans(self) -> None: + out = self.brief() + self.assertIn("- 1. Change set 1", out) + self.assertNotIn("`src/store.ts`", out) + + def test_brief_carries_earlier_work_and_deviations_but_no_test_inventory(self) -> None: + self.write_notes( + "# Implementation notes\n\n## Change set 1: Change set 1\n" + "- What was done: added the coupon table\n" + "- Seams tested: the store\n" + "- Tests added: test/store.test.ts::reads a coupon\n" + "- Deviations from spec: codes are stored upper-case\n" + ) + out = self.brief() + self.assertIn("### Change set 1: Change set 1", out) + self.assertIn("- What was done: added the coupon table", out) + self.assertIn("- Deviations from spec: codes are stored upper-case", out) + self.assertNotIn("Tests added", out) + self.assertNotIn("Seams tested", out) + + def test_linked_decision_missing_from_research_is_said_so(self) -> None: + self.write_spec(change_set(1, "`src/a.ts`", decision="D-ghost (✓ x)")) + self.assertIn("D-ghost: not argued in the spec's research section", self.brief(1)) + + def test_out_dir_writes_one_brief_per_change_set_and_prints_each_path(self) -> None: + out_dir = self.plan.parent / "briefs" + result = run(BRIEF, str(self.plan), "1", "2", "--out-dir", str(out_dir)) + self.assertEqual(result.returncode, 0, result.stderr) + for number in (1, 2): + written = out_dir / f"change-set-{number}.md" + self.assertIn(f"# Brief: change set {number} of coupons", written.read_text(encoding="utf-8")) + self.assertIn(str(written.resolve()), result.stdout) + + def test_unknown_change_set_exits_2_naming_the_ones_that_exist(self) -> None: + result = run(BRIEF, str(self.plan), "9") + self.assertEqual(result.returncode, 2) + self.assertIn("no change set 9 (change sets: 1, 2)", result.stderr) + + def test_several_change_sets_without_out_dir_exit_2(self) -> None: + self.assertEqual(run(BRIEF, str(self.plan), "1", "2").returncode, 2) + + def test_missing_spec_exits_2(self) -> None: + self.assertEqual(run(BRIEF, str(self.plan.parent / "absent"), "1").returncode, 2) + + +class TestNameWithCommaTest(PlanCase): + """A test name is prose; a comma in it must not send the build back around the check loop.""" + + def check(self, tests_added: str) -> subprocess.CompletedProcess[str]: + self.write_spec(change_set(1, "`src/a.ts`", 2)) + repo = self.plan.parent + (repo / "a.test.ts").write_text( + 'test("a set name, and a longer text counts more", () => {});\n' + 'test("an empty text gives a unit vector", () => {});\n', + encoding="utf-8", + ) + self.write_notes(f"## Change set 1: Change set 1\n- Tests added: {tests_added}\n") + return run(CHECK, str(self.plan), "--repo-root", str(repo)) + + def test_test_name_holding_a_comma_is_one_test(self) -> None: + result = self.check( + "a.test.ts::a set name, and a longer text counts more, a.test.ts::an empty text gives a unit vector" + ) + self.assertEqual(result.returncode, 0, result.stdout) + + def test_named_test_that_is_not_in_the_file_still_fails(self) -> None: + result = self.check("a.test.ts::a set name, and a longer text counts more, a.test.ts::never written") + self.assertEqual(result.returncode, 1) + self.assertIn("contains no test named 'never written'", result.stdout) + + +class FactoryCopiesTest(unittest.TestCase): + """The factory build and scope phases ship the same scripts.""" + + def test_factory_copies_match_dev(self) -> None: + for dev, factory in ( + (LINT, REPO_ROOT / "factory" / "phases" / "scope" / "scripts" / LINT.name), + (BRIEF, REPO_ROOT / "factory" / "phases" / "build" / "scripts" / BRIEF.name), + (CHECK, REPO_ROOT / "factory" / "phases" / "build" / "scripts" / CHECK.name), + ): + with self.subTest(script=dev.name): + self.assertEqual(dev.read_bytes(), factory.read_bytes()) + + +if __name__ == "__main__": + unittest.main() diff --git a/dev/references/ci-parity.md b/dev/references/ci-parity.md index 382ac06..92157c9 100644 --- a/dev/references/ci-parity.md +++ b/dev/references/ci-parity.md @@ -21,6 +21,8 @@ Record the commands and outcomes in the plan's `implementation-notes.md`. Run every reproducible project-owned required-check command against the final checkout, after feature E2E and after any hardening edits. +A command that already ran green on this exact tree - no file changed since - +is not run a second time: record its result and where it came from. - A failure is work to fix, including a test described as flaky, unrelated, or pre-existing. Diagnose and remove its nondeterminism; do not add retries, diff --git a/dev/skills/build/SKILL.md b/dev/skills/build/SKILL.md index f1f696e..962d4b0 100644 --- a/dev/skills/build/SKILL.md +++ b/dev/skills/build/SKILL.md @@ -17,11 +17,11 @@ Read [references/layers.md](references/layers.md), [references/tests.md](referen 1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. 2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. -3. For each wave, run its change sets in parallel per the same reference, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). +3. For each wave, run its change sets in parallel per the same reference, pass its wave gate once, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). 4. Move straight to the next wave. Never stop after one change set or wave to ask about review. 5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. 6. Then run the full e2e pass per "The e2e layer" below over the whole spec, and loop on failures until it is green. -7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. +7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), plus every Validation command the wave gates left for the end, starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. 8. After the e2e report and CI-parity gate, close per "Closing message". Build never pushes or opens a PR; that is `ship`'s phase 3. ## Jira sync @@ -58,12 +58,12 @@ Neither does a failed e2e loop: say what is blocked, then still point at `ship`. Each change set, whether you run it yourself or a subagent runs it, follows the same loop: - Test at the seams the spec's scope section declares, per [references/tests.md](references/tests.md); if the declared boundary is wrong or missing, follow its fallback and log the change under Deviations - do not stall on it. -- Implement in **vertical slices**: one scenario's behavior at a time, its test written before or right after the code - the enforced outcome is what matters, not the ritual order. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. -- Run the change set's own tests and typecheck continuously; once the change set is green, run the spec's Validation block verbatim - it is the wider suite plus typecheck/lint - and only report done when it passes clean. +- Implement in **vertical slices**: one scenario's behavior at a time, its test written before the code or right after it - before is the cheap order, because its first run is then the red the rules below ask for. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. +- Run the change set's own tests and typecheck continuously - the test files it adds or edits, never the whole suite - and report done when they are green. The spec's Validation block is not run per change set: it is the wave gate, run once per wave before its commits, per "The wave gate" in [references/parallel.md](references/parallel.md). ## Rules of the loop -- **Every test must have been seen red.** A test that has never failed proves nothing: earn its green by writing it before the code, or by briefly breaking the behavior once after. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. +- **Every test must have been seen red, once and cheaply.** A test that has never failed proves nothing. Written before the code, its first run is that red and costs nothing. Written after, break the behavior once per slice - one break covers every test of the slice - and run only the affected test file: never one break per test, never the suite for a red. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. - **One slice at a time.** One seam, one behavior, one test, one minimal implementation per cycle. - **Refactoring is not part of the loop.** It belongs to `ship`'s review phase. - **Keep going.** A red test, a failing e2e scenario, or an edge case that contradicts the spec is work to do, not a reason to hand back. Fix it, log the deviation, continue. Stop early only when a blocking question makes further work unsafe or wasted. diff --git a/dev/skills/build/references/parallel.md b/dev/skills/build/references/parallel.md index d3b6a05..3aac303 100644 --- a/dev/skills/build/references/parallel.md +++ b/dev/skills/build/references/parallel.md @@ -20,17 +20,27 @@ Never start work belonging to the next wave while the current wave is in flight. Launch one `Agent` per change set, **all in a single message** so they run concurrently. A wave of one change set needs no subagent: implement it yourself in the main thread. +First write the wave's briefs, one command for the whole wave: + +```bash +python3 {build-skill-root}/scripts/change-set-brief.py .dev/{plan-name} {N} {N} --out-dir /tmp/{project-slug}/briefs/{plan-name} +``` + +A brief is the spec cut down to one change set: the scope section, the decisions that change set links, its own plan, and what earlier change sets did and deviated on. +Every line is the spec's own, so an agent starts from about a third of the reading and loses nothing it builds against. +Write them per wave, never once up front: the notes they carry grow with every committed change set. + Each agent prompt contains: 1. The role: `You implement exactly one change set of a spec. Other agents implement sibling change sets concurrently; stay inside your change set's file list.` -2. The absolute path to `spec.md`, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The spec is a self-contained handoff by design; the agent reads all of it, then implements only its own change set. +2. The absolute path to the change set's brief, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The brief stands in for `spec.md` and `implementation-notes.md`: the agent reads neither whole, and opens the spec only to follow something the brief points at. 3. The absolute path to this skill's `SKILL.md`, with the instruction to follow its "The change-set loop" and "Rules of the loop" sections - read like the other references, not pasted into the prompt. 4. Hard constraints: - Implement only this change set's `[unit]` and `[integration]` scenarios. `[e2e]` scenarios are run once per spec by the orchestrator afterwards - do not launch the app. - Do not commit, stage, or touch git state. The orchestrator commits. - Do not edit files outside your change set's file list. If the change set genuinely needs a file another change set owns, stop and report it as a conflict instead of editing it. - Do not edit `implementation-notes.md`. Report your entry; the orchestrator appends it. - - Run the spec's Validation block and report its real result. A red result is a fact to report, not something to hide or paper over. + - Run only your own checks: the test files this change set adds or edits, plus typecheck and lint over the packages it touches. Never run the spec's Validation block, the whole suite, the e2e suite, or a benchmark - siblings are mid-edit in the same tree, so a wider run measures their half-finished work, and the orchestrator runs the Validation block once for the whole wave. Report the commands and their real result. A red result is a fact to report, not something to hide or paper over. 5. The output contract below. Output contract (the agent's final message must be exactly one fenced JSON block): @@ -50,13 +60,22 @@ Output contract (the agent's final message must be exactly one fenced JSON block ## After a wave -1. Verify rather than trust the reports: run the spec's Validation block yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. +1. Verify rather than trust the reports: run the wave gate below yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. 2. Commit each change set's work on the current branch, in number order, one commit per change set. A skipped-ahead change set's commit waits until every earlier change set is committed, so history keeps the spec's order. 3. Append each change set's entry to `implementation-notes.md` from `whatWasDone`, `seamsTested`, `testsAdded`, and `deviations`. `testsAdded` becomes the entry's `Tests added:` line, which the scenario checker reads. A `status: blocked` change set, a failing validation, or a reported conflict is yours to finish in the main thread before the next wave starts - do not carry a red change set forward and do not relaunch the same agent on the same failure more than once. If two change sets in a wave edited the same file anyway, reconcile it yourself and log it under Deviations. +## The wave gate + +The spec's Validation block is the gate between a wave and its commits, and the only full run a wave gets - a wave of one you implemented yourself included. +Each of its commands runs once per tree: a green result stands until a file changes, so nothing is re-run "to be sure" before committing, and after a fix the failed command runs first and the rest only once it is green. +Start commands that share no state together (lint, typecheck, and the unit suite), and write the wave's notes entries while they run. + +A command the block marks `(end of build)` is left out of the wave gate, and so are three kinds even when a spec lists one unmarked, because each costs minutes and tells a wave nothing it needs to commit: the e2e suite (the e2e pass runs it), benchmarks, and any analysis that re-runs the suite to measure it - coverage, complexity, mutation. +Each of those runs once, on the final tree, with the CI-parity gate. + ## When not to parallelize - The spec has one change set, or every change set's files overlap with the one before it: run them sequentially yourself. diff --git a/dev/skills/build/scripts/change-set-brief.py b/dev/skills/build/scripts/change-set-brief.py new file mode 100755 index 0000000..a44b2ba --- /dev/null +++ b/dev/skills/build/scripts/change-set-brief.py @@ -0,0 +1,169 @@ +#!/usr/bin/env python3 +"""Cut a spec down to what one change set's implementer needs. + +Usage: python3 change-set-brief.py .dev/{plan-name} N [N ...] [--out-dir DIR] + +A brief is spec.md minus the decisions the change set does not link and minus +the other change sets, followed by what earlier change sets did and deviated on +from implementation-notes.md. Every kept line is verbatim, so an agent reading +the brief reads the spec's own words, a third of them. + +With --out-dir, writes DIR/change-set-N.md per change set and prints each path +with its size against the spec. Without it, prints the one brief to stdout. +Exit 2 on a missing spec or an unknown change set. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +DECISION = re.compile(r"^(D-[a-z0-9]+(?:-[a-z0-9]+)*):") +LINKED = re.compile(r"\bD-[a-z0-9]+(?:-[a-z0-9]+)*") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +# The two note lines only the scenario checker and the reviewers read. +NOTE_NOISE = re.compile(r"^\s*-\s*(Tests added|Seams tested):", re.IGNORECASE) + + +def split_sections(lines: list[str]) -> list[tuple[str, list[str]]]: + """The spec as (lower-cased ## heading, lines) pairs; the text before the first is ''.""" + sections: list[tuple[str, list[str]]] = [("", [])] + for line in lines: + heading = HEADING.match(line) + if heading: + sections.append((heading.group(1).lower(), [line])) + else: + sections[-1][1].append(line) + return sections + + +def decision_entries(research: list[str]) -> dict[str, list[str]]: + """Each decision's header and its indented alternative lines, by slug.""" + entries: dict[str, list[str]] = {} + current = None + for line in research[1:]: + header = DECISION.match(line) + if header: + current = entries.setdefault(header.group(1), []) + current.append(line) + elif current is not None and line[:1] in (" ", "\t") and line.strip(): + current.append(line) + elif line.strip(): + current = None + return entries + + +def change_set_blocks(plan: list[str]) -> dict[int, list[str]]: + """Each change set's lines, from its numbered line to the next one.""" + blocks: dict[int, list[str]] = {} + current = None + for line in plan[1:]: + change_set = CHANGE_SET.match(line) + if change_set: + current = blocks.setdefault(int(change_set.group(1)), []) + if current is not None: + current.append(line) + for block in blocks.values(): + while block and not block[-1].strip(): + block.pop() + return blocks + + +def notes_digest(notes: Path) -> list[str]: + """implementation-notes.md without its test and seam inventories.""" + if not notes.is_file(): + return [] + kept = [ + line for line in notes.read_text(encoding="utf-8").splitlines() + if not NOTE_NOISE.match(line) and not line.startswith("# ") + ] + return kept if any(NOTE_ENTRY.match(line) for line in kept) else [] + + +def brief(plan_dir: Path, number: int) -> str: + spec = plan_dir / "spec.md" + sections = split_sections(spec.read_text(encoding="utf-8").splitlines()) + by_name = dict(sections) + blocks = change_set_blocks(by_name.get("change plan", [""])) + if number not in blocks: + known = ", ".join(str(n) for n in sorted(blocks)) or "none" + raise LookupError(f"{spec} has no change set {number} (change sets: {known})") + block = blocks[number] + decisions = decision_entries(by_name.get("research", [""])) + linked = list(dict.fromkeys(slug for line in block for slug in LINKED.findall(line))) + + out = [ + f"# Brief: change set {number} of {plan_dir.name}", + "", + f"This is {spec.resolve()} cut down to change set {number}: every line below is the spec's own.", + "Left out are the decisions this change set does not link and the other change sets' plans.", + "Open the spec only to follow something this brief points at, never to read it whole.", + "", + ] + for name, lines in sections: + if name == "research": + out += ["## Research (the decisions this change set links)", ""] + for slug in linked: + out += decisions.get(slug, [f"{slug}: not argued in the spec's research section"]) + [""] + if not linked: + out += ["This change set links no decision.", ""] + elif name == "change plan": + out += [f"## Change plan (change set {number} only)", ""] + block + [""] + others = [blocks[n][0].strip() for n in sorted(blocks) if n != number] + if others: + out += ["The other change sets, whose files are not yours:"] + [f"- {o[:140]}" for o in others] + [""] + else: + out += lines + ([""] if lines and lines[-1].strip() else []) + digest = notes_digest(plan_dir / "implementation-notes.md") + if digest: + out += ["## What earlier change sets did (from implementation-notes.md)", ""] + out += [("#" + line) if line.startswith("## ") else line for line in digest] + return "\n".join(out).rstrip() + "\n" + + +def main(argv: list[str]) -> int: + args: list[str] = [] + out_dir = None + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--out-dir" and rest: + out_dir = Path(rest.pop(0)) + else: + args.append(value) + numbers = args[1:] + if not numbers or not all(n.isdigit() for n in numbers) or (out_dir is None and len(numbers) != 1): + print( + "usage: change-set-brief.py [ ...] [--out-dir DIR]\n" + " several change sets need --out-dir", + file=sys.stderr, + ) + return 2 + plan_dir = Path(args[0]) + spec = plan_dir / "spec.md" + if not spec.is_file(): + print(f"no spec.md at {spec}", file=sys.stderr) + return 2 + try: + briefs = {int(n): brief(plan_dir, int(n)) for n in numbers} + except LookupError as error: + print(error, file=sys.stderr) + return 2 + if out_dir is None: + sys.stdout.write(briefs[int(numbers[0])]) + return 0 + out_dir.mkdir(parents=True, exist_ok=True) + spec_bytes = len(spec.read_bytes()) + for number, text in briefs.items(): + target = out_dir / f"change-set-{number}.md" + target.write_text(text, encoding="utf-8") + size = len(text.encode("utf-8")) + print(f"{target.resolve()}: {size} bytes, {round(100 * size / spec_bytes)}% of spec.md") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/dev/skills/build/scripts/check-tests.py b/dev/skills/build/scripts/check-tests.py index 7d90d25..ccbb1aa 100755 --- a/dev/skills/build/scripts/check-tests.py +++ b/dev/skills/build/scripts/check-tests.py @@ -18,6 +18,8 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) NOTE_TESTS = re.compile(r"^\s*-\s*Tests added:\s*(.*)$", re.IGNORECASE) +# A comma separates two references only when a path::name follows it; a test name may hold commas. +NOTE_SEPARATOR = re.compile(r",\s*(?=[^,]*::)") def spec_scenarios(spec: Path, problem) -> dict[int, int]: @@ -65,7 +67,7 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in value.split(",") if t.strip()) + entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) return entries diff --git a/dev/skills/scope-review/references/lenses.md b/dev/skills/scope-review/references/lenses.md index 0e5fa5f..ed4f6d9 100644 --- a/dev/skills/scope-review/references/lenses.md +++ b/dev/skills/scope-review/references/lenses.md @@ -11,7 +11,8 @@ Does the change plan survive contact with the repo? - Every file a change set names exists, or the set says it is new; the described edit is possible at that site - the function, hook, or config it assumes is really there. - Prior art and idioms the spec cites exist where it says they do. - Premises about current behavior are checked against the code, never trusted: an "X already handles Y" claim that is false is a BLOCK naming the site. -- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on. +- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on - and nothing slower: an e2e suite, benchmark, or coverage re-run listed there without its `(end of build)` mark is a CONCERN, because build pays for the block every wave. +- The plan is as wide as the change allows: change sets queued behind one shared file that a single change set could own are a CONCERN naming the file (`lint-spec.py` prints the waves). ## completeness diff --git a/dev/skills/scope/SKILL.md b/dev/skills/scope/SKILL.md index e7d432f..298c3bb 100644 --- a/dev/skills/scope/SKILL.md +++ b/dev/skills/scope/SKILL.md @@ -90,6 +90,7 @@ Name the components and flows the change adds, removes, or reshapes, in the over Efforts have second-order effects - capture them as nested sub-efforts, each carrying its own decisions back into the research section (rate limiting in scope means Redis setup, which carries config and deploy decisions). Record considered non-goals as `⊘` lines with a because clause - things someone weighed and cut, not mere omissions. End the scope with a `### Validation` block listing the repo's real typecheck/test/lint/build commands, discovered from `package.json`, a `Makefile`, CI config, or equivalent - never guess `npm test` into a `pytest` repo; ask if you cannot determine them. +`build` runs this block once per wave, so list each check once and mark every command a wave does not need - the e2e suite, a benchmark, a coverage or complexity run that repeats the suite - `(end of build)`: build runs those once, on the final tree. Writing style for the spec: ELI12, no similes or metaphors. ## 5. Review and research @@ -104,7 +105,9 @@ Writing style for the spec: ELI12, no similes or metaphors. Short fragmented sentences. Link decisions by ID wherever one applies, echoing the choice. Each change set ends with one `;`-separated `tests:` line - concrete scenarios as input -> expected outcome, each tagged `[unit]`, `[integration]`, or `[e2e]`, covering happy path, edge cases, and failure paths; a set with nothing to test says `tests: none - {reason}`. Specific enough that whoever writes the tests invents nothing; the author tags layers here because a fresh implementation session can't recover that intent. -Order change sets so each builds only on the ones before it; keep file lists disjoint where possible - `build` parallelizes consecutive change sets whose files don't overlap. +Order change sets so each builds only on the ones before it, and shape the plan wide: `build` runs change sets in parallel only while their file lists are disjoint, so a file three change sets edit makes them queue. +Land what several change sets share - a port, a schema, a registry, the composition root, a regenerated artifact - in one change set, and let the ones building on it own disjoint files. +Size each change set for one agent: past 25 scenarios it is several change sets, split along a seam. ``` 1. Change set 1 @@ -119,6 +122,7 @@ Order change sets so each builds only on the ones before it; keep file lists dis Then loop `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` until it exits clean. It owns the mechanics above; the spec is not final while it reports anything. +Once clean it prints the build waves the file lists allow and the files that make a change set wait; where change sets queue behind a shared file, move that file's edits into one change set and lint again. ## 7. Visualize diff --git a/dev/skills/scope/scripts/lint-spec.py b/dev/skills/scope/scripts/lint-spec.py index c45fdde..2cb73b3 100755 --- a/dev/skills/scope/scripts/lint-spec.py +++ b/dev/skills/scope/scripts/lint-spec.py @@ -3,6 +3,8 @@ Usage: python3 lint-spec.py .dev/{plan-name}/spec.md Exit 0 when the spec is clean, 1 with one problem per line otherwise. +A clean spec also gets the build waves its file lists allow, so the author sees +which shared files make change sets wait for each other. """ from __future__ import annotations @@ -20,8 +22,19 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +BACKTICKED = re.compile(r"`([^`\s]+)`") +# A backticked token is a file when it ends in a known file extension; +# `embedder.model` and `job.stages` are code, not paths. +FILE_NAME = re.compile( + r"(?:^|/)(?:Makefile|Dockerfile|[\w.@+-]+\." + r"(?:c|cc|cpp|cs|css|go|gradle|graphql|h|html|java|js|json|jsx|kt|lock|md|mjs|cjs|php|proto" + r"|py|rb|rs|scss|sh|sql|svelte|swift|tf|toml|ts|tsx|txt|vue|xml|yaml|yml))$" +) CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" +# One agent builds one change set; past this many scenarios it stops fitting in one sitting. +SCENARIO_LIMIT = 25 def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: @@ -122,7 +135,16 @@ def check_echoes( problem(number, f"the change plan links {slug}, which is still open or flagged") -def check_change_plan(body: list[tuple[int, str]], problem) -> None: +def built_change_sets(spec: Path) -> set[int]: + """Change sets build already logged; their size is history, and they never renumber.""" + notes = spec.parent / "implementation-notes.md" + if not notes.is_file(): + return set() + entries = (NOTE_ENTRY.match(line) for line in notes.read_text(encoding="utf-8").splitlines()) + return {int(entry.group(1)) for entry in entries if entry} + + +def check_change_plan(body: list[tuple[int, str]], built: set[int], problem) -> None: counts: dict[int, int] = {} current = None for number, line in body: @@ -148,15 +170,87 @@ def check_change_plan(body: list[tuple[int, str]], problem) -> None: if not value: problem(number, "tests: line is empty") continue - for scenario in value.split(";"): - scenario = scenario.strip() - if scenario and not LAYER.match(scenario): + scenarios = [scenario.strip() for scenario in value.split(";") if scenario.strip()] + for scenario in scenarios: + if not LAYER.match(scenario): problem(number, f"scenario '{scenario[:40]}' carries no [unit]/[integration]/[e2e] tag") + if len(scenarios) > SCENARIO_LIMIT and current not in built: + problem( + number, + f"change set {current} carries {len(scenarios)} scenarios; split it along a seam " + f"so no change set carries more than {SCENARIO_LIMIT}", + ) for change_set, seen in sorted(counts.items()): if seen != 1: problem(None, f"change set {change_set} has {seen} tests: lines; expected exactly one") +def change_set_files(body: list[tuple[int, str]]) -> dict[int, set[str]]: + """Files each change set names, read from the backticked paths outside its tests: line.""" + files: dict[int, set[str]] = {} + current = None + for _, line in body: + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + files.setdefault(current, set()) + if current is None or TESTS.match(line): + continue + files[current].update(t for t in BACKTICKED.findall(line) if FILE_NAME.search(t)) + return files + + +def shared_files(one: set[str], other: set[str]) -> list[str]: + """Paths two change sets both name; a bare file name matches any path that ends in it.""" + shared = one & other + for mine, theirs in ((one, other), (other, one)): + bare = {path for path in theirs if "/" not in path} + shared |= {path for path in mine if "/" in path and path.rsplit("/", 1)[-1] in bare} + return sorted(shared) + + +def build_waves(files: dict[int, set[str]]) -> tuple[list[list[int]], dict[int, tuple[int, list[str]]]]: + """The waves build's consecutive-disjoint batching gives when only file lists decide. + + A change set joins the current wave when it shares no file with the wave or with + any earlier change set still held back. Also returns, per change set that had to + wait, the change set it waited on last and the files they share. + """ + waves: list[list[int]] = [] + waited: dict[int, tuple[int, list[str]]] = {} + remaining = sorted(files) + while remaining: + wave: list[int] = [] + held: list[int] = [] + for change_set in remaining: + for earlier in wave + held: + shared = shared_files(files[change_set], files[earlier]) + if shared: + waited[change_set] = (earlier, shared) + held.append(change_set) + break + else: + wave.append(change_set) + waves.append(wave) + remaining = held + return waves, waited + + +def wave_report(body: list[tuple[int, str]], built: set[int]) -> list[str]: + files = {number: paths for number, paths in change_set_files(body).items() if number not in built} + if len(files) < 2: + return [] + waves, waited = build_waves(files) + layout = " ".join("[" + " ".join(str(n) for n in wave) + "]" for wave in waves) + lines = [ + f"build waves by file lists alone: {len(files)} change sets in {len(waves)} waves - {layout}" + ] + for change_set, (other, shared) in sorted(waited.items()): + more = f" (+{len(shared) - 3} more)" if len(shared) > 3 else "" + lines.append(f" {change_set} waits on {other}: {', '.join(shared[:3])}{more}") + return lines + + def check_validation(lines: list[str], problem) -> None: for index, line in enumerate(lines): if line.strip().lower() == "### validation": @@ -192,7 +286,8 @@ def problem(number: int | None, message: str) -> None: check_decisions(decisions, problem) check_echoes(found.get("scope", []), decisions, False, problem) check_echoes(found.get("change plan", []), decisions, True, problem) - check_change_plan(found.get("change plan", []), problem) + built = built_change_sets(path) + check_change_plan(found.get("change plan", []), built, problem) check_validation(lines, problem) for _, message in sorted(problems): @@ -201,6 +296,8 @@ def problem(number: int | None, message: str) -> None: print(f"\n{len(problems)} problem(s); the spec is not final.") return 1 print(f"{path}: clean - {len(decisions)} decisions argued.") + for line in wave_report(found.get("change plan", []), built): + print(line) return 0 diff --git a/docs/architecture.md b/docs/architecture.md index df734a0..4a4a254 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -29,7 +29,7 @@ Captured: 2026-09-17 (full, scope) - Updated: 2026-09-25 (the bootstrap plugin w ### Building the Codex distribution 1. `scripts/build_codex_plugin.py` copies skills/, references/, and `scripts/` of the source plugin (and factory/phases/ for the factory) into plugins/{name}/, strips Claude-only frontmatter, and writes agents/openai.yaml for invocable skills. -2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), the Pi package and transport (P01), the factory plugin's unattended wording and result protocol, and the bootstrap checker's tests (B01); `--check` mode of the generator fails when `plugins/` is stale. +2. `scripts/validate.sh` checks structure, links, skill length, the Codex distribution (C01), the Pi package and transport (P01), the factory plugin's unattended wording and result protocol, the bootstrap checker's tests (B01), and the dev spec-lint and change-set-brief tests (D01); `--check` mode of the generator fails when `plugins/` is stale. ### Bootstrapping a repository 1. A person invokes `/bootstrap:agents-md` (or `$bootstrap:agents-md`) in a consuming repository. diff --git a/docs/decisions.md b/docs/decisions.md index a933948..0d2d95e 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -304,6 +304,29 @@ D-foreign-metrics: Do foreign phases report skill metrics? (2026-09-20, config-d ✗ require the `skill-metrics.py` call in the launch prompt - imposes our tooling on a foreign skill ⊘ not doing - the orchestrator records its own per-attempt timestamps under D-phase-timings, so nothing is required of the skill; reopen if a custom pipeline needs per-skill metrics ? verify: the closing report shows per-phase timings on a custom pipeline +## Build speed + +D-wave-gate: How often does build run a spec's Validation block? (2026-10-01, user feedback on the contexia searchable-kb build) + ✓ once per wave, by the orchestrator, as the gate before the wave's commits; an implementer runs only the test files its change set adds or edits plus typecheck and lint, and e2e, benchmark, and suite-repeating analysis commands wait for the end of the build - the block ran more than once per change set, and it held the e2e suite, the benchmarks, and a complexity script that re-runs the suite with coverage (user, 2026-10-01; contexia `.dev/metrics.jsonl`: builds of 11,773 s and 17,565 s); in a shared tree an agent's full run also measures its siblings' half-finished work ⚠ a regression outside a change set's own tests is found at the wave gate, not by the agent that caused it, and a complexity or benchmark miss only at the end + ✗ once per build - the commits in between would be unverified, and a late red has to be bisected across change sets + ✗ two separate blocks in the spec, a fast one and a full one - a format change every reader of the block would have to learn; an `(end of build)` mark on a command says the same inside the one block, and build recognizes the three kinds unmarked, so older specs get the gate too +D-seen-red-cost: What may proving a test red cost? (2026-10-01, user feedback on the contexia searchable-kb build) + ✓ nothing when the test is written first, and one break per slice, running one test file, when it is written after - the rule guarantees that no test is unable to fail, and one break proves that for every test of the slice; one change set spent 47 break-and-rerun cycles on 75 tests (user, 2026-10-01) + ✗ drop the rule and lean on ship's mutation testing - that check is reported, not a gate (D-mutation-threshold), and it runs after the tests are already trusted + ✗ require test-first everywhere - build enforces the outcome, not the order; strict test-first stays reserved for bug fixes +D-change-set-size: How large may a change set be? (2026-10-01, user feedback on the contexia searchable-kb build) + ✓ at most 25 scenarios, enforced by `lint-spec.py`; a change set already logged in `implementation-notes.md` is exempt, because change sets never renumber - a 38-scenario change set took 73 minutes against 18 and 23 for the waves before it (contexia git log, 2026-10-01), and the limit flags 4 of the 61 change sets in the four contexia specs (27, 32, 38, 62 scenarios) ⚠ it counts scenarios, not files, so a change set with few scenarios and many files passes + ✗ a file-count limit - file lists are prose with abbreviations, so the count would be a guess the author can argue with + ✗ prose guidance alone - "iterate small" was already asked and did not hold; per P-deterministic-guards-over-prose +D-wave-report: How does scope see the parallelism its plan allows? (2026-10-01, user feedback on the contexia searchable-kb build) + ✓ `lint-spec.py` prints, on a clean spec, the build waves the file lists allow and the shared files that make a change set wait - "keep file lists disjoint where possible" was already asked and still produced three change sets queued behind `compose.ts`; printed rather than failed, because a chain is sometimes the honest shape of a change ⚠ file detection is a heuristic over backticked names, and what one change set consumes from another stays build's judgment + ✗ fail the lint on a chain - no threshold separates a careless chain from a necessary one + ✗ build overlapping change sets in separate worktrees and merge - spec order is the dependency order, so an overlap usually is a real dependency +D-change-set-brief: What does a change-set agent read before it starts? (2026-10-01, user feedback on the contexia searchable-kb build) + ✓ a brief `change-set-brief.py` cuts from the spec: every section but research and the change plan, the decisions its change set links, its own plan, and the notes without their test inventories - 42 KB against 130 KB of spec and notes for change set 5 of the contexia searchable-kb plan (measured 2026-10-01); every line is verbatim, so nothing is paraphrased ⚠ a decision a change set leans on but does not link is left out; the spec's path stays in the prompt for that + ✗ the whole spec and the notes - every agent pays for every other change set's plan + ✗ a summary the orchestrator writes per agent - output tokens are the slow ones, and a paraphrase can be wrong + ## Bootstrap plugin D-bootstrap-plugin: Where does the skill that writes a repository's AGENTS.md live? (2026-09-25, user request) diff --git a/factory/phases/build/SKILL.md b/factory/phases/build/SKILL.md index 797a9a1..61724ee 100644 --- a/factory/phases/build/SKILL.md +++ b/factory/phases/build/SKILL.md @@ -17,11 +17,11 @@ Read [references/layers.md](references/layers.md), [references/tests.md](referen 1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. 2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. -3. For each wave, run its change sets in parallel per the same reference, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). +3. For each wave, run its change sets in parallel per the same reference, pass its wave gate once, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). 4. Move straight to the next wave. Never stop after one change set or wave to ask about review. 5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. 6. Then run the full e2e pass per "The e2e layer" below over the whole spec, and loop on failures until it is green. -7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. +7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), plus every Validation command the wave gates left for the end, starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. 8. After the e2e report and CI-parity gate, close per "Closing message". Build never pushes or opens a PR; that is `ship`'s phase 3. ## Jira sync @@ -58,12 +58,12 @@ Neither does a failed e2e loop: say what is blocked, then still point at `ship`. Each change set, whether you run it yourself or a subagent runs it, follows the same loop: - Test at the seams the spec's scope section declares, per [references/tests.md](references/tests.md); if the declared boundary is wrong or missing, follow its fallback and log the change under Deviations - do not stall on it. -- Implement in **vertical slices**: one scenario's behavior at a time, its test written before or right after the code - the enforced outcome is what matters, not the ritual order. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. -- Run the change set's own tests and typecheck continuously; once the change set is green, run the spec's Validation block verbatim - it is the wider suite plus typecheck/lint - and only report done when it passes clean. +- Implement in **vertical slices**: one scenario's behavior at a time, its test written before the code or right after it - before is the cheap order, because its first run is then the red the rules below ask for. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. +- Run the change set's own tests and typecheck continuously - the test files it adds or edits, never the whole suite - and report done when they are green. The spec's Validation block is not run per change set: it is the wave gate, run once per wave before its commits, per "The wave gate" in [references/parallel.md](references/parallel.md). ## Rules of the loop -- **Every test must have been seen red.** A test that has never failed proves nothing: earn its green by writing it before the code, or by briefly breaking the behavior once after. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. +- **Every test must have been seen red, once and cheaply.** A test that has never failed proves nothing. Written before the code, its first run is that red and costs nothing. Written after, break the behavior once per slice - one break covers every test of the slice - and run only the affected test file: never one break per test, never the suite for a red. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. - **One slice at a time.** One seam, one behavior, one test, one minimal implementation per cycle. - **Refactoring is not part of the loop.** It belongs to `ship`'s review phase. - **Keep going.** A red test, a failing e2e scenario, or an edge case that contradicts the spec is work to do, not a reason to hand back. Fix it, log the deviation, continue. Stop early only when a blocking question makes further work unsafe or wasted. diff --git a/factory/phases/build/references/parallel.md b/factory/phases/build/references/parallel.md index d3b6a05..3aac303 100644 --- a/factory/phases/build/references/parallel.md +++ b/factory/phases/build/references/parallel.md @@ -20,17 +20,27 @@ Never start work belonging to the next wave while the current wave is in flight. Launch one `Agent` per change set, **all in a single message** so they run concurrently. A wave of one change set needs no subagent: implement it yourself in the main thread. +First write the wave's briefs, one command for the whole wave: + +```bash +python3 {build-skill-root}/scripts/change-set-brief.py .dev/{plan-name} {N} {N} --out-dir /tmp/{project-slug}/briefs/{plan-name} +``` + +A brief is the spec cut down to one change set: the scope section, the decisions that change set links, its own plan, and what earlier change sets did and deviated on. +Every line is the spec's own, so an agent starts from about a third of the reading and loses nothing it builds against. +Write them per wave, never once up front: the notes they carry grow with every committed change set. + Each agent prompt contains: 1. The role: `You implement exactly one change set of a spec. Other agents implement sibling change sets concurrently; stay inside your change set's file list.` -2. The absolute path to `spec.md`, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The spec is a self-contained handoff by design; the agent reads all of it, then implements only its own change set. +2. The absolute path to the change set's brief, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The brief stands in for `spec.md` and `implementation-notes.md`: the agent reads neither whole, and opens the spec only to follow something the brief points at. 3. The absolute path to this skill's `SKILL.md`, with the instruction to follow its "The change-set loop" and "Rules of the loop" sections - read like the other references, not pasted into the prompt. 4. Hard constraints: - Implement only this change set's `[unit]` and `[integration]` scenarios. `[e2e]` scenarios are run once per spec by the orchestrator afterwards - do not launch the app. - Do not commit, stage, or touch git state. The orchestrator commits. - Do not edit files outside your change set's file list. If the change set genuinely needs a file another change set owns, stop and report it as a conflict instead of editing it. - Do not edit `implementation-notes.md`. Report your entry; the orchestrator appends it. - - Run the spec's Validation block and report its real result. A red result is a fact to report, not something to hide or paper over. + - Run only your own checks: the test files this change set adds or edits, plus typecheck and lint over the packages it touches. Never run the spec's Validation block, the whole suite, the e2e suite, or a benchmark - siblings are mid-edit in the same tree, so a wider run measures their half-finished work, and the orchestrator runs the Validation block once for the whole wave. Report the commands and their real result. A red result is a fact to report, not something to hide or paper over. 5. The output contract below. Output contract (the agent's final message must be exactly one fenced JSON block): @@ -50,13 +60,22 @@ Output contract (the agent's final message must be exactly one fenced JSON block ## After a wave -1. Verify rather than trust the reports: run the spec's Validation block yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. +1. Verify rather than trust the reports: run the wave gate below yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. 2. Commit each change set's work on the current branch, in number order, one commit per change set. A skipped-ahead change set's commit waits until every earlier change set is committed, so history keeps the spec's order. 3. Append each change set's entry to `implementation-notes.md` from `whatWasDone`, `seamsTested`, `testsAdded`, and `deviations`. `testsAdded` becomes the entry's `Tests added:` line, which the scenario checker reads. A `status: blocked` change set, a failing validation, or a reported conflict is yours to finish in the main thread before the next wave starts - do not carry a red change set forward and do not relaunch the same agent on the same failure more than once. If two change sets in a wave edited the same file anyway, reconcile it yourself and log it under Deviations. +## The wave gate + +The spec's Validation block is the gate between a wave and its commits, and the only full run a wave gets - a wave of one you implemented yourself included. +Each of its commands runs once per tree: a green result stands until a file changes, so nothing is re-run "to be sure" before committing, and after a fix the failed command runs first and the rest only once it is green. +Start commands that share no state together (lint, typecheck, and the unit suite), and write the wave's notes entries while they run. + +A command the block marks `(end of build)` is left out of the wave gate, and so are three kinds even when a spec lists one unmarked, because each costs minutes and tells a wave nothing it needs to commit: the e2e suite (the e2e pass runs it), benchmarks, and any analysis that re-runs the suite to measure it - coverage, complexity, mutation. +Each of those runs once, on the final tree, with the CI-parity gate. + ## When not to parallelize - The spec has one change set, or every change set's files overlap with the one before it: run them sequentially yourself. diff --git a/factory/phases/build/scripts/change-set-brief.py b/factory/phases/build/scripts/change-set-brief.py new file mode 100755 index 0000000..a44b2ba --- /dev/null +++ b/factory/phases/build/scripts/change-set-brief.py @@ -0,0 +1,169 @@ +#!/usr/bin/env python3 +"""Cut a spec down to what one change set's implementer needs. + +Usage: python3 change-set-brief.py .dev/{plan-name} N [N ...] [--out-dir DIR] + +A brief is spec.md minus the decisions the change set does not link and minus +the other change sets, followed by what earlier change sets did and deviated on +from implementation-notes.md. Every kept line is verbatim, so an agent reading +the brief reads the spec's own words, a third of them. + +With --out-dir, writes DIR/change-set-N.md per change set and prints each path +with its size against the spec. Without it, prints the one brief to stdout. +Exit 2 on a missing spec or an unknown change set. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +DECISION = re.compile(r"^(D-[a-z0-9]+(?:-[a-z0-9]+)*):") +LINKED = re.compile(r"\bD-[a-z0-9]+(?:-[a-z0-9]+)*") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +# The two note lines only the scenario checker and the reviewers read. +NOTE_NOISE = re.compile(r"^\s*-\s*(Tests added|Seams tested):", re.IGNORECASE) + + +def split_sections(lines: list[str]) -> list[tuple[str, list[str]]]: + """The spec as (lower-cased ## heading, lines) pairs; the text before the first is ''.""" + sections: list[tuple[str, list[str]]] = [("", [])] + for line in lines: + heading = HEADING.match(line) + if heading: + sections.append((heading.group(1).lower(), [line])) + else: + sections[-1][1].append(line) + return sections + + +def decision_entries(research: list[str]) -> dict[str, list[str]]: + """Each decision's header and its indented alternative lines, by slug.""" + entries: dict[str, list[str]] = {} + current = None + for line in research[1:]: + header = DECISION.match(line) + if header: + current = entries.setdefault(header.group(1), []) + current.append(line) + elif current is not None and line[:1] in (" ", "\t") and line.strip(): + current.append(line) + elif line.strip(): + current = None + return entries + + +def change_set_blocks(plan: list[str]) -> dict[int, list[str]]: + """Each change set's lines, from its numbered line to the next one.""" + blocks: dict[int, list[str]] = {} + current = None + for line in plan[1:]: + change_set = CHANGE_SET.match(line) + if change_set: + current = blocks.setdefault(int(change_set.group(1)), []) + if current is not None: + current.append(line) + for block in blocks.values(): + while block and not block[-1].strip(): + block.pop() + return blocks + + +def notes_digest(notes: Path) -> list[str]: + """implementation-notes.md without its test and seam inventories.""" + if not notes.is_file(): + return [] + kept = [ + line for line in notes.read_text(encoding="utf-8").splitlines() + if not NOTE_NOISE.match(line) and not line.startswith("# ") + ] + return kept if any(NOTE_ENTRY.match(line) for line in kept) else [] + + +def brief(plan_dir: Path, number: int) -> str: + spec = plan_dir / "spec.md" + sections = split_sections(spec.read_text(encoding="utf-8").splitlines()) + by_name = dict(sections) + blocks = change_set_blocks(by_name.get("change plan", [""])) + if number not in blocks: + known = ", ".join(str(n) for n in sorted(blocks)) or "none" + raise LookupError(f"{spec} has no change set {number} (change sets: {known})") + block = blocks[number] + decisions = decision_entries(by_name.get("research", [""])) + linked = list(dict.fromkeys(slug for line in block for slug in LINKED.findall(line))) + + out = [ + f"# Brief: change set {number} of {plan_dir.name}", + "", + f"This is {spec.resolve()} cut down to change set {number}: every line below is the spec's own.", + "Left out are the decisions this change set does not link and the other change sets' plans.", + "Open the spec only to follow something this brief points at, never to read it whole.", + "", + ] + for name, lines in sections: + if name == "research": + out += ["## Research (the decisions this change set links)", ""] + for slug in linked: + out += decisions.get(slug, [f"{slug}: not argued in the spec's research section"]) + [""] + if not linked: + out += ["This change set links no decision.", ""] + elif name == "change plan": + out += [f"## Change plan (change set {number} only)", ""] + block + [""] + others = [blocks[n][0].strip() for n in sorted(blocks) if n != number] + if others: + out += ["The other change sets, whose files are not yours:"] + [f"- {o[:140]}" for o in others] + [""] + else: + out += lines + ([""] if lines and lines[-1].strip() else []) + digest = notes_digest(plan_dir / "implementation-notes.md") + if digest: + out += ["## What earlier change sets did (from implementation-notes.md)", ""] + out += [("#" + line) if line.startswith("## ") else line for line in digest] + return "\n".join(out).rstrip() + "\n" + + +def main(argv: list[str]) -> int: + args: list[str] = [] + out_dir = None + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--out-dir" and rest: + out_dir = Path(rest.pop(0)) + else: + args.append(value) + numbers = args[1:] + if not numbers or not all(n.isdigit() for n in numbers) or (out_dir is None and len(numbers) != 1): + print( + "usage: change-set-brief.py [ ...] [--out-dir DIR]\n" + " several change sets need --out-dir", + file=sys.stderr, + ) + return 2 + plan_dir = Path(args[0]) + spec = plan_dir / "spec.md" + if not spec.is_file(): + print(f"no spec.md at {spec}", file=sys.stderr) + return 2 + try: + briefs = {int(n): brief(plan_dir, int(n)) for n in numbers} + except LookupError as error: + print(error, file=sys.stderr) + return 2 + if out_dir is None: + sys.stdout.write(briefs[int(numbers[0])]) + return 0 + out_dir.mkdir(parents=True, exist_ok=True) + spec_bytes = len(spec.read_bytes()) + for number, text in briefs.items(): + target = out_dir / f"change-set-{number}.md" + target.write_text(text, encoding="utf-8") + size = len(text.encode("utf-8")) + print(f"{target.resolve()}: {size} bytes, {round(100 * size / spec_bytes)}% of spec.md") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/factory/phases/build/scripts/check-tests.py b/factory/phases/build/scripts/check-tests.py index 7d90d25..ccbb1aa 100755 --- a/factory/phases/build/scripts/check-tests.py +++ b/factory/phases/build/scripts/check-tests.py @@ -18,6 +18,8 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) NOTE_TESTS = re.compile(r"^\s*-\s*Tests added:\s*(.*)$", re.IGNORECASE) +# A comma separates two references only when a path::name follows it; a test name may hold commas. +NOTE_SEPARATOR = re.compile(r",\s*(?=[^,]*::)") def spec_scenarios(spec: Path, problem) -> dict[int, int]: @@ -65,7 +67,7 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in value.split(",") if t.strip()) + entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) return entries diff --git a/factory/phases/scope-review/references/lenses.md b/factory/phases/scope-review/references/lenses.md index 0e5fa5f..ed4f6d9 100644 --- a/factory/phases/scope-review/references/lenses.md +++ b/factory/phases/scope-review/references/lenses.md @@ -11,7 +11,8 @@ Does the change plan survive contact with the repo? - Every file a change set names exists, or the set says it is new; the described edit is possible at that site - the function, hook, or config it assumes is really there. - Prior art and idioms the spec cites exist where it says they do. - Premises about current behavior are checked against the code, never trusted: an "X already handles Y" claim that is false is a BLOCK naming the site. -- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on. +- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on - and nothing slower: an e2e suite, benchmark, or coverage re-run listed there without its `(end of build)` mark is a CONCERN, because build pays for the block every wave. +- The plan is as wide as the change allows: change sets queued behind one shared file that a single change set could own are a CONCERN naming the file (`lint-spec.py` prints the waves). ## completeness diff --git a/factory/phases/scope/SKILL.md b/factory/phases/scope/SKILL.md index c14c8c0..523ea3a 100644 --- a/factory/phases/scope/SKILL.md +++ b/factory/phases/scope/SKILL.md @@ -90,6 +90,7 @@ Name the components and flows the change adds, removes, or reshapes, in the over Efforts have second-order effects - capture them as nested sub-efforts, each carrying its own decisions back into the research section (rate limiting in scope means Redis setup, which carries config and deploy decisions). Record considered non-goals as `⊘` lines with a because clause - things someone weighed and cut, not mere omissions. End the scope with a `### Validation` block listing the repo's real typecheck/test/lint/build commands, discovered from `package.json`, a `Makefile`, CI config, or equivalent - never guess `npm test` into a `pytest` repo; ask if you cannot determine them. +`build` runs this block once per wave, so list each check once and mark every command a wave does not need - the e2e suite, a benchmark, a coverage or complexity run that repeats the suite - `(end of build)`: build runs those once, on the final tree. Writing style for the spec: ELI12, no similes or metaphors. ## 5. Review and research @@ -104,7 +105,9 @@ Writing style for the spec: ELI12, no similes or metaphors. Short fragmented sentences. Link decisions by ID wherever one applies, echoing the choice. Each change set ends with one `;`-separated `tests:` line - concrete scenarios as input -> expected outcome, each tagged `[unit]`, `[integration]`, or `[e2e]`, covering happy path, edge cases, and failure paths; a set with nothing to test says `tests: none - {reason}`. Specific enough that whoever writes the tests invents nothing; the author tags layers here because a fresh implementation session can't recover that intent. -Order change sets so each builds only on the ones before it; keep file lists disjoint where possible - `build` parallelizes consecutive change sets whose files don't overlap. +Order change sets so each builds only on the ones before it, and shape the plan wide: `build` runs change sets in parallel only while their file lists are disjoint, so a file three change sets edit makes them queue. +Land what several change sets share - a port, a schema, a registry, the composition root, a regenerated artifact - in one change set, and let the ones building on it own disjoint files. +Size each change set for one agent: past 25 scenarios it is several change sets, split along a seam. ``` 1. Change set 1 @@ -119,6 +122,7 @@ Order change sets so each builds only on the ones before it; keep file lists dis Then loop `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` until it exits clean. It owns the mechanics above; the spec is not final while it reports anything. +Once clean it prints the build waves the file lists allow and the files that make a change set wait; where change sets queue behind a shared file, move that file's edits into one change set and lint again. ## 7. Visualize diff --git a/factory/phases/scope/scripts/lint-spec.py b/factory/phases/scope/scripts/lint-spec.py index c45fdde..2cb73b3 100755 --- a/factory/phases/scope/scripts/lint-spec.py +++ b/factory/phases/scope/scripts/lint-spec.py @@ -3,6 +3,8 @@ Usage: python3 lint-spec.py .dev/{plan-name}/spec.md Exit 0 when the spec is clean, 1 with one problem per line otherwise. +A clean spec also gets the build waves its file lists allow, so the author sees +which shared files make change sets wait for each other. """ from __future__ import annotations @@ -20,8 +22,19 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +BACKTICKED = re.compile(r"`([^`\s]+)`") +# A backticked token is a file when it ends in a known file extension; +# `embedder.model` and `job.stages` are code, not paths. +FILE_NAME = re.compile( + r"(?:^|/)(?:Makefile|Dockerfile|[\w.@+-]+\." + r"(?:c|cc|cpp|cs|css|go|gradle|graphql|h|html|java|js|json|jsx|kt|lock|md|mjs|cjs|php|proto" + r"|py|rb|rs|scss|sh|sql|svelte|swift|tf|toml|ts|tsx|txt|vue|xml|yaml|yml))$" +) CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" +# One agent builds one change set; past this many scenarios it stops fitting in one sitting. +SCENARIO_LIMIT = 25 def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: @@ -122,7 +135,16 @@ def check_echoes( problem(number, f"the change plan links {slug}, which is still open or flagged") -def check_change_plan(body: list[tuple[int, str]], problem) -> None: +def built_change_sets(spec: Path) -> set[int]: + """Change sets build already logged; their size is history, and they never renumber.""" + notes = spec.parent / "implementation-notes.md" + if not notes.is_file(): + return set() + entries = (NOTE_ENTRY.match(line) for line in notes.read_text(encoding="utf-8").splitlines()) + return {int(entry.group(1)) for entry in entries if entry} + + +def check_change_plan(body: list[tuple[int, str]], built: set[int], problem) -> None: counts: dict[int, int] = {} current = None for number, line in body: @@ -148,15 +170,87 @@ def check_change_plan(body: list[tuple[int, str]], problem) -> None: if not value: problem(number, "tests: line is empty") continue - for scenario in value.split(";"): - scenario = scenario.strip() - if scenario and not LAYER.match(scenario): + scenarios = [scenario.strip() for scenario in value.split(";") if scenario.strip()] + for scenario in scenarios: + if not LAYER.match(scenario): problem(number, f"scenario '{scenario[:40]}' carries no [unit]/[integration]/[e2e] tag") + if len(scenarios) > SCENARIO_LIMIT and current not in built: + problem( + number, + f"change set {current} carries {len(scenarios)} scenarios; split it along a seam " + f"so no change set carries more than {SCENARIO_LIMIT}", + ) for change_set, seen in sorted(counts.items()): if seen != 1: problem(None, f"change set {change_set} has {seen} tests: lines; expected exactly one") +def change_set_files(body: list[tuple[int, str]]) -> dict[int, set[str]]: + """Files each change set names, read from the backticked paths outside its tests: line.""" + files: dict[int, set[str]] = {} + current = None + for _, line in body: + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + files.setdefault(current, set()) + if current is None or TESTS.match(line): + continue + files[current].update(t for t in BACKTICKED.findall(line) if FILE_NAME.search(t)) + return files + + +def shared_files(one: set[str], other: set[str]) -> list[str]: + """Paths two change sets both name; a bare file name matches any path that ends in it.""" + shared = one & other + for mine, theirs in ((one, other), (other, one)): + bare = {path for path in theirs if "/" not in path} + shared |= {path for path in mine if "/" in path and path.rsplit("/", 1)[-1] in bare} + return sorted(shared) + + +def build_waves(files: dict[int, set[str]]) -> tuple[list[list[int]], dict[int, tuple[int, list[str]]]]: + """The waves build's consecutive-disjoint batching gives when only file lists decide. + + A change set joins the current wave when it shares no file with the wave or with + any earlier change set still held back. Also returns, per change set that had to + wait, the change set it waited on last and the files they share. + """ + waves: list[list[int]] = [] + waited: dict[int, tuple[int, list[str]]] = {} + remaining = sorted(files) + while remaining: + wave: list[int] = [] + held: list[int] = [] + for change_set in remaining: + for earlier in wave + held: + shared = shared_files(files[change_set], files[earlier]) + if shared: + waited[change_set] = (earlier, shared) + held.append(change_set) + break + else: + wave.append(change_set) + waves.append(wave) + remaining = held + return waves, waited + + +def wave_report(body: list[tuple[int, str]], built: set[int]) -> list[str]: + files = {number: paths for number, paths in change_set_files(body).items() if number not in built} + if len(files) < 2: + return [] + waves, waited = build_waves(files) + layout = " ".join("[" + " ".join(str(n) for n in wave) + "]" for wave in waves) + lines = [ + f"build waves by file lists alone: {len(files)} change sets in {len(waves)} waves - {layout}" + ] + for change_set, (other, shared) in sorted(waited.items()): + more = f" (+{len(shared) - 3} more)" if len(shared) > 3 else "" + lines.append(f" {change_set} waits on {other}: {', '.join(shared[:3])}{more}") + return lines + + def check_validation(lines: list[str], problem) -> None: for index, line in enumerate(lines): if line.strip().lower() == "### validation": @@ -192,7 +286,8 @@ def problem(number: int | None, message: str) -> None: check_decisions(decisions, problem) check_echoes(found.get("scope", []), decisions, False, problem) check_echoes(found.get("change plan", []), decisions, True, problem) - check_change_plan(found.get("change plan", []), problem) + built = built_change_sets(path) + check_change_plan(found.get("change plan", []), built, problem) check_validation(lines, problem) for _, message in sorted(problems): @@ -201,6 +296,8 @@ def problem(number: int | None, message: str) -> None: print(f"\n{len(problems)} problem(s); the spec is not final.") return 1 print(f"{path}: clean - {len(decisions)} decisions argued.") + for line in wave_report(found.get("change plan", []), built): + print(line) return 0 diff --git a/factory/references/ci-parity.md b/factory/references/ci-parity.md index 382ac06..92157c9 100644 --- a/factory/references/ci-parity.md +++ b/factory/references/ci-parity.md @@ -21,6 +21,8 @@ Record the commands and outcomes in the plan's `implementation-notes.md`. Run every reproducible project-owned required-check command against the final checkout, after feature E2E and after any hardening edits. +A command that already ran green on this exact tree - no file changed since - +is not run a second time: record its result and where it came from. - A failure is work to fix, including a test described as flaky, unrelated, or pre-existing. Diagnose and remove its nondeterminism; do not add retries, diff --git a/package.json b/package.json index 39d5398..628c1b3 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@tobrun/dev-workflow", - "version": "3.2.0", + "version": "3.3.0", "description": "Test-focused development workflow skills for Claude Code, Codex, and Pi.", "keywords": [ "pi-package", diff --git a/plugins/dev/.codex-plugin/plugin.json b/plugins/dev/.codex-plugin/plugin.json index 8201b0a..d8c10dc 100644 --- a/plugins/dev/.codex-plugin/plugin.json +++ b/plugins/dev/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "dev", - "version": "3.2.0", + "version": "3.3.0", "description": "Development workflow skills: scope changes with argued decisions, build across unit/integration/e2e with every scenario proven by tests, ship with a deterministic quality gauntlet and an adversarially verified review, create structured commits that feed a decision ledger, and render pitches or comprehension quizzes.", "author": { "name": "Tobrun" diff --git a/plugins/dev/references/ci-parity.md b/plugins/dev/references/ci-parity.md index 382ac06..92157c9 100644 --- a/plugins/dev/references/ci-parity.md +++ b/plugins/dev/references/ci-parity.md @@ -21,6 +21,8 @@ Record the commands and outcomes in the plan's `implementation-notes.md`. Run every reproducible project-owned required-check command against the final checkout, after feature E2E and after any hardening edits. +A command that already ran green on this exact tree - no file changed since - +is not run a second time: record its result and where it came from. - A failure is work to fix, including a test described as flaky, unrelated, or pre-existing. Diagnose and remove its nondeterminism; do not add retries, diff --git a/plugins/dev/skills/build/SKILL.md b/plugins/dev/skills/build/SKILL.md index 4659ac3..4cb02f4 100644 --- a/plugins/dev/skills/build/SKILL.md +++ b/plugins/dev/skills/build/SKILL.md @@ -16,11 +16,11 @@ Read [references/layers.md](references/layers.md), [references/tests.md](referen 1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. 2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. -3. For each wave, run its change sets in parallel per the same reference, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). +3. For each wave, run its change sets in parallel per the same reference, pass its wave gate once, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). 4. Move straight to the next wave. Never stop after one change set or wave to ask about review. 5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. 6. Then run the full e2e pass per "The e2e layer" below over the whole spec, and loop on failures until it is green. -7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. +7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), plus every Validation command the wave gates left for the end, starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. 8. After the e2e report and CI-parity gate, close per "Closing message". Build never pushes or opens a PR; that is `ship`'s phase 3. ## Jira sync @@ -57,12 +57,12 @@ Neither does a failed e2e loop: say what is blocked, then still point at `ship`. Each change set, whether you run it yourself or a subagent runs it, follows the same loop: - Test at the seams the spec's scope section declares, per [references/tests.md](references/tests.md); if the declared boundary is wrong or missing, follow its fallback and log the change under Deviations - do not stall on it. -- Implement in **vertical slices**: one scenario's behavior at a time, its test written before or right after the code - the enforced outcome is what matters, not the ritual order. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. -- Run the change set's own tests and typecheck continuously; once the change set is green, run the spec's Validation block verbatim - it is the wider suite plus typecheck/lint - and only report done when it passes clean. +- Implement in **vertical slices**: one scenario's behavior at a time, its test written before the code or right after it - before is the cheap order, because its first run is then the red the rules below ask for. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. +- Run the change set's own tests and typecheck continuously - the test files it adds or edits, never the whole suite - and report done when they are green. The spec's Validation block is not run per change set: it is the wave gate, run once per wave before its commits, per "The wave gate" in [references/parallel.md](references/parallel.md). ## Rules of the loop -- **Every test must have been seen red.** A test that has never failed proves nothing: earn its green by writing it before the code, or by briefly breaking the behavior once after. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. +- **Every test must have been seen red, once and cheaply.** A test that has never failed proves nothing. Written before the code, its first run is that red and costs nothing. Written after, break the behavior once per slice - one break covers every test of the slice - and run only the affected test file: never one break per test, never the suite for a red. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. - **One slice at a time.** One seam, one behavior, one test, one minimal implementation per cycle. - **Refactoring is not part of the loop.** It belongs to `ship`'s review phase. - **Keep going.** A red test, a failing e2e scenario, or an edge case that contradicts the spec is work to do, not a reason to hand back. Fix it, log the deviation, continue. Stop early only when a blocking question makes further work unsafe or wasted. diff --git a/plugins/dev/skills/build/references/parallel.md b/plugins/dev/skills/build/references/parallel.md index d3b6a05..3aac303 100644 --- a/plugins/dev/skills/build/references/parallel.md +++ b/plugins/dev/skills/build/references/parallel.md @@ -20,17 +20,27 @@ Never start work belonging to the next wave while the current wave is in flight. Launch one `Agent` per change set, **all in a single message** so they run concurrently. A wave of one change set needs no subagent: implement it yourself in the main thread. +First write the wave's briefs, one command for the whole wave: + +```bash +python3 {build-skill-root}/scripts/change-set-brief.py .dev/{plan-name} {N} {N} --out-dir /tmp/{project-slug}/briefs/{plan-name} +``` + +A brief is the spec cut down to one change set: the scope section, the decisions that change set links, its own plan, and what earlier change sets did and deviated on. +Every line is the spec's own, so an agent starts from about a third of the reading and loses nothing it builds against. +Write them per wave, never once up front: the notes they carry grow with every committed change set. + Each agent prompt contains: 1. The role: `You implement exactly one change set of a spec. Other agents implement sibling change sets concurrently; stay inside your change set's file list.` -2. The absolute path to `spec.md`, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The spec is a self-contained handoff by design; the agent reads all of it, then implements only its own change set. +2. The absolute path to the change set's brief, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The brief stands in for `spec.md` and `implementation-notes.md`: the agent reads neither whole, and opens the spec only to follow something the brief points at. 3. The absolute path to this skill's `SKILL.md`, with the instruction to follow its "The change-set loop" and "Rules of the loop" sections - read like the other references, not pasted into the prompt. 4. Hard constraints: - Implement only this change set's `[unit]` and `[integration]` scenarios. `[e2e]` scenarios are run once per spec by the orchestrator afterwards - do not launch the app. - Do not commit, stage, or touch git state. The orchestrator commits. - Do not edit files outside your change set's file list. If the change set genuinely needs a file another change set owns, stop and report it as a conflict instead of editing it. - Do not edit `implementation-notes.md`. Report your entry; the orchestrator appends it. - - Run the spec's Validation block and report its real result. A red result is a fact to report, not something to hide or paper over. + - Run only your own checks: the test files this change set adds or edits, plus typecheck and lint over the packages it touches. Never run the spec's Validation block, the whole suite, the e2e suite, or a benchmark - siblings are mid-edit in the same tree, so a wider run measures their half-finished work, and the orchestrator runs the Validation block once for the whole wave. Report the commands and their real result. A red result is a fact to report, not something to hide or paper over. 5. The output contract below. Output contract (the agent's final message must be exactly one fenced JSON block): @@ -50,13 +60,22 @@ Output contract (the agent's final message must be exactly one fenced JSON block ## After a wave -1. Verify rather than trust the reports: run the spec's Validation block yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. +1. Verify rather than trust the reports: run the wave gate below yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. 2. Commit each change set's work on the current branch, in number order, one commit per change set. A skipped-ahead change set's commit waits until every earlier change set is committed, so history keeps the spec's order. 3. Append each change set's entry to `implementation-notes.md` from `whatWasDone`, `seamsTested`, `testsAdded`, and `deviations`. `testsAdded` becomes the entry's `Tests added:` line, which the scenario checker reads. A `status: blocked` change set, a failing validation, or a reported conflict is yours to finish in the main thread before the next wave starts - do not carry a red change set forward and do not relaunch the same agent on the same failure more than once. If two change sets in a wave edited the same file anyway, reconcile it yourself and log it under Deviations. +## The wave gate + +The spec's Validation block is the gate between a wave and its commits, and the only full run a wave gets - a wave of one you implemented yourself included. +Each of its commands runs once per tree: a green result stands until a file changes, so nothing is re-run "to be sure" before committing, and after a fix the failed command runs first and the rest only once it is green. +Start commands that share no state together (lint, typecheck, and the unit suite), and write the wave's notes entries while they run. + +A command the block marks `(end of build)` is left out of the wave gate, and so are three kinds even when a spec lists one unmarked, because each costs minutes and tells a wave nothing it needs to commit: the e2e suite (the e2e pass runs it), benchmarks, and any analysis that re-runs the suite to measure it - coverage, complexity, mutation. +Each of those runs once, on the final tree, with the CI-parity gate. + ## When not to parallelize - The spec has one change set, or every change set's files overlap with the one before it: run them sequentially yourself. diff --git a/plugins/dev/skills/build/scripts/change-set-brief.py b/plugins/dev/skills/build/scripts/change-set-brief.py new file mode 100755 index 0000000..a44b2ba --- /dev/null +++ b/plugins/dev/skills/build/scripts/change-set-brief.py @@ -0,0 +1,169 @@ +#!/usr/bin/env python3 +"""Cut a spec down to what one change set's implementer needs. + +Usage: python3 change-set-brief.py .dev/{plan-name} N [N ...] [--out-dir DIR] + +A brief is spec.md minus the decisions the change set does not link and minus +the other change sets, followed by what earlier change sets did and deviated on +from implementation-notes.md. Every kept line is verbatim, so an agent reading +the brief reads the spec's own words, a third of them. + +With --out-dir, writes DIR/change-set-N.md per change set and prints each path +with its size against the spec. Without it, prints the one brief to stdout. +Exit 2 on a missing spec or an unknown change set. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +DECISION = re.compile(r"^(D-[a-z0-9]+(?:-[a-z0-9]+)*):") +LINKED = re.compile(r"\bD-[a-z0-9]+(?:-[a-z0-9]+)*") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +# The two note lines only the scenario checker and the reviewers read. +NOTE_NOISE = re.compile(r"^\s*-\s*(Tests added|Seams tested):", re.IGNORECASE) + + +def split_sections(lines: list[str]) -> list[tuple[str, list[str]]]: + """The spec as (lower-cased ## heading, lines) pairs; the text before the first is ''.""" + sections: list[tuple[str, list[str]]] = [("", [])] + for line in lines: + heading = HEADING.match(line) + if heading: + sections.append((heading.group(1).lower(), [line])) + else: + sections[-1][1].append(line) + return sections + + +def decision_entries(research: list[str]) -> dict[str, list[str]]: + """Each decision's header and its indented alternative lines, by slug.""" + entries: dict[str, list[str]] = {} + current = None + for line in research[1:]: + header = DECISION.match(line) + if header: + current = entries.setdefault(header.group(1), []) + current.append(line) + elif current is not None and line[:1] in (" ", "\t") and line.strip(): + current.append(line) + elif line.strip(): + current = None + return entries + + +def change_set_blocks(plan: list[str]) -> dict[int, list[str]]: + """Each change set's lines, from its numbered line to the next one.""" + blocks: dict[int, list[str]] = {} + current = None + for line in plan[1:]: + change_set = CHANGE_SET.match(line) + if change_set: + current = blocks.setdefault(int(change_set.group(1)), []) + if current is not None: + current.append(line) + for block in blocks.values(): + while block and not block[-1].strip(): + block.pop() + return blocks + + +def notes_digest(notes: Path) -> list[str]: + """implementation-notes.md without its test and seam inventories.""" + if not notes.is_file(): + return [] + kept = [ + line for line in notes.read_text(encoding="utf-8").splitlines() + if not NOTE_NOISE.match(line) and not line.startswith("# ") + ] + return kept if any(NOTE_ENTRY.match(line) for line in kept) else [] + + +def brief(plan_dir: Path, number: int) -> str: + spec = plan_dir / "spec.md" + sections = split_sections(spec.read_text(encoding="utf-8").splitlines()) + by_name = dict(sections) + blocks = change_set_blocks(by_name.get("change plan", [""])) + if number not in blocks: + known = ", ".join(str(n) for n in sorted(blocks)) or "none" + raise LookupError(f"{spec} has no change set {number} (change sets: {known})") + block = blocks[number] + decisions = decision_entries(by_name.get("research", [""])) + linked = list(dict.fromkeys(slug for line in block for slug in LINKED.findall(line))) + + out = [ + f"# Brief: change set {number} of {plan_dir.name}", + "", + f"This is {spec.resolve()} cut down to change set {number}: every line below is the spec's own.", + "Left out are the decisions this change set does not link and the other change sets' plans.", + "Open the spec only to follow something this brief points at, never to read it whole.", + "", + ] + for name, lines in sections: + if name == "research": + out += ["## Research (the decisions this change set links)", ""] + for slug in linked: + out += decisions.get(slug, [f"{slug}: not argued in the spec's research section"]) + [""] + if not linked: + out += ["This change set links no decision.", ""] + elif name == "change plan": + out += [f"## Change plan (change set {number} only)", ""] + block + [""] + others = [blocks[n][0].strip() for n in sorted(blocks) if n != number] + if others: + out += ["The other change sets, whose files are not yours:"] + [f"- {o[:140]}" for o in others] + [""] + else: + out += lines + ([""] if lines and lines[-1].strip() else []) + digest = notes_digest(plan_dir / "implementation-notes.md") + if digest: + out += ["## What earlier change sets did (from implementation-notes.md)", ""] + out += [("#" + line) if line.startswith("## ") else line for line in digest] + return "\n".join(out).rstrip() + "\n" + + +def main(argv: list[str]) -> int: + args: list[str] = [] + out_dir = None + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--out-dir" and rest: + out_dir = Path(rest.pop(0)) + else: + args.append(value) + numbers = args[1:] + if not numbers or not all(n.isdigit() for n in numbers) or (out_dir is None and len(numbers) != 1): + print( + "usage: change-set-brief.py [ ...] [--out-dir DIR]\n" + " several change sets need --out-dir", + file=sys.stderr, + ) + return 2 + plan_dir = Path(args[0]) + spec = plan_dir / "spec.md" + if not spec.is_file(): + print(f"no spec.md at {spec}", file=sys.stderr) + return 2 + try: + briefs = {int(n): brief(plan_dir, int(n)) for n in numbers} + except LookupError as error: + print(error, file=sys.stderr) + return 2 + if out_dir is None: + sys.stdout.write(briefs[int(numbers[0])]) + return 0 + out_dir.mkdir(parents=True, exist_ok=True) + spec_bytes = len(spec.read_bytes()) + for number, text in briefs.items(): + target = out_dir / f"change-set-{number}.md" + target.write_text(text, encoding="utf-8") + size = len(text.encode("utf-8")) + print(f"{target.resolve()}: {size} bytes, {round(100 * size / spec_bytes)}% of spec.md") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/plugins/dev/skills/build/scripts/check-tests.py b/plugins/dev/skills/build/scripts/check-tests.py index 7d90d25..ccbb1aa 100755 --- a/plugins/dev/skills/build/scripts/check-tests.py +++ b/plugins/dev/skills/build/scripts/check-tests.py @@ -18,6 +18,8 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) NOTE_TESTS = re.compile(r"^\s*-\s*Tests added:\s*(.*)$", re.IGNORECASE) +# A comma separates two references only when a path::name follows it; a test name may hold commas. +NOTE_SEPARATOR = re.compile(r",\s*(?=[^,]*::)") def spec_scenarios(spec: Path, problem) -> dict[int, int]: @@ -65,7 +67,7 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in value.split(",") if t.strip()) + entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) return entries diff --git a/plugins/dev/skills/scope-review/references/lenses.md b/plugins/dev/skills/scope-review/references/lenses.md index 0e5fa5f..ed4f6d9 100644 --- a/plugins/dev/skills/scope-review/references/lenses.md +++ b/plugins/dev/skills/scope-review/references/lenses.md @@ -11,7 +11,8 @@ Does the change plan survive contact with the repo? - Every file a change set names exists, or the set says it is new; the described edit is possible at that site - the function, hook, or config it assumes is really there. - Prior art and idioms the spec cites exist where it says they do. - Premises about current behavior are checked against the code, never trusted: an "X already handles Y" claim that is false is a BLOCK naming the site. -- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on. +- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on - and nothing slower: an e2e suite, benchmark, or coverage re-run listed there without its `(end of build)` mark is a CONCERN, because build pays for the block every wave. +- The plan is as wide as the change allows: change sets queued behind one shared file that a single change set could own are a CONCERN naming the file (`lint-spec.py` prints the waves). ## completeness diff --git a/plugins/dev/skills/scope/SKILL.md b/plugins/dev/skills/scope/SKILL.md index f621d9c..8767cde 100644 --- a/plugins/dev/skills/scope/SKILL.md +++ b/plugins/dev/skills/scope/SKILL.md @@ -89,6 +89,7 @@ Name the components and flows the change adds, removes, or reshapes, in the over Efforts have second-order effects - capture them as nested sub-efforts, each carrying its own decisions back into the research section (rate limiting in scope means Redis setup, which carries config and deploy decisions). Record considered non-goals as `⊘` lines with a because clause - things someone weighed and cut, not mere omissions. End the scope with a `### Validation` block listing the repo's real typecheck/test/lint/build commands, discovered from `package.json`, a `Makefile`, CI config, or equivalent - never guess `npm test` into a `pytest` repo; ask if you cannot determine them. +`build` runs this block once per wave, so list each check once and mark every command a wave does not need - the e2e suite, a benchmark, a coverage or complexity run that repeats the suite - `(end of build)`: build runs those once, on the final tree. Writing style for the spec: ELI12, no similes or metaphors. ## 5. Review and research @@ -103,7 +104,9 @@ Writing style for the spec: ELI12, no similes or metaphors. Short fragmented sentences. Link decisions by ID wherever one applies, echoing the choice. Each change set ends with one `;`-separated `tests:` line - concrete scenarios as input -> expected outcome, each tagged `[unit]`, `[integration]`, or `[e2e]`, covering happy path, edge cases, and failure paths; a set with nothing to test says `tests: none - {reason}`. Specific enough that whoever writes the tests invents nothing; the author tags layers here because a fresh implementation session can't recover that intent. -Order change sets so each builds only on the ones before it; keep file lists disjoint where possible - `build` parallelizes consecutive change sets whose files don't overlap. +Order change sets so each builds only on the ones before it, and shape the plan wide: `build` runs change sets in parallel only while their file lists are disjoint, so a file three change sets edit makes them queue. +Land what several change sets share - a port, a schema, a registry, the composition root, a regenerated artifact - in one change set, and let the ones building on it own disjoint files. +Size each change set for one agent: past 25 scenarios it is several change sets, split along a seam. ``` 1. Change set 1 @@ -118,6 +121,7 @@ Order change sets so each builds only on the ones before it; keep file lists dis Then loop `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` until it exits clean. It owns the mechanics above; the spec is not final while it reports anything. +Once clean it prints the build waves the file lists allow and the files that make a change set wait; where change sets queue behind a shared file, move that file's edits into one change set and lint again. ## 7. Visualize diff --git a/plugins/dev/skills/scope/scripts/lint-spec.py b/plugins/dev/skills/scope/scripts/lint-spec.py index c45fdde..2cb73b3 100755 --- a/plugins/dev/skills/scope/scripts/lint-spec.py +++ b/plugins/dev/skills/scope/scripts/lint-spec.py @@ -3,6 +3,8 @@ Usage: python3 lint-spec.py .dev/{plan-name}/spec.md Exit 0 when the spec is clean, 1 with one problem per line otherwise. +A clean spec also gets the build waves its file lists allow, so the author sees +which shared files make change sets wait for each other. """ from __future__ import annotations @@ -20,8 +22,19 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +BACKTICKED = re.compile(r"`([^`\s]+)`") +# A backticked token is a file when it ends in a known file extension; +# `embedder.model` and `job.stages` are code, not paths. +FILE_NAME = re.compile( + r"(?:^|/)(?:Makefile|Dockerfile|[\w.@+-]+\." + r"(?:c|cc|cpp|cs|css|go|gradle|graphql|h|html|java|js|json|jsx|kt|lock|md|mjs|cjs|php|proto" + r"|py|rb|rs|scss|sh|sql|svelte|swift|tf|toml|ts|tsx|txt|vue|xml|yaml|yml))$" +) CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" +# One agent builds one change set; past this many scenarios it stops fitting in one sitting. +SCENARIO_LIMIT = 25 def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: @@ -122,7 +135,16 @@ def check_echoes( problem(number, f"the change plan links {slug}, which is still open or flagged") -def check_change_plan(body: list[tuple[int, str]], problem) -> None: +def built_change_sets(spec: Path) -> set[int]: + """Change sets build already logged; their size is history, and they never renumber.""" + notes = spec.parent / "implementation-notes.md" + if not notes.is_file(): + return set() + entries = (NOTE_ENTRY.match(line) for line in notes.read_text(encoding="utf-8").splitlines()) + return {int(entry.group(1)) for entry in entries if entry} + + +def check_change_plan(body: list[tuple[int, str]], built: set[int], problem) -> None: counts: dict[int, int] = {} current = None for number, line in body: @@ -148,15 +170,87 @@ def check_change_plan(body: list[tuple[int, str]], problem) -> None: if not value: problem(number, "tests: line is empty") continue - for scenario in value.split(";"): - scenario = scenario.strip() - if scenario and not LAYER.match(scenario): + scenarios = [scenario.strip() for scenario in value.split(";") if scenario.strip()] + for scenario in scenarios: + if not LAYER.match(scenario): problem(number, f"scenario '{scenario[:40]}' carries no [unit]/[integration]/[e2e] tag") + if len(scenarios) > SCENARIO_LIMIT and current not in built: + problem( + number, + f"change set {current} carries {len(scenarios)} scenarios; split it along a seam " + f"so no change set carries more than {SCENARIO_LIMIT}", + ) for change_set, seen in sorted(counts.items()): if seen != 1: problem(None, f"change set {change_set} has {seen} tests: lines; expected exactly one") +def change_set_files(body: list[tuple[int, str]]) -> dict[int, set[str]]: + """Files each change set names, read from the backticked paths outside its tests: line.""" + files: dict[int, set[str]] = {} + current = None + for _, line in body: + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + files.setdefault(current, set()) + if current is None or TESTS.match(line): + continue + files[current].update(t for t in BACKTICKED.findall(line) if FILE_NAME.search(t)) + return files + + +def shared_files(one: set[str], other: set[str]) -> list[str]: + """Paths two change sets both name; a bare file name matches any path that ends in it.""" + shared = one & other + for mine, theirs in ((one, other), (other, one)): + bare = {path for path in theirs if "/" not in path} + shared |= {path for path in mine if "/" in path and path.rsplit("/", 1)[-1] in bare} + return sorted(shared) + + +def build_waves(files: dict[int, set[str]]) -> tuple[list[list[int]], dict[int, tuple[int, list[str]]]]: + """The waves build's consecutive-disjoint batching gives when only file lists decide. + + A change set joins the current wave when it shares no file with the wave or with + any earlier change set still held back. Also returns, per change set that had to + wait, the change set it waited on last and the files they share. + """ + waves: list[list[int]] = [] + waited: dict[int, tuple[int, list[str]]] = {} + remaining = sorted(files) + while remaining: + wave: list[int] = [] + held: list[int] = [] + for change_set in remaining: + for earlier in wave + held: + shared = shared_files(files[change_set], files[earlier]) + if shared: + waited[change_set] = (earlier, shared) + held.append(change_set) + break + else: + wave.append(change_set) + waves.append(wave) + remaining = held + return waves, waited + + +def wave_report(body: list[tuple[int, str]], built: set[int]) -> list[str]: + files = {number: paths for number, paths in change_set_files(body).items() if number not in built} + if len(files) < 2: + return [] + waves, waited = build_waves(files) + layout = " ".join("[" + " ".join(str(n) for n in wave) + "]" for wave in waves) + lines = [ + f"build waves by file lists alone: {len(files)} change sets in {len(waves)} waves - {layout}" + ] + for change_set, (other, shared) in sorted(waited.items()): + more = f" (+{len(shared) - 3} more)" if len(shared) > 3 else "" + lines.append(f" {change_set} waits on {other}: {', '.join(shared[:3])}{more}") + return lines + + def check_validation(lines: list[str], problem) -> None: for index, line in enumerate(lines): if line.strip().lower() == "### validation": @@ -192,7 +286,8 @@ def problem(number: int | None, message: str) -> None: check_decisions(decisions, problem) check_echoes(found.get("scope", []), decisions, False, problem) check_echoes(found.get("change plan", []), decisions, True, problem) - check_change_plan(found.get("change plan", []), problem) + built = built_change_sets(path) + check_change_plan(found.get("change plan", []), built, problem) check_validation(lines, problem) for _, message in sorted(problems): @@ -201,6 +296,8 @@ def problem(number: int | None, message: str) -> None: print(f"\n{len(problems)} problem(s); the spec is not final.") return 1 print(f"{path}: clean - {len(decisions)} decisions argued.") + for line in wave_report(found.get("change plan", []), built): + print(line) return 0 diff --git a/plugins/factory/phases/build/SKILL.md b/plugins/factory/phases/build/SKILL.md index 45d2736..e88d955 100644 --- a/plugins/factory/phases/build/SKILL.md +++ b/plugins/factory/phases/build/SKILL.md @@ -16,11 +16,11 @@ Read [references/layers.md](references/layers.md), [references/tests.md](referen 1. Run `python3 {build-skill-root}/../../scripts/skill-metrics.py start build`, then read `spec.md` in full: the research section (the decisions and their rationale), the scope section (including its Validation block of real repo commands), and the change plan. Explore the relevant code. If the Validation block is absent, discover the repo's real test and typecheck commands yourself from `package.json`, a `Makefile`, or CI config, and log them in `implementation-notes.md`. 2. Build waves by disjoint batching per [references/parallel.md](references/parallel.md): sequential in spec order by default, batched only when file lists are disjoint and nothing a wave-mate or earlier unfinished change set introduces is consumed. Every change set in the plan is in scope, not just the first. -3. For each wave, run its change sets in parallel per the same reference, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). +3. For each wave, run its change sets in parallel per the same reference, pass its wave gate once, then commit each finished change set on the current branch and append its entry to `implementation-notes.md`. A change set that adds, removes, moves, or rewires a component, flow, or boundary updates `docs/architecture.md` in the same commit and passes `architecture-check.py` first, per [../../references/architecture.md](../../references/architecture.md). 4. Move straight to the next wave. Never stop after one change set or wave to ask about review. 5. When every change set is committed, loop `python3 {build-skill-root}/scripts/check-tests.py .dev/{plan-name}` until it exits clean: it proves every specced scenario has a test that really exists, rather than one that was reported. 6. Then run the full e2e pass per "The e2e layer" below over the whole spec, and loop on failures until it is green. -7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. +7. Run the repository's required pull-request commands per [../../references/ci-parity.md](../../references/ci-parity.md), plus every Validation command the wave gates left for the end, starting them in the background as soon as the e2e loop is green and rendering the e2e report while they run - the two share nothing. A known-red CI scenario is not an acceptable deviation. 8. After the e2e report and CI-parity gate, close per "Closing message". Build never pushes or opens a PR; that is `ship`'s phase 3. ## Jira sync @@ -57,12 +57,12 @@ Neither does a failed e2e loop: say what is blocked, then still point at `ship`. Each change set, whether you run it yourself or a subagent runs it, follows the same loop: - Test at the seams the spec's scope section declares, per [references/tests.md](references/tests.md); if the declared boundary is wrong or missing, follow its fallback and log the change under Deviations - do not stall on it. -- Implement in **vertical slices**: one scenario's behavior at a time, its test written before or right after the code - the enforced outcome is what matters, not the ritual order. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. -- Run the change set's own tests and typecheck continuously; once the change set is green, run the spec's Validation block verbatim - it is the wider suite plus typecheck/lint - and only report done when it passes clean. +- Implement in **vertical slices**: one scenario's behavior at a time, its test written before the code or right after it - before is the cheap order, because its first run is then the red the rules below ask for. Each `tests:` scenario's test lives at its tagged layer ([references/layers.md](references/layers.md)); a scenario isn't met until a real test exists there. +- Run the change set's own tests and typecheck continuously - the test files it adds or edits, never the whole suite - and report done when they are green. The spec's Validation block is not run per change set: it is the wave gate, run once per wave before its commits, per "The wave gate" in [references/parallel.md](references/parallel.md). ## Rules of the loop -- **Every test must have been seen red.** A test that has never failed proves nothing: earn its green by writing it before the code, or by briefly breaking the behavior once after. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. +- **Every test must have been seen red, once and cheaply.** A test that has never failed proves nothing. Written before the code, its first run is that red and costs nothing. Written after, break the behavior once per slice - one break covers every test of the slice - and run only the affected test file: never one break per test, never the suite for a red. Bug fixes are strictly test-first: a defect change set starts with a failing test that reproduces the reported issue - red is the proof it was actually reproduced - only then fix, and watch that same test go green. - **One slice at a time.** One seam, one behavior, one test, one minimal implementation per cycle. - **Refactoring is not part of the loop.** It belongs to `ship`'s review phase. - **Keep going.** A red test, a failing e2e scenario, or an edge case that contradicts the spec is work to do, not a reason to hand back. Fix it, log the deviation, continue. Stop early only when a blocking question makes further work unsafe or wasted. diff --git a/plugins/factory/phases/build/references/parallel.md b/plugins/factory/phases/build/references/parallel.md index d3b6a05..3aac303 100644 --- a/plugins/factory/phases/build/references/parallel.md +++ b/plugins/factory/phases/build/references/parallel.md @@ -20,17 +20,27 @@ Never start work belonging to the next wave while the current wave is in flight. Launch one `Agent` per change set, **all in a single message** so they run concurrently. A wave of one change set needs no subagent: implement it yourself in the main thread. +First write the wave's briefs, one command for the whole wave: + +```bash +python3 {build-skill-root}/scripts/change-set-brief.py .dev/{plan-name} {N} {N} --out-dir /tmp/{project-slug}/briefs/{plan-name} +``` + +A brief is the spec cut down to one change set: the scope section, the decisions that change set links, its own plan, and what earlier change sets did and deviated on. +Every line is the spec's own, so an agent starts from about a third of the reading and loses nothing it builds against. +Write them per wave, never once up front: the notes they carry grow with every committed change set. + Each agent prompt contains: 1. The role: `You implement exactly one change set of a spec. Other agents implement sibling change sets concurrently; stay inside your change set's file list.` -2. The absolute path to `spec.md`, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The spec is a self-contained handoff by design; the agent reads all of it, then implements only its own change set. +2. The absolute path to the change set's brief, the number of the change set the agent owns, and this skill's `layers.md`, `tests.md`, and `mocking.md` - the agent reads them itself rather than receiving them inlined. The brief stands in for `spec.md` and `implementation-notes.md`: the agent reads neither whole, and opens the spec only to follow something the brief points at. 3. The absolute path to this skill's `SKILL.md`, with the instruction to follow its "The change-set loop" and "Rules of the loop" sections - read like the other references, not pasted into the prompt. 4. Hard constraints: - Implement only this change set's `[unit]` and `[integration]` scenarios. `[e2e]` scenarios are run once per spec by the orchestrator afterwards - do not launch the app. - Do not commit, stage, or touch git state. The orchestrator commits. - Do not edit files outside your change set's file list. If the change set genuinely needs a file another change set owns, stop and report it as a conflict instead of editing it. - Do not edit `implementation-notes.md`. Report your entry; the orchestrator appends it. - - Run the spec's Validation block and report its real result. A red result is a fact to report, not something to hide or paper over. + - Run only your own checks: the test files this change set adds or edits, plus typecheck and lint over the packages it touches. Never run the spec's Validation block, the whole suite, the e2e suite, or a benchmark - siblings are mid-edit in the same tree, so a wider run measures their half-finished work, and the orchestrator runs the Validation block once for the whole wave. Report the commands and their real result. A red result is a fact to report, not something to hide or paper over. 5. The output contract below. Output contract (the agent's final message must be exactly one fenced JSON block): @@ -50,13 +60,22 @@ Output contract (the agent's final message must be exactly one fenced JSON block ## After a wave -1. Verify rather than trust the reports: run the spec's Validation block yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. +1. Verify rather than trust the reports: run the wave gate below yourself, once per wave - the shared tree already holds the whole wave's changes, so one run covers every change set in it. 2. Commit each change set's work on the current branch, in number order, one commit per change set. A skipped-ahead change set's commit waits until every earlier change set is committed, so history keeps the spec's order. 3. Append each change set's entry to `implementation-notes.md` from `whatWasDone`, `seamsTested`, `testsAdded`, and `deviations`. `testsAdded` becomes the entry's `Tests added:` line, which the scenario checker reads. A `status: blocked` change set, a failing validation, or a reported conflict is yours to finish in the main thread before the next wave starts - do not carry a red change set forward and do not relaunch the same agent on the same failure more than once. If two change sets in a wave edited the same file anyway, reconcile it yourself and log it under Deviations. +## The wave gate + +The spec's Validation block is the gate between a wave and its commits, and the only full run a wave gets - a wave of one you implemented yourself included. +Each of its commands runs once per tree: a green result stands until a file changes, so nothing is re-run "to be sure" before committing, and after a fix the failed command runs first and the rest only once it is green. +Start commands that share no state together (lint, typecheck, and the unit suite), and write the wave's notes entries while they run. + +A command the block marks `(end of build)` is left out of the wave gate, and so are three kinds even when a spec lists one unmarked, because each costs minutes and tells a wave nothing it needs to commit: the e2e suite (the e2e pass runs it), benchmarks, and any analysis that re-runs the suite to measure it - coverage, complexity, mutation. +Each of those runs once, on the final tree, with the CI-parity gate. + ## When not to parallelize - The spec has one change set, or every change set's files overlap with the one before it: run them sequentially yourself. diff --git a/plugins/factory/phases/build/scripts/change-set-brief.py b/plugins/factory/phases/build/scripts/change-set-brief.py new file mode 100755 index 0000000..a44b2ba --- /dev/null +++ b/plugins/factory/phases/build/scripts/change-set-brief.py @@ -0,0 +1,169 @@ +#!/usr/bin/env python3 +"""Cut a spec down to what one change set's implementer needs. + +Usage: python3 change-set-brief.py .dev/{plan-name} N [N ...] [--out-dir DIR] + +A brief is spec.md minus the decisions the change set does not link and minus +the other change sets, followed by what earlier change sets did and deviated on +from implementation-notes.md. Every kept line is verbatim, so an agent reading +the brief reads the spec's own words, a third of them. + +With --out-dir, writes DIR/change-set-N.md per change set and prints each path +with its size against the spec. Without it, prints the one brief to stdout. +Exit 2 on a missing spec or an unknown change set. +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +HEADING = re.compile(r"^##\s+(.*?)\s*$") +DECISION = re.compile(r"^(D-[a-z0-9]+(?:-[a-z0-9]+)*):") +LINKED = re.compile(r"\bD-[a-z0-9]+(?:-[a-z0-9]+)*") +CHANGE_SET = re.compile(r"^\s*(\d+)\.\s+\S") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +# The two note lines only the scenario checker and the reviewers read. +NOTE_NOISE = re.compile(r"^\s*-\s*(Tests added|Seams tested):", re.IGNORECASE) + + +def split_sections(lines: list[str]) -> list[tuple[str, list[str]]]: + """The spec as (lower-cased ## heading, lines) pairs; the text before the first is ''.""" + sections: list[tuple[str, list[str]]] = [("", [])] + for line in lines: + heading = HEADING.match(line) + if heading: + sections.append((heading.group(1).lower(), [line])) + else: + sections[-1][1].append(line) + return sections + + +def decision_entries(research: list[str]) -> dict[str, list[str]]: + """Each decision's header and its indented alternative lines, by slug.""" + entries: dict[str, list[str]] = {} + current = None + for line in research[1:]: + header = DECISION.match(line) + if header: + current = entries.setdefault(header.group(1), []) + current.append(line) + elif current is not None and line[:1] in (" ", "\t") and line.strip(): + current.append(line) + elif line.strip(): + current = None + return entries + + +def change_set_blocks(plan: list[str]) -> dict[int, list[str]]: + """Each change set's lines, from its numbered line to the next one.""" + blocks: dict[int, list[str]] = {} + current = None + for line in plan[1:]: + change_set = CHANGE_SET.match(line) + if change_set: + current = blocks.setdefault(int(change_set.group(1)), []) + if current is not None: + current.append(line) + for block in blocks.values(): + while block and not block[-1].strip(): + block.pop() + return blocks + + +def notes_digest(notes: Path) -> list[str]: + """implementation-notes.md without its test and seam inventories.""" + if not notes.is_file(): + return [] + kept = [ + line for line in notes.read_text(encoding="utf-8").splitlines() + if not NOTE_NOISE.match(line) and not line.startswith("# ") + ] + return kept if any(NOTE_ENTRY.match(line) for line in kept) else [] + + +def brief(plan_dir: Path, number: int) -> str: + spec = plan_dir / "spec.md" + sections = split_sections(spec.read_text(encoding="utf-8").splitlines()) + by_name = dict(sections) + blocks = change_set_blocks(by_name.get("change plan", [""])) + if number not in blocks: + known = ", ".join(str(n) for n in sorted(blocks)) or "none" + raise LookupError(f"{spec} has no change set {number} (change sets: {known})") + block = blocks[number] + decisions = decision_entries(by_name.get("research", [""])) + linked = list(dict.fromkeys(slug for line in block for slug in LINKED.findall(line))) + + out = [ + f"# Brief: change set {number} of {plan_dir.name}", + "", + f"This is {spec.resolve()} cut down to change set {number}: every line below is the spec's own.", + "Left out are the decisions this change set does not link and the other change sets' plans.", + "Open the spec only to follow something this brief points at, never to read it whole.", + "", + ] + for name, lines in sections: + if name == "research": + out += ["## Research (the decisions this change set links)", ""] + for slug in linked: + out += decisions.get(slug, [f"{slug}: not argued in the spec's research section"]) + [""] + if not linked: + out += ["This change set links no decision.", ""] + elif name == "change plan": + out += [f"## Change plan (change set {number} only)", ""] + block + [""] + others = [blocks[n][0].strip() for n in sorted(blocks) if n != number] + if others: + out += ["The other change sets, whose files are not yours:"] + [f"- {o[:140]}" for o in others] + [""] + else: + out += lines + ([""] if lines and lines[-1].strip() else []) + digest = notes_digest(plan_dir / "implementation-notes.md") + if digest: + out += ["## What earlier change sets did (from implementation-notes.md)", ""] + out += [("#" + line) if line.startswith("## ") else line for line in digest] + return "\n".join(out).rstrip() + "\n" + + +def main(argv: list[str]) -> int: + args: list[str] = [] + out_dir = None + rest = argv[1:] + while rest: + value = rest.pop(0) + if value == "--out-dir" and rest: + out_dir = Path(rest.pop(0)) + else: + args.append(value) + numbers = args[1:] + if not numbers or not all(n.isdigit() for n in numbers) or (out_dir is None and len(numbers) != 1): + print( + "usage: change-set-brief.py [ ...] [--out-dir DIR]\n" + " several change sets need --out-dir", + file=sys.stderr, + ) + return 2 + plan_dir = Path(args[0]) + spec = plan_dir / "spec.md" + if not spec.is_file(): + print(f"no spec.md at {spec}", file=sys.stderr) + return 2 + try: + briefs = {int(n): brief(plan_dir, int(n)) for n in numbers} + except LookupError as error: + print(error, file=sys.stderr) + return 2 + if out_dir is None: + sys.stdout.write(briefs[int(numbers[0])]) + return 0 + out_dir.mkdir(parents=True, exist_ok=True) + spec_bytes = len(spec.read_bytes()) + for number, text in briefs.items(): + target = out_dir / f"change-set-{number}.md" + target.write_text(text, encoding="utf-8") + size = len(text.encode("utf-8")) + print(f"{target.resolve()}: {size} bytes, {round(100 * size / spec_bytes)}% of spec.md") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv)) diff --git a/plugins/factory/phases/build/scripts/check-tests.py b/plugins/factory/phases/build/scripts/check-tests.py index 7d90d25..ccbb1aa 100755 --- a/plugins/factory/phases/build/scripts/check-tests.py +++ b/plugins/factory/phases/build/scripts/check-tests.py @@ -18,6 +18,8 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) NOTE_TESTS = re.compile(r"^\s*-\s*Tests added:\s*(.*)$", re.IGNORECASE) +# A comma separates two references only when a path::name follows it; a test name may hold commas. +NOTE_SEPARATOR = re.compile(r",\s*(?=[^,]*::)") def spec_scenarios(spec: Path, problem) -> dict[int, int]: @@ -65,7 +67,7 @@ def note_entries(notes: Path) -> dict[int, list[str]]: value = named.group(1).strip() if value.lower().startswith("none"): continue - entries[current].extend(t.strip() for t in value.split(",") if t.strip()) + entries[current].extend(t.strip() for t in NOTE_SEPARATOR.split(value) if t.strip()) return entries diff --git a/plugins/factory/phases/scope-review/references/lenses.md b/plugins/factory/phases/scope-review/references/lenses.md index 0e5fa5f..ed4f6d9 100644 --- a/plugins/factory/phases/scope-review/references/lenses.md +++ b/plugins/factory/phases/scope-review/references/lenses.md @@ -11,7 +11,8 @@ Does the change plan survive contact with the repo? - Every file a change set names exists, or the set says it is new; the described edit is possible at that site - the function, hook, or config it assumes is really there. - Prior art and idioms the spec cites exist where it says they do. - Premises about current behavior are checked against the code, never trusted: an "X already handles Y" claim that is false is a BLOCK naming the site. -- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on. +- The Validation block's commands exist in the repo's manifests and run the layers the plan relies on - and nothing slower: an e2e suite, benchmark, or coverage re-run listed there without its `(end of build)` mark is a CONCERN, because build pays for the block every wave. +- The plan is as wide as the change allows: change sets queued behind one shared file that a single change set could own are a CONCERN naming the file (`lint-spec.py` prints the waves). ## completeness diff --git a/plugins/factory/phases/scope/SKILL.md b/plugins/factory/phases/scope/SKILL.md index 4dc4446..3a3a3ee 100644 --- a/plugins/factory/phases/scope/SKILL.md +++ b/plugins/factory/phases/scope/SKILL.md @@ -89,6 +89,7 @@ Name the components and flows the change adds, removes, or reshapes, in the over Efforts have second-order effects - capture them as nested sub-efforts, each carrying its own decisions back into the research section (rate limiting in scope means Redis setup, which carries config and deploy decisions). Record considered non-goals as `⊘` lines with a because clause - things someone weighed and cut, not mere omissions. End the scope with a `### Validation` block listing the repo's real typecheck/test/lint/build commands, discovered from `package.json`, a `Makefile`, CI config, or equivalent - never guess `npm test` into a `pytest` repo; ask if you cannot determine them. +`build` runs this block once per wave, so list each check once and mark every command a wave does not need - the e2e suite, a benchmark, a coverage or complexity run that repeats the suite - `(end of build)`: build runs those once, on the final tree. Writing style for the spec: ELI12, no similes or metaphors. ## 5. Review and research @@ -103,7 +104,9 @@ Writing style for the spec: ELI12, no similes or metaphors. Short fragmented sentences. Link decisions by ID wherever one applies, echoing the choice. Each change set ends with one `;`-separated `tests:` line - concrete scenarios as input -> expected outcome, each tagged `[unit]`, `[integration]`, or `[e2e]`, covering happy path, edge cases, and failure paths; a set with nothing to test says `tests: none - {reason}`. Specific enough that whoever writes the tests invents nothing; the author tags layers here because a fresh implementation session can't recover that intent. -Order change sets so each builds only on the ones before it; keep file lists disjoint where possible - `build` parallelizes consecutive change sets whose files don't overlap. +Order change sets so each builds only on the ones before it, and shape the plan wide: `build` runs change sets in parallel only while their file lists are disjoint, so a file three change sets edit makes them queue. +Land what several change sets share - a port, a schema, a registry, the composition root, a regenerated artifact - in one change set, and let the ones building on it own disjoint files. +Size each change set for one agent: past 25 scenarios it is several change sets, split along a seam. ``` 1. Change set 1 @@ -118,6 +121,7 @@ Order change sets so each builds only on the ones before it; keep file lists dis Then loop `python3 {scope-skill-root}/scripts/lint-spec.py .dev/{plan-name}/spec.md` until it exits clean. It owns the mechanics above; the spec is not final while it reports anything. +Once clean it prints the build waves the file lists allow and the files that make a change set wait; where change sets queue behind a shared file, move that file's edits into one change set and lint again. ## 7. Visualize diff --git a/plugins/factory/phases/scope/scripts/lint-spec.py b/plugins/factory/phases/scope/scripts/lint-spec.py index c45fdde..2cb73b3 100755 --- a/plugins/factory/phases/scope/scripts/lint-spec.py +++ b/plugins/factory/phases/scope/scripts/lint-spec.py @@ -3,6 +3,8 @@ Usage: python3 lint-spec.py .dev/{plan-name}/spec.md Exit 0 when the spec is clean, 1 with one problem per line otherwise. +A clean spec also gets the build waves its file lists allow, so the author sees +which shared files make change sets wait for each other. """ from __future__ import annotations @@ -20,8 +22,19 @@ TESTS = re.compile(r"^\s*tests:\s*(.*)$", re.IGNORECASE) LAYER = re.compile(r"^\[(unit|integration|e2e)\]\s+\S") HEADING = re.compile(r"^##\s+(.*?)\s*$") +NOTE_ENTRY = re.compile(r"^##\s+Change set\s+(\d+)\b", re.IGNORECASE) +BACKTICKED = re.compile(r"`([^`\s]+)`") +# A backticked token is a file when it ends in a known file extension; +# `embedder.model` and `job.stages` are code, not paths. +FILE_NAME = re.compile( + r"(?:^|/)(?:Makefile|Dockerfile|[\w.@+-]+\." + r"(?:c|cc|cpp|cs|css|go|gradle|graphql|h|html|java|js|json|jsx|kt|lock|md|mjs|cjs|php|proto" + r"|py|rb|rs|scss|sh|sql|svelte|swift|tf|toml|ts|tsx|txt|vue|xml|yaml|yml))$" +) CHOSEN, OPEN, NOT_DOING = "✓", "?", "⊘" +# One agent builds one change set; past this many scenarios it stops fitting in one sitting. +SCENARIO_LIMIT = 25 def sections(lines: list[str]) -> dict[str, list[tuple[int, str]]]: @@ -122,7 +135,16 @@ def check_echoes( problem(number, f"the change plan links {slug}, which is still open or flagged") -def check_change_plan(body: list[tuple[int, str]], problem) -> None: +def built_change_sets(spec: Path) -> set[int]: + """Change sets build already logged; their size is history, and they never renumber.""" + notes = spec.parent / "implementation-notes.md" + if not notes.is_file(): + return set() + entries = (NOTE_ENTRY.match(line) for line in notes.read_text(encoding="utf-8").splitlines()) + return {int(entry.group(1)) for entry in entries if entry} + + +def check_change_plan(body: list[tuple[int, str]], built: set[int], problem) -> None: counts: dict[int, int] = {} current = None for number, line in body: @@ -148,15 +170,87 @@ def check_change_plan(body: list[tuple[int, str]], problem) -> None: if not value: problem(number, "tests: line is empty") continue - for scenario in value.split(";"): - scenario = scenario.strip() - if scenario and not LAYER.match(scenario): + scenarios = [scenario.strip() for scenario in value.split(";") if scenario.strip()] + for scenario in scenarios: + if not LAYER.match(scenario): problem(number, f"scenario '{scenario[:40]}' carries no [unit]/[integration]/[e2e] tag") + if len(scenarios) > SCENARIO_LIMIT and current not in built: + problem( + number, + f"change set {current} carries {len(scenarios)} scenarios; split it along a seam " + f"so no change set carries more than {SCENARIO_LIMIT}", + ) for change_set, seen in sorted(counts.items()): if seen != 1: problem(None, f"change set {change_set} has {seen} tests: lines; expected exactly one") +def change_set_files(body: list[tuple[int, str]]) -> dict[int, set[str]]: + """Files each change set names, read from the backticked paths outside its tests: line.""" + files: dict[int, set[str]] = {} + current = None + for _, line in body: + change_set = CHANGE_SET.match(line) + if change_set: + current = int(change_set.group(1)) + files.setdefault(current, set()) + if current is None or TESTS.match(line): + continue + files[current].update(t for t in BACKTICKED.findall(line) if FILE_NAME.search(t)) + return files + + +def shared_files(one: set[str], other: set[str]) -> list[str]: + """Paths two change sets both name; a bare file name matches any path that ends in it.""" + shared = one & other + for mine, theirs in ((one, other), (other, one)): + bare = {path for path in theirs if "/" not in path} + shared |= {path for path in mine if "/" in path and path.rsplit("/", 1)[-1] in bare} + return sorted(shared) + + +def build_waves(files: dict[int, set[str]]) -> tuple[list[list[int]], dict[int, tuple[int, list[str]]]]: + """The waves build's consecutive-disjoint batching gives when only file lists decide. + + A change set joins the current wave when it shares no file with the wave or with + any earlier change set still held back. Also returns, per change set that had to + wait, the change set it waited on last and the files they share. + """ + waves: list[list[int]] = [] + waited: dict[int, tuple[int, list[str]]] = {} + remaining = sorted(files) + while remaining: + wave: list[int] = [] + held: list[int] = [] + for change_set in remaining: + for earlier in wave + held: + shared = shared_files(files[change_set], files[earlier]) + if shared: + waited[change_set] = (earlier, shared) + held.append(change_set) + break + else: + wave.append(change_set) + waves.append(wave) + remaining = held + return waves, waited + + +def wave_report(body: list[tuple[int, str]], built: set[int]) -> list[str]: + files = {number: paths for number, paths in change_set_files(body).items() if number not in built} + if len(files) < 2: + return [] + waves, waited = build_waves(files) + layout = " ".join("[" + " ".join(str(n) for n in wave) + "]" for wave in waves) + lines = [ + f"build waves by file lists alone: {len(files)} change sets in {len(waves)} waves - {layout}" + ] + for change_set, (other, shared) in sorted(waited.items()): + more = f" (+{len(shared) - 3} more)" if len(shared) > 3 else "" + lines.append(f" {change_set} waits on {other}: {', '.join(shared[:3])}{more}") + return lines + + def check_validation(lines: list[str], problem) -> None: for index, line in enumerate(lines): if line.strip().lower() == "### validation": @@ -192,7 +286,8 @@ def problem(number: int | None, message: str) -> None: check_decisions(decisions, problem) check_echoes(found.get("scope", []), decisions, False, problem) check_echoes(found.get("change plan", []), decisions, True, problem) - check_change_plan(found.get("change plan", []), problem) + built = built_change_sets(path) + check_change_plan(found.get("change plan", []), built, problem) check_validation(lines, problem) for _, message in sorted(problems): @@ -201,6 +296,8 @@ def problem(number: int | None, message: str) -> None: print(f"\n{len(problems)} problem(s); the spec is not final.") return 1 print(f"{path}: clean - {len(decisions)} decisions argued.") + for line in wave_report(found.get("change plan", []), built): + print(line) return 0 diff --git a/plugins/factory/references/ci-parity.md b/plugins/factory/references/ci-parity.md index 382ac06..92157c9 100644 --- a/plugins/factory/references/ci-parity.md +++ b/plugins/factory/references/ci-parity.md @@ -21,6 +21,8 @@ Record the commands and outcomes in the plan's `implementation-notes.md`. Run every reproducible project-owned required-check command against the final checkout, after feature E2E and after any hardening edits. +A command that already ran green on this exact tree - no file changed since - +is not run a second time: record its result and where it came from. - A failure is work to fix, including a test described as flaky, unrelated, or pre-existing. Diagnose and remove its nondeterminism; do not add retries, diff --git a/scripts/validate.sh b/scripts/validate.sh index c9b8c05..2b1dfa1 100755 --- a/scripts/validate.sh +++ b/scripts/validate.sh @@ -1016,6 +1016,27 @@ PYEOF fi } +# =========================================================================== +# D01: the dev plugin's spec-lint and change-set-brief tests pass +# =========================================================================== +check_dev_script() { + local dir="dev/evals/tests" + [ -d "$dir" ] || return + if ! python3 - "$dir" >"$LOG_DIR/dev-script-tests.log" 2>&1 <<'PYEOF' +import sys, unittest + +loader = unittest.TestLoader() +suite = loader.discover(start_dir=sys.argv[1], top_level_dir=".") +result = unittest.TextTestRunner(verbosity=2).run(suite) +sys.exit(0 if result.wasSuccessful() and result.testsRun > 0 else 1) +PYEOF + then + tail -60 "$LOG_DIR/dev-script-tests.log" >&2 + fail "D01" "$dir" \ + "Dev script tests failed (full output: $LOG_DIR/dev-script-tests.log); run: python3 -m unittest discover -s dev/evals/tests -t ." + fi +} + # =========================================================================== # Main # =========================================================================== @@ -1041,6 +1062,7 @@ check_factory_unattended check_factory_protocol check_factory_script check_bootstrap_script +check_dev_script if [ -s "$ERROR_FILE" ]; then echo ""