diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 1eefd38..a53f0ee 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,7 +2,7 @@ # Contributing -Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/44cdd76953435a74daac53faa5917bbb19deb014/labs/12-product-engineering-loop). +Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/ebc2162014928c17debb1ce5186f1d6e20608f36/labs/12-product-engineering-loop). The Boatstack repository receives product/runtime changes through a generated pull request. Review the PR's `UPSTREAM.json`, tests, adapter diff, and context-size change; do not hand-edit generated output on `main`. `.github/workflows` is the exception: it is Boatstack's executable control plane, excluded from scheduled projection and changed only through a separate manually reviewed Boatstack PR. diff --git a/UPSTREAM.json b/UPSTREAM.json index 3343dda..d422886 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -1,7 +1,7 @@ { "canonical_context": { - "characters": 83752, - "estimated_tokens": 20938, + "characters": 85503, + "estimated_tokens": 21376, "estimator": "ceil(total characters / 4); compactness signal, not provider billing", "files": [ "product-engineering-loop/references/workflow.md", @@ -12,7 +12,7 @@ }, "files": { ".gitignore": "a7079e923a776f14f1bb3a6aa0a11a133a8e1dfb35af020f327623357b7e3957", - "CONTRIBUTING.md": "25f280f883a57bdacee7b279bf2e3c621fe3f7d1d2337362447dac446ed1da76", + "CONTRIBUTING.md": "64703c794db57cec757149c25037a0f7ad28a922be86436261a5094a9bb26f30", "README.md": "3ce3e95e511089b44e946a44b8d5f4f81d019ece5336db65b2cab1f9dc4d4dad", "assets/boatstack-journey.svg": "e465befc50c8ce30f3e07e8fd97012931beeb053392c8fbf38ad645023b3cc63", "assets/boatstack-mark.svg": "be1f984da1bfa69fa5d1f986d8343d21f7e20921b71db888c928b4d2e54b09b5", @@ -71,11 +71,11 @@ "boatstack/detached_test.go": "6cc70d15baa9a69afacf66ea29ce112efeb166836acb0a52bf9c4bb4c898cee5", "boatstack/docs/control-law-scoping.md": "0ae984821248eabda8c0eeaf201b367991e6742984e7c718df20ecc24caee475", "boatstack/evidence.go": "497a31e6ff632cb1d7c3adfc9f269af3f6aa84e948dd5d417c162767542a27df", - "boatstack/export.go": "a7fc4039efb49d6692af1961c24be585e896a834777a68bf1cb7dd264f11be5f", + "boatstack/export.go": "ca5c284fc658b112f58145e80b2bbfdd33a72b153a97922e2cf8c3f6e98fefb3", "boatstack/export_test.go": "dce5aa3ab5499c82d05859cf86b46dfcee308482491366d83e10ca3fb8605bb6", "boatstack/flow_coding.go": "9fa53a0204f98a25f97775c3acf37392a591c14ce850b44aa587b5806e770bb9", "boatstack/flow_coding_test.go": "dddcd7a85892d4fa10af42739d4c1ff265721b0313e27b6e7a1bbb019d5c3b51", - "boatstack/flow_control.go": "d1cc9268fafd37a322222b26e041d89d8f95e7ca8c7ea0726679801c2836c536", + "boatstack/flow_control.go": "43204356017b837148bd1505779611fbe781097839c2944f41a437a4ec755a82", "boatstack/flow_control_test.go": "02d788c83be55ebd79ffc73875bfd019de45325151eb1f70f506980eb8e77f29", "boatstack/flow_drive.go": "a501ceda390dfd3605e22cf7ecfa15f9d50240b3fac6ebb2bb2d80c615d0a9fc", "boatstack/flow_drive_conformance_test.go": "23edea926c271a1f5718fb9dae1da11e4bf03cceb1357290cd61cd8ffb73beda", @@ -130,8 +130,9 @@ "boatstack/mutation_undo.go": "697d11b600a276ddbcabe6a9f8040d4f7283e017a0e8fd689ef53a274638946c", "boatstack/mutation_undo_test.go": "39540e717e3f2136bf975594043a3db9072b28ebe61c6cb0b982cea5e8b1e14e", "boatstack/next.go": "c133dbf907dc86ca5aacec154f6e4a63e7aa9f5f0ebdd76e1803375b5700675a", + "boatstack/next_actor_conformance_test.go": "23c055bcc99d889344c3f86c7940eb9ef2ef6f4ddab34e91cf215c9d078f5d42", "boatstack/next_banner_test.go": "c431a6987ed1e479442fc9f5db4371632880b92aa790fa9dd0f5285293352c41", - "boatstack/next_response.go": "f63f9593cf4adb1217cc65337fe737c4e5ef07b9c6f311641a77269b2c264b6b", + "boatstack/next_response.go": "11decf2e3b236cbaa183980946ec17ffbbbb1af9c08bd11a466a8487bf229d5f", "boatstack/next_response_conformance_test.go": "be4f3bc7507abfb0ae9f86310eb29e34b166dcc40b6fa103e05babb81f2bd928", "boatstack/next_test.go": "6b5ec46ecf1a197d7644846cecbb6d99873a06b7c4e5562772b5016fa0a4cb11", "boatstack/operation.go": "86c6d1a82cfd0b3f2aadef044a8047fb49c612131ef8ba5f44e92c9f18e45162", @@ -162,7 +163,7 @@ "boatstack/references/host-hook-contracts.md": "2a89d44d0e418a53f2e3b6300fed957cdf878f45ea97ce24b55b66065f0eaa1d", "boatstack/references/irreversible-operation-boundary.md": "e0076f0fea3bf729b2e9bdf353eaeaaf7cdafabfaf26b8d9b27287e5414c2441", "boatstack/references/portability.md": "fb683095991bb0cb06ec56fb8884c49038b283172a7d2f8b203483b7cacb4bae", - "boatstack/references/workflow.md": "d6174f8d546e75829a35418741e94a5606d18314e4a2d8e0bb04394745f335bd", + "boatstack/references/workflow.md": "f370167fbd82ceadf23e21a49e0662bd1fcf8e6eefcda88e987f51496a74aaee", "boatstack/release.go": "82dcb4ca59e8c79a68d5333d650f90e64abd448d04e0c6f504fdf07f42b5ed76", "boatstack/release_test.go": "5cf2d76fe9b836a91ca68eba53d5585e2c4be5b9421aaf939ea0723063a24690", "boatstack/repair_state_test.go": "f3779ac47c3db3927175a545728d3b2e020dbc85f41394d8235753b52afc3739", @@ -203,10 +204,10 @@ "docs/benchmark-corpus-audit.md": "f2d206fe8579a514f9da82b2c96c19b343ac004be67617e1bd34f0f8e0e5e6c6", "docs/benchmark-submission-audit.md": "9518abdd17690729c6423f87cab20418ed47b0915b5faa44b9ef975e9e9c3b79", "docs/configuration.md": "060775c73431f28bd16066bdf9e0f89034d2855c7ca0f5544f660d24b91211d0", - "docs/evidence-engineered-coding.md": "c0a2e7c4e55f9501ac2eb0fa2071d220f12f382c2804f5587088878bafd5d84a", + "docs/evidence-engineered-coding.md": "c83785a7651082402839257369b488aa906abb6395b9b99130cea42df19d7a33", "docs/generated-files.md": "437791765b0a4015032ae21d1a6618563cad92b7402819e4f963bf5ae16284a3", "docs/getting-started.md": "51c2823f21e35140d31e6d5083dc4b89fddd24721ac6acc474154a4da53ee9f8", - "docs/public-claims.json": "3aa44c8b5ad28c6227573a766d720bf161d47bf470c806ace774d76ef185b7f4", + "docs/public-claims.json": "b5dd5544c932a4dbd549047fc90cecec8e9c14842daae12feb2500a0a566896f", "docs/public-surface.md": "713f7a050b5f339cf948299103ef3800417dccfecf2cc1a4166397ea6f978907", "docs/research-and-design.md": "8d78678108f0a6c924e1ff9b32c0f81aae9d1f779e0082843b6f99ad993ae2b6", "docs/safety.md": "7b9b5c515d36e683767ec8d3d9d6d119ac93650b2f629d351deadd4c600ed6a6", @@ -220,7 +221,7 @@ "labs/diagram-json/compiled/evidence.md": "1ba1c989ade070a8ef9a508fbd788d100d7292f2dbacbb2bce895468019f619d", "labs/diagram-json/compiled/tasks.json": "88f60851abf79d851e9fccc754ff3040034ae595306bc87d64784c19eb403e71", "labs/diagram-json/compiled/test-matrix.json": "424657ff505768e50fa113801fd8363364a18269d5297480907a993d44063a39", - "labs/diagram-json/plan.lock.json": "b8d70f1da70841b389e7b3ae2fbc2129d8c1c8328d0fe2be786b5e65f1a8ce72", + "labs/diagram-json/plan.lock.json": "e8b820e1c2f9bfc28e661842324f7c4455f57be29290d0909634a00b71e5919f", "labs/diagram-json/plan.md": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "labs/diagram-json/questions.md": "74733b015002c8a6777c558e7e997fa48c94850b9bd39054fe9366c97ecf728d", "labs/diagram-json/request.md": "0808fc41c36779c404f4a3a121167da6e76cac56df526e70f9ed6d3e0d4c02ed", @@ -351,12 +352,13 @@ "release-notes/2026-07-27-state-ownership-map.md": "d032547aafc1a4acbeb520f6cbb59d7757de4f33fe824701d7b5ea8cd8c8b9e7", "release-notes/2026-07-28-minimum-app-permissions.md": "9ef97e32e5591966ef34ba23f4e0a7aa14d061a4ac5f72f5f4dcecc9bfb62a89", "release-notes/2026-07-28-native-auto-merge-conformance.md": "67d0fab76fd4911b5836d19d06b319537cb1807650d35dbd534627a2c3622757", + "release-notes/2026-07-28-operator-frontier-next-actor.md": "7e769625a2beb8a204d2d79158c18bdac350656de5c55998988a8a5aba2c319a", "release-notes/2026-07-28-protected-native-auto-merge.md": "67dc76a6e7ce51034a0eadc541ba7a8946cfcabe5433cedc25db55321dfb8b62" }, "generator": "operatorstack/intelligence-flow:boatstack-distribution", "schema_version": 1, "source": { - "commit": "44cdd76953435a74daac53faa5917bbb19deb014", + "commit": "ebc2162014928c17debb1ce5186f1d6e20608f36", "path": "labs/12-product-engineering-loop", "repository": "operatorstack/intelligence-flow" } diff --git a/boatstack/export.go b/boatstack/export.go index f47842f..e9ebdfa 100644 --- a/boatstack/export.go +++ b/boatstack/export.go @@ -298,7 +298,7 @@ func BuildExportBundle(configPath string, config ProjectConfig, rawConfig []byte } operations := map[string]string{ - "boatstack-next": "Run the project-local helper next-status --repo . --format response and present its output as the response. This operation is strictly read-only: do not run the reported operation, edit artifacts, contact GitHub beyond the helper's bounded published-PR inspection, or advance a gate. The helper renders the canonical response contract deterministically — the outcome line and the single ### Next step block with the exact runnable command when one is prescribable; never override, re-derive, or add a second next action. Conversation, terminal, worktree, or process observations may be included as clearly labeled context only and must never override the repository-backed result.", + "boatstack-next": "Run the project-local helper next-status --repo . --format response and present its output as the response. This operation is strictly read-only: do not run the reported operation, edit artifacts, contact GitHub beyond the helper's bounded published-PR inspection, or advance a gate. The helper renders the canonical response contract deterministically — the outcome line and the single ### Next step block with the exact runnable command when one is prescribable; never override, re-derive, or add a second next action. The helper also types the step's actor: when the rendered step is marked \"This step is mine to do\", the step is the agent's, and the one next action is the delegation reply g. Only after the exact reply g, execute the prescribed step, re-render next-status --repo . --format response, and continue through further agent-owned steps until the next step belongs to the operator (an approval, a publish or cleanup reply, a feature choice, a product fact) or no action is required. Stop immediately when a step does not change the prescribed next step — repetition without progress is a stall; report the block and hand the turn to the operator. Never end a response by describing work the agent still has to do. Conversation, terminal, worktree, or process observations may be included as clearly labeled context only and must never override the repository-backed result.", "boatstack-run": "First run the read-only next-status --repo . --json and operation-status --repo . --json. If an operation is executing, wait and report it instead of launching it again; if reconciliation is required, verify its exact postcondition before retrying. If NOT_STARTED, respond Start a Boatstack feature and ask the user for the plan produced in the host conversation, then execute auto-plan with its path via --plan (Boatstack does not scan directories for plans) without Git preflight, pausing at its normal decision or approval boundary; do not fetch or require a feature branch. If PUBLISHED, report that the PR is awaiting or lacks verified completion and make reviewing its checks the one next action; do not claim completion. If FEATURE_COMPLETE, respond Feature complete with No action required. Stop on UNVERIFIED, BLOCKED, ambiguous, stale, or invalid state. Before executing the first delivery-stage next_operation (build, repair, test-gate, review-gate, or ship-gate), run the project-local helper run-preflight --repo . --json; planning and plan-gate do not require it. Stop on a blocked preflight; never merge, rebase, force-push, discard changes, switch branches, or create a constrained delivery branch to repair freshness. Then execute exactly the verified next_operation using the canonical operation semantics, verify the resulting repository state, and resolve again. Continue across every declared delivery slice. Pause for the exact plan approval reply a, any material product decision, and the exact PR publication reply o or u; after a valid reply in the current host session, automatically continue the run. A run request never supplies approval or publication authority. For a same-intent test or review failure, use repair, record the observation, and retry from the returned stage. The delivery state's durable repair_attempt is the budget; stop after three complete automated repair-and-gate cycles even across new turns, host restarts, or async notifications. Stop immediately on an amendment, ambiguity, unsafe or destructive capability, stale evidence, branch mismatch, unsupported recovery, or exhausted repair budget. If Cursor reports MainThreadShellExec not initialized, explain that Cursor failed before the Boatstack hook started and make Developer: Reload Window the one recovery action; do not recommend reinstall unless Boatstack reports a missing, drifted, unsafe, or checksum-invalid runtime. Do not use conversation as workflow evidence. Durable operation receipts store execution facts and retry budgets, never autonomous workflow intent. Report the feature, active slice, stages completed, completion or pause reason, durable repair-cycle count, and exactly one next action. Ship means publishing every declared slice PR for review; never merge or deploy.", "root-cause": "Perform failure-mode elimination on a bug, not a patch. This operation is strictly read-only: do not edit product code, create or update artifacts, advance a gate, or contact GitHub; the user supplies the symptom, stack trace, error log, or failing signal as the argument. Locate the failure below its surface symptom and classify it against the failure classes in @.product-loop/failure-moves.md; name the failure CLASS, not the one instance, and if no class fits, name the new class in that vocabulary. Investigate with read-only tools and produce a numbered root-cause chain in which every step is cited to file:line and which distinguishes the crashing frame (the victim) from the true origin (the cause); label authoritative repository facts DISCOVERED and any inference PROPOSED. State the blast radius: every other call site or path exposed to the same class. Propose the minimal STRUCTURAL elimination that makes the whole class unreachable and covers every exposed site, reusing an existing repository pattern or utility where one exists, rather than a local guard on the single line in the trace. Present this as a material product decision with the same tiered paths auto-plan uses under boundary_analysis: [1a] Symptom Patch or [1b] Programmatic Enforcement (a boundary that eliminates the class), and recommend one. Require a regression that reproduces the failure mode before the fix plus the project's own gates as the proof the class is gone, and name related latent hazards left out of scope as non-goals. Then format the result as a host Plan-mode source plan (symptom, root-cause chain, failure mode, blast radius, elimination, non-goals, verification, delivery base branch) and respond Root cause found, making the one next action: save this plan to a durable in-repo path and run auto-plan with it via --plan. Do not implement the fix; hand off to the plan gate.", "auto-plan": "Take the plan produced in the host conversation, supplied explicitly via --plan (Boatstack never scans directories for plans), and refine it into a Markdown-only draft feature package whose canonical structured artifact is plan.md. Run check-plan read-only. If workflow.boundary_analysis is true, evaluate if the change is a symptom of a missing systemic boundary and perform a rapid codebase scan for other vulnerabilities. Present this as a material product decision with tiered paths: [1a] Symptom Patch or [1b] Programmatic Enforcement (Slice 1 for the boundary, Slice 2 for the feature). When workflow.pr_visual_evidence is suggest or require, record a structural pr_visual_evidence decision: relevant with one to three entry/state/viewport/expected scenarios, or not_relevant with a reason. Discover existing visual tooling but never require a frontend framework or add repository tooling during planning. When a scenario is relevant but no capability command resolves, surface a material provisioning decision with tiered paths: [1a] provision the capture capability now as its own ordered delivery slice, [1b] bundle the capture harness into the feature slice, or [1c] record the gap and defer; this is a surfaced choice, never an imposed framework. Record affected_paths and structured side_effects for external writes; use an immutable target identity, transactional or fix-forward recovery, and destructive=false. When workflow.maintain_changelog is true, include CHANGELOG.md in every delivery slice's affected paths. Keep internal phases as tasks in one delivery slice. Only when the accepted outcome explicitly needs multiple PRs, declare ordered delivery_slices and assign every task exactly once; plan approval never authorizes publication. Do not implement, create JSON or locks, or imply acceptance. If ready, respond with Plan ready and make Run /plan-gate the one next action. If decisions remain, respond with I need your input and ask only 1-3 material questions. If an earlier hand-authored draft was never registered and its plan cannot be verified, the guard denies every product mutation at INVALID_STATE with next operation repair-state; run repair-state to quarantine that unregistered malformed draft and return to auto-plan, then re-author the planning Markdown through the owned planning-write channel (stdin), never a raw file write. It is reversible, refuses any feature carrying a plan lock, pr.md, delivery state, tracked files, or an active or published delivery, and never edits product code.", diff --git a/boatstack/flow_control.go b/boatstack/flow_control.go index d4953cf..6d39eaf 100644 --- a/boatstack/flow_control.go +++ b/boatstack/flow_control.go @@ -99,6 +99,73 @@ const ( MarkerRecoveryRepair = deliverycontrol.TransitionID("recovery.repair_state") ) +// NextActor names who performs the prescribed next step. The operator owns a +// step only when it owes operator knowledge or authority — an approval, a +// publish decision, a feature choice, a plan path, a correction fact. The +// agent owns every other step, including steps whose owed inputs are evidence +// the agent produces by doing the work (test runs, the review protocol). A +// working response may end only on an operator-owned step or a terminal +// state; that boundary is the operator frontier. +// control-law: turn-ends-only-at-the-operator-frontier +type NextActor string + +const ( + // NextActorAgent — the coding agent performs this step now. A working + // response never ends on an agent-owned step; the read-only status view + // renders it as a one-key delegation instead of executing it. + NextActorAgent NextActor = "agent" + // NextActorOperator — the step owes operator knowledge or authority; the + // response may end here. + NextActorOperator NextActor = "operator" + // NextActorNone — terminal; nobody owes an action. + NextActorNone NextActor = "none" +) + +// operatorOwedFlags are the prescribed-command inputs that carry operator +// knowledge or authority rather than work-derivable evidence. A prescription +// owing any of these belongs to the operator. The evidence flags (--status, +// --evidence, --reviewer-identity, --review-method) are deliberately absent: +// the agent obtains those by doing the work — never by fabrication — so owing +// them does not move the step across the frontier. +// control-law: turn-ends-only-at-the-operator-frontier +var operatorOwedFlags = map[string]bool{ + "--plan": true, // which source plan: product knowledge + "--feature": true, // which delivery: the operator names the slug + "--mutation": true, // which receipt to reverse: an operator decision + "--preview-fingerprint": true, // publish authority is human-confirmed + "--message": true, // correction facts are human knowledge + "--source-stage": true, + "--classification": true, +} + +// classifyNextActor types the next step by who must act. Fail-closed: anything +// it cannot place returns operator, which preserves prescribe-and-stop — the +// worst misclassification is today's behavior, never a runaway agent. +// control-law: turn-ends-only-at-the-operator-frontier +func classifyNextActor(status NextStatus, next FlowNext) NextActor { + switch { + case status.ObservedStage == "FEATURE_COMPLETE", + status.ObservedStage == "PUBLISHED" && status.Lifecycle == "PUBLISHED_MERGED": + return NextActorNone + case status.ObservedStage == "PUBLISHED": + // Reviewing the open pull request is the operator's act. + return NextActorOperator + case next.Prescribed == nil: + // Ambiguity and unprescribed blocks resolve only by operator choice. + return NextActorOperator + case next.Prescribed.Transition == PublishTransition: + // Opening or updating a PR is operator-confirmed (`o`/`u`), regardless + // of which flags happen to be owed. + return NextActorOperator + } + for _, flag := range next.Prescribed.RequiresHumanInput { + if operatorOwedFlags[flag] { + return NextActorOperator + } + } + return NextActorAgent +} + // FlowNext is the advisory answer for `flow next`: the current delivery-flow // state, the real recommended operation (from ResolveNext — the authoritative // next-move table), and the oracle's lowest-cost next control plus the remaining @@ -138,6 +205,12 @@ type FlowNext struct { // Advisory, never a second primary: the rendering keeps exactly one Run line. // control-law: solution-set-derives-from-guard-declarations Alternatives []PrescribedCommand `json:"alternatives,omitempty"` + // Actor names who performs the next step: "agent" when the step is the + // coding agent's to do now, "operator" when it owes operator knowledge or + // authority (the response may end there — the operator frontier), "none" + // when the flow is terminal. + // control-law: turn-ends-only-at-the-operator-frontier + Actor NextActor `json:"next_actor"` } // PrescribedCommand is the exact next command that makes the oracle's lowest-cost @@ -377,6 +450,7 @@ func nextControlFromStatus(repo string, status NextStatus) (FlowNext, error) { out.FollowUp = followUp } out.Alternatives = alternativesFor(repo, status, out) + out.Actor = classifyNextActor(status, out) return out, nil } out.State = state @@ -402,6 +476,7 @@ func nextControlFromStatus(repo string, status NextStatus) (FlowNext, error) { } } out.Alternatives = alternativesFor(repo, status, out) + out.Actor = classifyNextActor(status, out) return out, nil } @@ -415,6 +490,9 @@ func FormatFlowNext(next FlowNext) string { if next.Reason != "" { fmt.Fprintf(&b, "Reason: %s\n", next.Reason) } + if next.Actor != "" { + fmt.Fprintf(&b, "Next actor: %s\n", next.Actor) + } if next.Resolved { fmt.Fprintf(&b, "Flow state: %s -> goal %s\n", next.State, next.Goal) fmt.Fprintf(&b, "Advisory (flow oracle): next %s, remaining cost %d\n", next.OracleNext, next.RemainingCost) diff --git a/boatstack/next_actor_conformance_test.go b/boatstack/next_actor_conformance_test.go new file mode 100644 index 0000000..bc3ae27 --- /dev/null +++ b/boatstack/next_actor_conformance_test.go @@ -0,0 +1,187 @@ +package boatstack + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/operatorstack/boatstack/boatstack/internal/deliverycontrol" +) + +// control-law: turn-ends-only-at-the-operator-frontier +// (cross-reference: response-contract-is-helper-rendered) +// +// Every prescribed next step is typed by who performs it. A step belongs to +// the operator only when it owes operator knowledge or authority; every other +// step is the agent's, and a working response must not end on it. These tests +// hold the classifier to that boundary (fail-closed to operator), hold the +// renderer to marking agent-owned steps with the delegation line, and pin the +// exported instruction to the delegation reply so a status query stays +// read-only while the operator's next action is always a single key. + +const agentStepMarker = "This step is mine to do." + +// Per-stage classification through the real resolution path: pre-activation +// stages the agent can drive are agent-owned; stages owing operator knowledge +// (the source-plan path, a feature choice) are operator-owned. +func TestNextActorPerStage(t *testing.T) { + expectActor := func(t *testing.T, repo, feature string, want NextActor) FlowNext { + t.Helper() + status, err := ResolveNext(repo, feature) + if err != nil { + t.Fatal(err) + } + next, err := nextControlFromStatus(repo, status) + if err != nil { + t.Fatal(err) + } + if next.Actor != want { + t.Fatalf("actor = %q, want %q (stage %s)", next.Actor, want, status.ObservedStage) + } + return next + } + + t.Run("not_started_owes_plan_path", func(t *testing.T) { + expectActor(t, nextTestRepo(t), "", NextActorOperator) + }) + + t.Run("draft_plan_check_is_agents", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + expectActor(t, repo, "", NextActorAgent) + }) + + t.Run("approved_activation_is_agents", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + if err := os.WriteFile(filepath.Join(repo, ".product-loop", "features", "demo", "approval.md"), []byte("approved\n"), 0o644); err != nil { + t.Fatal(err) + } + expectActor(t, repo, "", NextActorAgent) + }) + + t.Run("build_evidence_is_agents", func(t *testing.T) { + repo, feature := activateTwoSliceDelivery(t) + next := expectActor(t, repo, feature, NextActorAgent) + if next.Prescribed == nil || len(next.Prescribed.RequiresHumanInput) == 0 { + t.Fatal("fixture must owe evidence flags — the point is they do not cross the frontier") + } + }) + + t.Run("ambiguous_choice_is_operators", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "plan-one") + writeSavedFeaturePlan(t, repo, "plan-two") + expectActor(t, repo, "", NextActorOperator) + }) +} + +// Boundary: the classifier is fail-closed. Terminal states owe nobody an +// action; publish authority, operator-owed knowledge flags, and unprescribed +// blocks all resolve to the operator — the worst misclassification is +// prescribe-and-stop, never a runaway agent. +func TestNextActorFrontierBoundaries(t *testing.T) { + cases := []struct { + name string + status NextStatus + next FlowNext + want NextActor + }{ + {"feature_complete_is_terminal", NextStatus{ObservedStage: "FEATURE_COMPLETE"}, FlowNext{}, NextActorNone}, + {"merged_publication_is_terminal", NextStatus{ObservedStage: "PUBLISHED", Lifecycle: "PUBLISHED_MERGED"}, FlowNext{}, NextActorNone}, + {"open_pr_review_is_operators", NextStatus{ObservedStage: "PUBLISHED"}, FlowNext{}, NextActorOperator}, + {"unprescribed_block_is_operators", NextStatus{ObservedStage: "INVALID_STATE"}, FlowNext{}, NextActorOperator}, + {"publish_authority_is_operators", NextStatus{ObservedStage: "REVIEW_PASSED"}, FlowNext{ + Prescribed: &PrescribedCommand{Verb: "publish", Transition: PublishTransition}, + }, NextActorOperator}, + {"owed_knowledge_is_operators", NextStatus{ObservedStage: "BUILD"}, FlowNext{ + Prescribed: &PrescribedCommand{Verb: "record-change", RequiresHumanInput: []string{"--message", "--source-stage", "--classification"}}, + }, NextActorOperator}, + {"owed_evidence_stays_agents", NextStatus{ObservedStage: "BUILD"}, FlowNext{ + Prescribed: &PrescribedCommand{Verb: "record-delivery-gate", RequiresHumanInput: []string{"--status", "--evidence"}, Transition: deliverycontrol.TransitionID("delivery.record_gate_test")}, + }, NextActorAgent}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := classifyNextActor(tc.status, tc.next); got != tc.want { + t.Fatalf("classifyNextActor = %q, want %q", got, tc.want) + } + }) + } +} + +// Rendering: the delegation line appears exactly when the step is the +// agent's, and it always names the single-key reply — the operator's next +// action is known even when the work is the agent's. +func TestResponseMarksAgentOwnedSteps(t *testing.T) { + t.Run("agent_owned_step_carries_delegation", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + _, output := renderedResponse(t, repo) + if !strings.Contains(output, agentStepMarker) { + t.Fatalf("agent-owned step must carry the marker: %q", output) + } + if !strings.Contains(output, "Reply `g`") { + t.Fatalf("agent-owned step must name the delegation reply: %q", output) + } + }) + + t.Run("operator_owned_step_carries_no_delegation", func(t *testing.T) { + _, output := renderedResponse(t, nextTestRepo(t)) + if strings.Contains(output, agentStepMarker) { + t.Fatalf("operator-owned step must not claim to be the agent's: %q", output) + } + }) + + t.Run("ambiguous_block_carries_no_delegation", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "plan-one") + writeSavedFeaturePlan(t, repo, "plan-two") + _, output := renderedResponse(t, repo) + if strings.Contains(output, agentStepMarker) { + t.Fatalf("a feature choice is a human act, never the agent's: %q", output) + } + }) +} + +// Bypass: the exported boatstack-next instruction carries the frontier rule — +// the delegation reply, the continue-until-operator loop, and the +// no-progress stall guard — so no host prose can quietly turn an agent-owned +// step back into an operator to-do item. +func TestExportedNextInstructionCarriesFrontierRule(t *testing.T) { + config := testConfig() + raw, err := MarshalJSON(config) + if err != nil { + t.Fatal(err) + } + bundle, err := BuildExportBundle(".boatstack-project.json", config, raw, "boatstack") + if err != nil { + t.Fatal(err) + } + inspected := 0 + for path, content := range bundle.Files { + if !strings.Contains(path, "boatstack-next") { + continue + } + inspected++ + text := string(content) + for _, rule := range []string{ + agentStepMarker[:len(agentStepMarker)-1], // marker phrase, unpunctuated + "delegation reply g", + "repetition without progress is a stall", + "Never end a response by describing work the agent still has to do", + } { + if !strings.Contains(text, rule) { + t.Fatalf("%s must carry the frontier rule %q", path, rule) + } + } + } + if inspected == 0 { + t.Fatal("no exported boatstack-next instruction found — the frontier rule would be vacuous") + } + workflow := string(bundle.Files[".product-loop/workflow.md"]) + if !strings.Contains(workflow, "### The operator frontier") { + t.Fatal("workflow.md must document the operator frontier") + } +} diff --git a/boatstack/next_response.go b/boatstack/next_response.go index 607a92a..82fcab7 100644 --- a/boatstack/next_response.go +++ b/boatstack/next_response.go @@ -33,6 +33,14 @@ func RenderNextStatusResponse(repo string, status NextStatus) (string, error) { switch { case next.Prescribed != nil: + if next.Actor == NextActorAgent { + // An agent-owned step is never the operator's to perform. This + // read-only status view renders it as a one-key delegation; a working + // response must not end here at all — it does the step, re-renders, + // and repeats until the next step reaches the operator frontier. + // control-law: turn-ends-only-at-the-operator-frontier + b.WriteString("This step is mine to do. Reply `g` and I will do it now, then continue to your next decision.\n") + } writePrescribed(&b, next.Prescribed) if next.FollowUp != "" { fmt.Fprintf(&b, "Then: %s\n", next.FollowUp) diff --git a/boatstack/references/workflow.md b/boatstack/references/workflow.md index 6ad3cb2..bb636cd 100644 --- a/boatstack/references/workflow.md +++ b/boatstack/references/workflow.md @@ -113,11 +113,19 @@ not change the information, ordering, gate semantics, or one-action boundary. Lead with a plain outcome, never a machine code such as `PASS`, `PLAN_APPROVED`, `BLOCKED`, `READY_FOR_BUILD`, `PASS_WITH_GAPS`, or `WAITING_FOR_INPUT`. Keep approval-relevant scope, non-goals, decisions, risks, and gaps visible. Move internal operations (`check-plan`, `record-approval`, `activate-plan`), hashes, paths, tables, receipts, locks, and raw output into **Technical details**. **Exactly one primary action:** end with the action that advances or unblocks the current state; a secondary option gets one short sentence. Never route past a blocked state. +### The operator frontier + +Every next step belongs to one actor: the operator or the agent. A step belongs to the operator only when it needs operator knowledge or authority — an approval (`a`), a publish decision (`o`/`u`), a cleanup decision (`c`/`k`), a feature choice, a source-plan path, or a correction fact. Every other step belongs to the agent, including steps whose evidence the agent produces by doing the work: build sub-actions, plan checks, test runs, and the review protocol. The helper computes the actor (`next_actor` in `flow next --json`) and marks agent-owned steps in the rendered response with "This step is mine to do." + +End a working response only at the operator frontier: the final `### Next step` must belong to the operator, or state that no action is required. Never end a working response by describing work the agent still has to do — do the work, re-render `next-status --repo . --format response`, and continue until the next step belongs to the operator. Presenting the agent's own pending work as the operator's next step is a contract violation. + +The read-only `next` status query is the one exception, because a status question must not mutate anything. When the rendered step is marked as the agent's, the one next action is the delegation reply: reply `g`, and the agent executes the marked step and continues to the operator frontier under the same bounds as the foreground run coordinator. Stop immediately when executing a step does not change the prescribed next step — repetition without progress is a stall, never a loop. Report the block plainly and hand the turn to the operator. + **Write in Simplified Technical English.** Use short sentences, the active voice, and the present tense. State one idea per sentence, put the condition first, and choose the simple, common word. Keep a term consistent, and write positively. This applies to every operation, including the review findings and the PR brief. It does not change the fixed outcome labels, the single `### Next step`, the collapsed **Technical details**, or the reply keys. | State | Outcome -> one next action | |---|---| -| `next`, `/boatstack-next`, `$boatstack next` not started / active / complete / ambiguous | **Start a Boatstack feature** -> save a Plan-mode file or run `auto-plan`; **Next Boatstack stage** -> run the one repository-backed operation; **Feature complete** -> no action required; **Boatstack state needs attention** -> resolve the named ambiguity (address the invalid evidence, or, when the block names only past deliveries, ignore a named past delivery after explicit user confirmation) | +| `next`, `/boatstack-next`, `$boatstack next` not started / active / complete / ambiguous | **Start a Boatstack feature** -> save a Plan-mode file or run `auto-plan`; **Next Boatstack stage** -> run the one repository-backed operation (when the rendered step is marked as the agent's, reply `g` to have the agent do it and continue to the operator frontier); **Feature complete** -> no action required; **Boatstack state needs attention** -> resolve the named ambiguity (address the invalid evidence, or, when the block names only past deliveries, ignore a named past delivery after explicit user confirmation) | | `run`, `/boatstack-run`, `$boatstack run` not started / complete / paused / blocked | **Start a Boatstack feature** -> save a Plan-mode file; **Feature ready for review** -> review the published PRs; **Boatstack run paused** -> provide the one required approval, confirmation, or product answer; **Boatstack run needs attention** -> resolve the named freshness, safety, state, or repair blocker | | `root-cause`, `/root-cause`, `$boatstack root-cause` | **Root cause found** -> save the diagnosis as a source plan and run `auto-plan` with it via `--plan`; the operation is read-only and never edits code or advances a gate | | `auto-plan` ready / needs answers | **Plan ready** -> run `/plan-gate`; **I need your input** -> answer with the displayed choice keys or `r` for all recommendations | diff --git a/docs/evidence-engineered-coding.md b/docs/evidence-engineered-coding.md index f30c093..d25d8cd 100644 --- a/docs/evidence-engineered-coding.md +++ b/docs/evidence-engineered-coding.md @@ -96,7 +96,7 @@ subject to acceptance criteria pass approval is current ``` -That is why context trimming is not automatically an optimization. If removing state increases rework or false acceptance, total cost rises. The canonical runtime references are approximately **20938 estimated tokens**, while host adapters point to one operation at a time. +That is why context trimming is not automatically an optimization. If removing state increases rework or false acceptance, total cost rises. The canonical runtime references are approximately **21376 estimated tokens**, while host adapters point to one operation at a time. ## Control appears at transitions @@ -146,6 +146,6 @@ Delivery and system improvement also remain separate. A failed task may suggest ## What is evidence-backed -The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`44cdd76953435a74daac53faa5917bbb19deb014`](https://github.com/operatorstack/intelligence-flow/tree/44cdd76953435a74daac53faa5917bbb19deb014/labs/12-product-engineering-loop). +The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`ebc2162014928c17debb1ce5186f1d6e20608f36`](https://github.com/operatorstack/intelligence-flow/tree/ebc2162014928c17debb1ce5186f1d6e20608f36/labs/12-product-engineering-loop). The evidence supports specific failure mechanisms and guardrails. It does not establish that Boatstack is optimal, that control-theory notation proves software quality, or that one workflow dominates every team. Those are evaluation questions, so the distribution preserves measurements, provenance, gaps, and negative results. diff --git a/docs/public-claims.json b/docs/public-claims.json index bf3b3aa..3d65872 100644 --- a/docs/public-claims.json +++ b/docs/public-claims.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "source_commit": "44cdd76953435a74daac53faa5917bbb19deb014", + "source_commit": "ebc2162014928c17debb1ce5186f1d6e20608f36", "statuses": ["verified", "observed", "still_being_evaluated"], "claims": [ { @@ -12,7 +12,7 @@ "readable_evidence": "why-these-steps.md#portable-workflow-and-state", "implementation": ["../boatstack/export.go", "../boatstack/references/artifacts.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "human-decisions", @@ -23,7 +23,7 @@ "readable_evidence": "why-these-steps.md#human-decisions", "implementation": ["../boatstack/references/workflow.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "validation-provenance", @@ -34,7 +34,7 @@ "readable_evidence": "why-these-steps.md#validation-provenance", "implementation": ["validation-and-evidence.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "irreversible-operations", @@ -46,7 +46,7 @@ "readable_evidence": "why-these-steps.md#irreversible-operations", "implementation": ["safety.md", "../boatstack/safety.go", "../boatstack/hooks.go"], "verification": ["../boatstack/safety_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "reviewer-ready-pr", @@ -57,7 +57,7 @@ "readable_evidence": "why-these-steps.md#reviewer-ready-pr", "implementation": ["../boatstack/pr.go", "getting-started.md"], "verification": ["../boatstack/pr_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "phase-scoped-delivery", @@ -68,7 +68,7 @@ "readable_evidence": "why-these-steps.md#phase-scoped-delivery", "implementation": ["../boatstack/delivery.go", "../boatstack/safety.go", "../boatstack/hooks.go", "../boatstack/references/workflow.md"], "verification": ["../boatstack/delivery_test.go", "../boatstack/pr_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "model-neutral-contract", @@ -79,7 +79,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "cross-model-failures", @@ -90,7 +90,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "lower-cost-outcomes", @@ -101,7 +101,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "git-worktree-activation", @@ -112,7 +112,7 @@ "readable_evidence": "why-these-steps.md#git-worktree-activation", "implementation": ["../boatstack/runtime_cache.go", "../boatstack/hooks.go"], "verification": ["../boatstack/runtime_cache_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" }, { "id": "visible-updates", @@ -123,7 +123,7 @@ "readable_evidence": "why-these-steps.md#visible-updates", "implementation": ["../boatstack/update.go", "../boatstack/init.go"], "verification": ["../boatstack/update_test.go", "../boatstack/init_test.go", "../boatstack/export_test.go"], - "last_verified_version": "source:44cdd76953435a74daac53faa5917bbb19deb014" + "last_verified_version": "source:ebc2162014928c17debb1ce5186f1d6e20608f36" } ] } diff --git a/labs/diagram-json/plan.lock.json b/labs/diagram-json/plan.lock.json index 0172e28..2c47415 100644 --- a/labs/diagram-json/plan.lock.json +++ b/labs/diagram-json/plan.lock.json @@ -6,7 +6,7 @@ "plan_path": "labs/diagram-json/plan.md", "plan_sha256": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "schema_version": 1, - "source_commit": "44cdd76953435a74daac53faa5917bbb19deb014", + "source_commit": "ebc2162014928c17debb1ce5186f1d6e20608f36", "source_plan_path": "labs/diagram-json/source-plan.md", "source_plan_sha256": "e10593ddaa7522ab80cc991d0a09399257139799e37f737794cd49d68a39985b", "spec_path": "labs/diagram-json/spec.md", diff --git a/release-notes/2026-07-28-operator-frontier-next-actor.md b/release-notes/2026-07-28-operator-frontier-next-actor.md new file mode 100644 index 0000000..bda87a3 --- /dev/null +++ b/release-notes/2026-07-28-operator-frontier-next-actor.md @@ -0,0 +1,16 @@ +### Next steps now name their actor, so status replies never assign you the agent's work + +Every prescribed next step is now typed by who performs it. A step belongs to +the operator only when it owes operator knowledge or authority — an approval, +a publish or cleanup reply, a feature choice, a source-plan path, a correction +fact. Every other step belongs to the agent, including steps whose evidence +the agent produces by doing the work, such as test runs and plan checks. + +The rendered response marks agent-owned steps with "This step is mine to do" +and offers a single delegation key: reply `g` and the agent executes the step, +re-renders, and continues until the next step reaches the operator frontier. +A working response may no longer end by describing work the agent still has +to do; when a step repeats without progress the agent stops and reports the +block instead of looping. `flow next --json` exposes the typing as +`next_actor`, and the classifier fails closed to the operator, so a step it +cannot place behaves exactly as before.