diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index fe25f19..835315d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,7 +2,7 @@ # Contributing -Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/4b31ab38875160d6ef71a85c65efcdb4bd7ad91b/labs/12-product-engineering-loop). +Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/7f71d4816931f34405b6c9a6bccca0eb4e29021f/labs/12-product-engineering-loop). The Boatstack repository receives product/runtime changes through a generated pull request. Review the PR's `UPSTREAM.json`, tests, adapter diff, and context-size change; do not hand-edit generated output on `main`. `.github/workflows` is the exception: it is Boatstack's executable control plane, excluded from scheduled projection and changed only through a separate manually reviewed Boatstack PR. diff --git a/UPSTREAM.json b/UPSTREAM.json index 1de88b0..3cff492 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -12,7 +12,7 @@ }, "files": { ".gitignore": "a7079e923a776f14f1bb3a6aa0a11a133a8e1dfb35af020f327623357b7e3957", - "CONTRIBUTING.md": "9361609e7207a8b0dc8b9257e180798e08e4ec88b9e39b00f93453d9699efb9f", + "CONTRIBUTING.md": "c84e0cbd421edb15d688e7a0c5a75e00a96da9e0fd7ff736c864a26ede7408b8", "README.md": "3ce3e95e511089b44e946a44b8d5f4f81d019ece5336db65b2cab1f9dc4d4dad", "assets/boatstack-journey.svg": "e465befc50c8ce30f3e07e8fd97012931beeb053392c8fbf38ad645023b3cc63", "assets/boatstack-mark.svg": "be1f984da1bfa69fa5d1f986d8343d21f7e20921b71db888c928b4d2e54b09b5", @@ -44,7 +44,7 @@ "boatstack/changelog_test.go": "ce792f23a7fe1e09fb3096cd1314130a6ab69321d4877b12a8e994027541baf7", "boatstack/cmd/boatstack-helper/coverage_conformance_test.go": "347810fec8cc65300ad58cf84570040a001f6dffd9d464f21038034bec6f00e9", "boatstack/cmd/boatstack-helper/flow.go": "5ab24541d3f85c2f442730d3122d18eb6676600bc11fde4805e803a25a8300c7", - "boatstack/cmd/boatstack-helper/main.go": "78b6e98cc9f7c8bddaacd7fb906e61a5c6ba6129d7cd691ea83be54c1d13363f", + "boatstack/cmd/boatstack-helper/main.go": "9e0712b0a3936a3066a33b9c97f92d8ebc07f6e200664b21e6a3af03a00d9f3b", "boatstack/cmd/boatstack-helper/main_test.go": "b36c52d6d5c9dd2428730de10ff18194b7e32a98722e41341c301c6f7a04cad5", "boatstack/command.go": "4726ac515dedab4947be7eb48f88c6cb8b53d674124504b69f03e6396b080ee8", "boatstack/command_test.go": "9f707abba3640add81c3e97ba7e72fedbf98f3394b1c060a9ca4b4a28e919968", @@ -67,11 +67,11 @@ "boatstack/detached_test.go": "6cc70d15baa9a69afacf66ea29ce112efeb166836acb0a52bf9c4bb4c898cee5", "boatstack/docs/control-law-scoping.md": "0ae984821248eabda8c0eeaf201b367991e6742984e7c718df20ecc24caee475", "boatstack/evidence.go": "497a31e6ff632cb1d7c3adfc9f269af3f6aa84e948dd5d417c162767542a27df", - "boatstack/export.go": "56dd395382e033cc919ae8894722d5c6d135ffdee9153edd900dd064c3d08962", + "boatstack/export.go": "a7fc4039efb49d6692af1961c24be585e896a834777a68bf1cb7dd264f11be5f", "boatstack/export_test.go": "dce5aa3ab5499c82d05859cf86b46dfcee308482491366d83e10ca3fb8605bb6", "boatstack/flow_coding.go": "9fa53a0204f98a25f97775c3acf37392a591c14ce850b44aa587b5806e770bb9", "boatstack/flow_coding_test.go": "dddcd7a85892d4fa10af42739d4c1ff265721b0313e27b6e7a1bbb019d5c3b51", - "boatstack/flow_control.go": "76b78c69305475f827ab512cf5e78ae22e6a1b40c4e8b2387b72e00fee09bf07", + "boatstack/flow_control.go": "da5759a2588acc8a67b46b480572fe2f00c1a68769bdf9127a4aef7351dcd44a", "boatstack/flow_control_test.go": "02d788c83be55ebd79ffc73875bfd019de45325151eb1f70f506980eb8e77f29", "boatstack/flow_drive.go": "a501ceda390dfd3605e22cf7ecfa15f9d50240b3fac6ebb2bb2d80c615d0a9fc", "boatstack/flow_drive_conformance_test.go": "23edea926c271a1f5718fb9dae1da11e4bf03cceb1357290cd61cd8ffb73beda", @@ -126,6 +126,8 @@ "boatstack/mutation_undo_test.go": "39540e717e3f2136bf975594043a3db9072b28ebe61c6cb0b982cea5e8b1e14e", "boatstack/next.go": "c133dbf907dc86ca5aacec154f6e4a63e7aa9f5f0ebdd76e1803375b5700675a", "boatstack/next_banner_test.go": "c431a6987ed1e479442fc9f5db4371632880b92aa790fa9dd0f5285293352c41", + "boatstack/next_response.go": "62777e52556098ca8a40197a5d7d42fddd49d5e89f7b9e1884c23bf2cdf07599", + "boatstack/next_response_conformance_test.go": "be4f3bc7507abfb0ae9f86310eb29e34b166dcc40b6fa103e05babb81f2bd928", "boatstack/next_test.go": "6b5ec46ecf1a197d7644846cecbb6d99873a06b7c4e5562772b5016fa0a4cb11", "boatstack/operation.go": "073113e1e7b6349417e70b704bd1a342b460604cbd97a7fab06b1a6494604112", "boatstack/operation_test.go": "59d3dc37319aa4d334c0cacbe886e2f757842e6a28dee8781448e528fecbde11", @@ -195,10 +197,10 @@ "docs/benchmark-corpus-audit.md": "f2d206fe8579a514f9da82b2c96c19b343ac004be67617e1bd34f0f8e0e5e6c6", "docs/benchmark-submission-audit.md": "9518abdd17690729c6423f87cab20418ed47b0915b5faa44b9ef975e9e9c3b79", "docs/configuration.md": "060775c73431f28bd16066bdf9e0f89034d2855c7ca0f5544f660d24b91211d0", - "docs/evidence-engineered-coding.md": "e4eb592093db7fb1ce18f41fe7dd1c89efbaa76760fd618c3375f27fa507b684", + "docs/evidence-engineered-coding.md": "8722b287bb91f5fd0209d61f0594672ab026ebe9f8f70b65f8df5214c4451d95", "docs/generated-files.md": "437791765b0a4015032ae21d1a6618563cad92b7402819e4f963bf5ae16284a3", "docs/getting-started.md": "51c2823f21e35140d31e6d5083dc4b89fddd24721ac6acc474154a4da53ee9f8", - "docs/public-claims.json": "a243fcd3b9a12ef51f491b77b4646db6f8d4b7d435d9b5f0402b97941d73821a", + "docs/public-claims.json": "6e804e45d843599b44b7f7733064ebf1a7799afc5debe7b694b5a0d4b1298edd", "docs/public-surface.md": "713f7a050b5f339cf948299103ef3800417dccfecf2cc1a4166397ea6f978907", "docs/research-and-design.md": "8d78678108f0a6c924e1ff9b32c0f81aae9d1f779e0082843b6f99ad993ae2b6", "docs/safety.md": "7b9b5c515d36e683767ec8d3d9d6d119ac93650b2f629d351deadd4c600ed6a6", @@ -212,7 +214,7 @@ "labs/diagram-json/compiled/evidence.md": "1ba1c989ade070a8ef9a508fbd788d100d7292f2dbacbb2bce895468019f619d", "labs/diagram-json/compiled/tasks.json": "88f60851abf79d851e9fccc754ff3040034ae595306bc87d64784c19eb403e71", "labs/diagram-json/compiled/test-matrix.json": "424657ff505768e50fa113801fd8363364a18269d5297480907a993d44063a39", - "labs/diagram-json/plan.lock.json": "06b5298ea7a2f9a3db31eb3b1a9bee41a74168a2f3ff02bf5d8a00183705392d", + "labs/diagram-json/plan.lock.json": "2c2fbf7ea885339dad90e362725a2ffe8b0edec513d90f4bb5de5c7feede47ea", "labs/diagram-json/plan.md": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "labs/diagram-json/questions.md": "74733b015002c8a6777c558e7e997fa48c94850b9bd39054fe9366c97ecf728d", "labs/diagram-json/request.md": "0808fc41c36779c404f4a3a121167da6e76cac56df526e70f9ed6d3e0d4c02ed", @@ -332,6 +334,7 @@ "release-notes/2026-07-27-document-content-is-data.md": "c6a35c222bf53ba465fadf21e7e95e1764b52e12e41852e51584ce9cb6513f4a", "release-notes/2026-07-27-first-planning-write-owned-channel.md": "7a37e7abf7fd5f8612aca4323d55af1748c9e668bb294518619a1c39e195309f", "release-notes/2026-07-27-guard-dual-reward-corpus.md": "6bec0385c6c553f00517259821e502796ca1b1907aeab718a287560e3e0fa0d6", + "release-notes/2026-07-27-helper-rendered-next-response.md": "08f53d8ade77c0ed6ffe7b63330c67ec0b88a5c71a466ed103f3b66d0952e4e2", "release-notes/2026-07-27-invalid-delivery-block-actionable.md": "8fac8e3921e2285291703efa46e624b72cb5bac1b8492beca4c4b633abb5ba16", "release-notes/2026-07-27-prescriptive-planning-closure.md": "e544408e1c3cceb0cb1979833ea120853c38020933439b39e7f009f454b9661e", "release-notes/2026-07-27-read-only-inspection-pipelines.md": "0963286371e9a12592915c23a958dd013bf2a35e9fca6921691bc8bb3c3d8dc8", @@ -341,7 +344,7 @@ "generator": "operatorstack/intelligence-flow:boatstack-distribution", "schema_version": 1, "source": { - "commit": "4b31ab38875160d6ef71a85c65efcdb4bd7ad91b", + "commit": "7f71d4816931f34405b6c9a6bccca0eb4e29021f", "path": "labs/12-product-engineering-loop", "repository": "operatorstack/intelligence-flow" } diff --git a/boatstack/cmd/boatstack-helper/main.go b/boatstack/cmd/boatstack-helper/main.go index 709739f..cd1e014 100644 --- a/boatstack/cmd/boatstack-helper/main.go +++ b/boatstack/cmd/boatstack-helper/main.go @@ -719,9 +719,13 @@ func nextStatusCommand(arguments []string) int { feature := flags.String("feature", "", "optional specific managed feature to inspect") jsonOutput := flags.Bool("json", false, "print the versioned structured status") render := flags.Bool("render", false, "print the branded, human-facing status banner") + format := flags.String("format", "", `optional output format: "response" renders the canonical response contract (banner, outcome line, one ### Next step)`) if err := flags.Parse(arguments); err != nil { return 2 } + if *format != "" && *format != "response" { + return fail(fmt.Errorf(`unsupported format %q; use --format response`, *format)) + } status, err := boatstack.ResolveNext(*repo, *feature) if err != nil { return fail(err) @@ -732,6 +736,12 @@ func nextStatusCommand(arguments []string) int { return fail(marshalErr) } fmt.Print(string(value)) + } else if *format == "response" { + output, renderErr := boatstack.RenderNextStatusResponse(*repo, status) + if renderErr != nil { + return fail(renderErr) + } + fmt.Print(output) } else if *render { fmt.Print(boatstack.RenderNextStatusBanner(status)) } else { diff --git a/boatstack/export.go b/boatstack/export.go index ad42e36..f47842f 100644 --- a/boatstack/export.go +++ b/boatstack/export.go @@ -298,7 +298,7 @@ func BuildExportBundle(configPath string, config ProjectConfig, rawConfig []byte } operations := map[string]string{ - "boatstack-next": "Run the project-local helper next-status --repo . --json. This operation is strictly read-only: do not run the reported operation, edit artifacts, contact GitHub beyond the helper's bounded published-PR inspection, or advance a gate. Translate the structured result into the canonical response contract. Show the verified feature and active slice when present. Distinguish NOT_STARTED, whose next operation is auto-plan run with the plan path via --plan, from PUBLISHED, which responds PR published and makes reviewing its checks the one action, and FEATURE_COMPLETE, which is reserved for a verified merged PR and responds Feature complete with No action required. If verification_status is BLOCKED, name the ambiguity or invalid evidence and make its safe restoration the one action; never clear artifacts. Conversation, terminal, worktree, or process observations may be included as clearly labeled context only and must never override the repository-backed result. Otherwise make the returned next_operation the one next action.", + "boatstack-next": "Run the project-local helper next-status --repo . --format response and present its output as the response. This operation is strictly read-only: do not run the reported operation, edit artifacts, contact GitHub beyond the helper's bounded published-PR inspection, or advance a gate. The helper renders the canonical response contract deterministically — the outcome line and the single ### Next step block with the exact runnable command when one is prescribable; never override, re-derive, or add a second next action. Conversation, terminal, worktree, or process observations may be included as clearly labeled context only and must never override the repository-backed result.", "boatstack-run": "First run the read-only next-status --repo . --json and operation-status --repo . --json. If an operation is executing, wait and report it instead of launching it again; if reconciliation is required, verify its exact postcondition before retrying. If NOT_STARTED, respond Start a Boatstack feature and ask the user for the plan produced in the host conversation, then execute auto-plan with its path via --plan (Boatstack does not scan directories for plans) without Git preflight, pausing at its normal decision or approval boundary; do not fetch or require a feature branch. If PUBLISHED, report that the PR is awaiting or lacks verified completion and make reviewing its checks the one next action; do not claim completion. If FEATURE_COMPLETE, respond Feature complete with No action required. Stop on UNVERIFIED, BLOCKED, ambiguous, stale, or invalid state. Before executing the first delivery-stage next_operation (build, repair, test-gate, review-gate, or ship-gate), run the project-local helper run-preflight --repo . --json; planning and plan-gate do not require it. Stop on a blocked preflight; never merge, rebase, force-push, discard changes, switch branches, or create a constrained delivery branch to repair freshness. Then execute exactly the verified next_operation using the canonical operation semantics, verify the resulting repository state, and resolve again. Continue across every declared delivery slice. Pause for the exact plan approval reply a, any material product decision, and the exact PR publication reply o or u; after a valid reply in the current host session, automatically continue the run. A run request never supplies approval or publication authority. For a same-intent test or review failure, use repair, record the observation, and retry from the returned stage. The delivery state's durable repair_attempt is the budget; stop after three complete automated repair-and-gate cycles even across new turns, host restarts, or async notifications. Stop immediately on an amendment, ambiguity, unsafe or destructive capability, stale evidence, branch mismatch, unsupported recovery, or exhausted repair budget. If Cursor reports MainThreadShellExec not initialized, explain that Cursor failed before the Boatstack hook started and make Developer: Reload Window the one recovery action; do not recommend reinstall unless Boatstack reports a missing, drifted, unsafe, or checksum-invalid runtime. Do not use conversation as workflow evidence. Durable operation receipts store execution facts and retry budgets, never autonomous workflow intent. Report the feature, active slice, stages completed, completion or pause reason, durable repair-cycle count, and exactly one next action. Ship means publishing every declared slice PR for review; never merge or deploy.", "root-cause": "Perform failure-mode elimination on a bug, not a patch. This operation is strictly read-only: do not edit product code, create or update artifacts, advance a gate, or contact GitHub; the user supplies the symptom, stack trace, error log, or failing signal as the argument. Locate the failure below its surface symptom and classify it against the failure classes in @.product-loop/failure-moves.md; name the failure CLASS, not the one instance, and if no class fits, name the new class in that vocabulary. Investigate with read-only tools and produce a numbered root-cause chain in which every step is cited to file:line and which distinguishes the crashing frame (the victim) from the true origin (the cause); label authoritative repository facts DISCOVERED and any inference PROPOSED. State the blast radius: every other call site or path exposed to the same class. Propose the minimal STRUCTURAL elimination that makes the whole class unreachable and covers every exposed site, reusing an existing repository pattern or utility where one exists, rather than a local guard on the single line in the trace. Present this as a material product decision with the same tiered paths auto-plan uses under boundary_analysis: [1a] Symptom Patch or [1b] Programmatic Enforcement (a boundary that eliminates the class), and recommend one. Require a regression that reproduces the failure mode before the fix plus the project's own gates as the proof the class is gone, and name related latent hazards left out of scope as non-goals. Then format the result as a host Plan-mode source plan (symptom, root-cause chain, failure mode, blast radius, elimination, non-goals, verification, delivery base branch) and respond Root cause found, making the one next action: save this plan to a durable in-repo path and run auto-plan with it via --plan. Do not implement the fix; hand off to the plan gate.", "auto-plan": "Take the plan produced in the host conversation, supplied explicitly via --plan (Boatstack never scans directories for plans), and refine it into a Markdown-only draft feature package whose canonical structured artifact is plan.md. Run check-plan read-only. If workflow.boundary_analysis is true, evaluate if the change is a symptom of a missing systemic boundary and perform a rapid codebase scan for other vulnerabilities. Present this as a material product decision with tiered paths: [1a] Symptom Patch or [1b] Programmatic Enforcement (Slice 1 for the boundary, Slice 2 for the feature). When workflow.pr_visual_evidence is suggest or require, record a structural pr_visual_evidence decision: relevant with one to three entry/state/viewport/expected scenarios, or not_relevant with a reason. Discover existing visual tooling but never require a frontend framework or add repository tooling during planning. When a scenario is relevant but no capability command resolves, surface a material provisioning decision with tiered paths: [1a] provision the capture capability now as its own ordered delivery slice, [1b] bundle the capture harness into the feature slice, or [1c] record the gap and defer; this is a surfaced choice, never an imposed framework. Record affected_paths and structured side_effects for external writes; use an immutable target identity, transactional or fix-forward recovery, and destructive=false. When workflow.maintain_changelog is true, include CHANGELOG.md in every delivery slice's affected paths. Keep internal phases as tasks in one delivery slice. Only when the accepted outcome explicitly needs multiple PRs, declare ordered delivery_slices and assign every task exactly once; plan approval never authorizes publication. Do not implement, create JSON or locks, or imply acceptance. If ready, respond with Plan ready and make Run /plan-gate the one next action. If decisions remain, respond with I need your input and ask only 1-3 material questions. If an earlier hand-authored draft was never registered and its plan cannot be verified, the guard denies every product mutation at INVALID_STATE with next operation repair-state; run repair-state to quarantine that unregistered malformed draft and return to auto-plan, then re-author the planning Markdown through the owned planning-write channel (stdin), never a raw file write. It is reversible, refuses any feature carrying a plan lock, pr.md, delivery state, tracked files, or an active or published delivery, and never edits product code.", diff --git a/boatstack/flow_control.go b/boatstack/flow_control.go index 4ac3cfe..4a6e9e5 100644 --- a/boatstack/flow_control.go +++ b/boatstack/flow_control.go @@ -304,6 +304,13 @@ func NextControl(repo, feature string) (FlowNext, error) { if err != nil { return FlowNext{}, err } + return nextControlFromStatus(repo, status) +} + +// nextControlFromStatus is NextControl on an already-resolved status, so a +// caller that renders both the friendly phrase and the prescription (the +// response contract) observes state exactly once — one resolution, no drift. +func nextControlFromStatus(repo string, status NextStatus) (FlowNext, error) { out := FlowNext{ Goal: flowGoal, RecommendedOp: status.NextOperation, @@ -334,14 +341,14 @@ func NextControl(repo, feature string) (FlowNext, error) { out.OracleNext = advice.NextTransition out.RemainingCost = advice.RemainingCost if advice.NextTransition != "" { - if prescribed, ok := prescribeCommand(repo, feature, status, advice.NextTransition); ok { + if prescribed, ok := prescribeCommand(repo, status.Feature, status, advice.NextTransition); ok { out.Prescribed = prescribed } } // While the slice is building, "build" is opaque; surface the read-only // dependency-ordered next sub-action from the plan's task DAG as a hint. if state == deliverycontrol.StateBuild { - if tasks, err := FlowTasksForActiveSlice(repo, feature); err == nil && tasks.Resolved && len(tasks.Ordered) > 0 { + if tasks, err := FlowTasksForActiveSlice(repo, status.Feature); err == nil && tasks.Resolved && len(tasks.Ordered) > 0 { hint := tasks.Ordered[0] out.SubAction = &hint } diff --git a/boatstack/next_response.go b/boatstack/next_response.go new file mode 100644 index 0000000..f19703a --- /dev/null +++ b/boatstack/next_response.go @@ -0,0 +1,75 @@ +package boatstack + +import ( + "fmt" + "strings" + "unicode" + "unicode/utf8" +) + +// RenderNextStatusResponse renders the canonical user-facing response contract +// for the current workflow position: the branded banner, one friendly outcome +// line, and exactly one "### Next step" block carrying the exact runnable +// command when one is prescribable. Adapters present this output verbatim +// instead of re-deriving the state table from prose — the decision was always +// deterministic (ResolveNext + the prescription layer); this makes the +// rendering deterministic too. +// +// It obeys the banner law (banner-hides-internal-machinery): machine stage +// names and operation codes never appear as status prose. The prescribed +// command line is the one legitimate place helper verbs appear — it is the +// runnable next step, exactly as `flow next` prints it. +// control-law: response-contract-is-helper-rendered +func RenderNextStatusResponse(repo string, status NextStatus) (string, error) { + next, err := nextControlFromStatus(repo, status) + if err != nil { + return "", err + } + + var b strings.Builder + b.WriteString(RenderNextStatusBanner(status)) + b.WriteString("\n" + sentenceCase(friendlyPhrase(status)) + ".\n") + b.WriteString("\n### Next step\n\n") + + switch { + case next.Prescribed != nil: + writePrescribed(&b, next.Prescribed) + if next.FollowUp != "" { + fmt.Fprintf(&b, "Then: %s\n", next.FollowUp) + } + if next.SubAction != nil { + title := next.SubAction.Title + if title != "" { + title = " — " + title + } + fmt.Fprintf(&b, "Next sub-action: %s%s (from the plan task DAG; see `flow tasks`)\n", next.SubAction.ID, title) + } + case status.ObservedStage == "FEATURE_COMPLETE", + status.ObservedStage == "PUBLISHED" && status.Lifecycle == "PUBLISHED_MERGED": + b.WriteString("No action required.\n") + case status.ObservedStage == "PUBLISHED": + if status.PRURL != "" { + fmt.Fprintf(&b, "Review the pull request: %s\n", status.PRURL) + } else { + b.WriteString("Review the pull request.\n") + } + case len(status.BlockingAmbiguity) > 0: + b.WriteString(sentenceCase(friendlyBlockReason(status)) + ":\n") + for _, candidate := range status.BlockingAmbiguity { + fmt.Fprintf(&b, "- %s\n", candidate) + } + default: + b.WriteString(sentenceCase(friendlyBlockReason(status)) + ".\n") + } + return b.String(), nil +} + +// sentenceCase upper-cases the first rune of a friendly phrase so it can open +// a sentence without changing the phrase vocabulary. +func sentenceCase(phrase string) string { + if phrase == "" { + return phrase + } + first, size := utf8.DecodeRuneInString(phrase) + return string(unicode.ToUpper(first)) + phrase[size:] +} diff --git a/boatstack/next_response_conformance_test.go b/boatstack/next_response_conformance_test.go new file mode 100644 index 0000000..dbd58ee --- /dev/null +++ b/boatstack/next_response_conformance_test.go @@ -0,0 +1,230 @@ +package boatstack + +import ( + "os" + "os/exec" + "path/filepath" + "regexp" + "strings" + "testing" +) + +// control-law: response-contract-is-helper-rendered +// (cross-reference: banner-hides-internal-machinery) +// +// The "one next action" decision was always deterministic — ResolveNext plus +// the prescription layer — but its user-facing rendering lived only as prose +// in the exported operation instructions, re-derived by every agent on every +// turn. RenderNextStatusResponse compiles that rendering into the helper: +// one friendly outcome line and exactly one "### Next step" block carrying +// the identical command line `flow next` prescribes. These tests hold the +// renderer to the phrase vocabulary, the single-next-step contract, parity +// with the prescription layer, and the banner law; and they pin the exported +// skill prose to POINTING at the renderer rather than restating a state table. + +// machineTokens must never appear as status prose in a rendered response. +// Lines that carry the runnable command (`boatstack-helper …`) are the one +// legitimate exception — the verb IS the next step there. +var machineTokens = regexp.MustCompile(`DRAFT_PLAN|APPROVED|POLICY_READY|NOT_INITIALIZED|INVALID_STATE|AMBIGUOUS|NOT_STARTED|TEST_PASSED|REVIEW_PASSED|PR_PREVIEW|FEATURE_COMPLETE|repair-state|discard-delivery|plan-gate|ship-gate|review-gate|auto-plan`) + +func renderedResponse(t *testing.T, repo string) (NextStatus, string) { + t.Helper() + status, err := ResolveNext(repo, "") + if err != nil { + t.Fatal(err) + } + output, err := RenderNextStatusResponse(repo, status) + if err != nil { + t.Fatal(err) + } + return status, output +} + +// Positive: per-stage fixtures render the friendly phrase and exactly one +// "### Next step"; when a command is prescribable the block carries it. +func TestResponseContractPerStage(t *testing.T) { + t.Run("not_started", func(t *testing.T) { + repo := nextTestRepo(t) + status, output := renderedResponse(t, repo) + assertResponseShape(t, status, output) + if !strings.Contains(output, "Run: boatstack-helper check-source-plan") { + t.Fatalf("NOT_STARTED must carry the prescribed command: %q", output) + } + }) + + t.Run("draft_plan", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + status, output := renderedResponse(t, repo) + assertResponseShape(t, status, output) + if !strings.Contains(output, "Run: boatstack-helper check-plan") { + t.Fatalf("DRAFT_PLAN must carry the prescribed command: %q", output) + } + if !strings.Contains(output, "Then: ") { + t.Fatalf("DRAFT_PLAN must carry the follow-up: %q", output) + } + }) + + t.Run("approved", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + if err := os.WriteFile(filepath.Join(repo, ".product-loop", "features", "demo", "approval.md"), []byte("approved\n"), 0o644); err != nil { + t.Fatal(err) + } + status, output := renderedResponse(t, repo) + assertResponseShape(t, status, output) + if !strings.Contains(output, "Run: boatstack-helper activate-plan") { + t.Fatalf("APPROVED must carry the prescribed command: %q", output) + } + }) + + t.Run("build", func(t *testing.T) { + repo, feature := activateTwoSliceDelivery(t) + status, err := ResolveNext(repo, feature) + if err != nil { + t.Fatal(err) + } + output, err := RenderNextStatusResponse(repo, status) + if err != nil { + t.Fatal(err) + } + assertResponseShape(t, status, output) + if !strings.Contains(output, "Run: boatstack-helper record-delivery-gate") { + t.Fatalf("BUILD must carry the oracle-prescribed command: %q", output) + } + }) + + t.Run("ambiguous_lists_candidates", func(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "plan-one") + writeSavedFeaturePlan(t, repo, "plan-two") + status, output := renderedResponse(t, repo) + assertResponseShape(t, status, output) + if strings.Contains(output, "Run: ") { + t.Fatalf("AMBIGUOUS must not fabricate a command: %q", output) + } + for _, candidate := range []string{"plan-one", "plan-two"} { + if !strings.Contains(output, "- "+candidate) { + t.Fatalf("candidates must be listed: %q", output) + } + } + }) +} + +func assertResponseShape(t *testing.T, status NextStatus, output string) { + t.Helper() + if got := strings.Count(output, "### Next step"); got != 1 { + t.Fatalf("response must carry exactly one next step, got %d: %q", got, output) + } + phrase := friendlyPhrase(status) + if !strings.Contains(strings.ToLower(output), strings.ToLower(phrase)) { + t.Fatalf("response must carry the friendly phrase %q: %q", phrase, output) + } +} + +// Negative: machine stage names and operation codes never appear as status +// prose — only command lines may name helper verbs. +func TestResponseHidesMachineTokens(t *testing.T) { + fixtures := map[string]func(t *testing.T) string{ + "not_started": func(t *testing.T) string { return nextTestRepo(t) }, + "draft_plan": func(t *testing.T) string { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + return repo + }, + "ambiguous": func(t *testing.T) string { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "plan-one") + writeSavedFeaturePlan(t, repo, "plan-two") + return repo + }, + } + for name, fixture := range fixtures { + t.Run(name, func(t *testing.T) { + _, output := renderedResponse(t, fixture(t)) + for _, line := range strings.Split(output, "\n") { + if strings.Contains(line, "boatstack-helper") { + continue // the runnable command line is the legitimate exception + } + if match := machineTokens.FindString(line); match != "" { + t.Fatalf("machine token %q leaked into status prose: %q", match, line) + } + } + }) + } +} + +// Relation: the response's command line is the identical CommandLine() the +// flow prescription layer renders — one decision, two surfaces, no drift. +func TestResponseCommandMatchesFlowPrescription(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + + status, output := renderedResponse(t, repo) + next, err := nextControlFromStatus(repo, status) + if err != nil { + t.Fatal(err) + } + if next.Prescribed == nil { + t.Fatal("fixture must prescribe a command") + } + if !strings.Contains(output, "Run: "+next.Prescribed.CommandLine()) { + t.Fatalf("response and flow prescription drifted:\nresponse: %q\nprescribed: %q", output, next.Prescribed.CommandLine()) + } + if !strings.Contains(FormatFlowNext(next), "Run: "+next.Prescribed.CommandLine()) { + t.Fatal("FormatFlowNext no longer renders the same command line — parity broken") + } +} + +// Bypass: the exported boatstack-next instruction points at the helper-rendered +// contract and no longer restates the per-stage decision table in prose — the +// prose channel cannot silently reintroduce a second decision surface. +func TestExportedNextInstructionDefersToRenderer(t *testing.T) { + config := testConfig() + raw, err := MarshalJSON(config) + if err != nil { + t.Fatal(err) + } + bundle, err := BuildExportBundle(".boatstack-project.json", config, raw, "boatstack") + if err != nil { + t.Fatal(err) + } + inspected := 0 + for path, content := range bundle.Files { + if !strings.Contains(path, "boatstack-next") { + continue + } + inspected++ + text := string(content) + if !strings.Contains(text, "--format response") { + t.Fatalf("%s must point at the helper-rendered response contract", path) + } + for _, restated := range []string{"Distinguish NOT_STARTED", "FEATURE_COMPLETE, which is reserved"} { + if strings.Contains(text, restated) { + t.Fatalf("%s restates the state table the helper now renders: %q", path, restated) + } + } + } + if inspected == 0 { + t.Fatal("no exported boatstack-next instruction found — the bypass guarantee would be vacuous") + } +} + +// Failure-state: rendering is read-only — the repository is byte-identical +// after resolving and rendering the response. +func TestResponseRenderingIsReadOnly(t *testing.T) { + repo := nextTestRepo(t) + writeSavedFeaturePlan(t, repo, "demo") + before, err := exec.Command("git", "-C", repo, "status", "--porcelain").CombinedOutput() + if err != nil { + t.Fatal(err) + } + _, _ = renderedResponse(t, repo) + after, err := exec.Command("git", "-C", repo, "status", "--porcelain").CombinedOutput() + if err != nil { + t.Fatal(err) + } + if string(before) != string(after) { + t.Fatalf("rendering mutated the repository: before=%q after=%q", before, after) + } +} diff --git a/docs/evidence-engineered-coding.md b/docs/evidence-engineered-coding.md index f40674f..066e420 100644 --- a/docs/evidence-engineered-coding.md +++ b/docs/evidence-engineered-coding.md @@ -146,6 +146,6 @@ Delivery and system improvement also remain separate. A failed task may suggest ## What is evidence-backed -The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`4b31ab38875160d6ef71a85c65efcdb4bd7ad91b`](https://github.com/operatorstack/intelligence-flow/tree/4b31ab38875160d6ef71a85c65efcdb4bd7ad91b/labs/12-product-engineering-loop). +The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`7f71d4816931f34405b6c9a6bccca0eb4e29021f`](https://github.com/operatorstack/intelligence-flow/tree/7f71d4816931f34405b6c9a6bccca0eb4e29021f/labs/12-product-engineering-loop). The evidence supports specific failure mechanisms and guardrails. It does not establish that Boatstack is optimal, that control-theory notation proves software quality, or that one workflow dominates every team. Those are evaluation questions, so the distribution preserves measurements, provenance, gaps, and negative results. diff --git a/docs/public-claims.json b/docs/public-claims.json index 2889e98..ef01ca3 100644 --- a/docs/public-claims.json +++ b/docs/public-claims.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "source_commit": "4b31ab38875160d6ef71a85c65efcdb4bd7ad91b", + "source_commit": "7f71d4816931f34405b6c9a6bccca0eb4e29021f", "statuses": ["verified", "observed", "still_being_evaluated"], "claims": [ { @@ -12,7 +12,7 @@ "readable_evidence": "why-these-steps.md#portable-workflow-and-state", "implementation": ["../boatstack/export.go", "../boatstack/references/artifacts.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "human-decisions", @@ -23,7 +23,7 @@ "readable_evidence": "why-these-steps.md#human-decisions", "implementation": ["../boatstack/references/workflow.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "validation-provenance", @@ -34,7 +34,7 @@ "readable_evidence": "why-these-steps.md#validation-provenance", "implementation": ["validation-and-evidence.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "irreversible-operations", @@ -46,7 +46,7 @@ "readable_evidence": "why-these-steps.md#irreversible-operations", "implementation": ["safety.md", "../boatstack/safety.go", "../boatstack/hooks.go"], "verification": ["../boatstack/safety_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "reviewer-ready-pr", @@ -57,7 +57,7 @@ "readable_evidence": "why-these-steps.md#reviewer-ready-pr", "implementation": ["../boatstack/pr.go", "getting-started.md"], "verification": ["../boatstack/pr_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "phase-scoped-delivery", @@ -68,7 +68,7 @@ "readable_evidence": "why-these-steps.md#phase-scoped-delivery", "implementation": ["../boatstack/delivery.go", "../boatstack/safety.go", "../boatstack/hooks.go", "../boatstack/references/workflow.md"], "verification": ["../boatstack/delivery_test.go", "../boatstack/pr_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "model-neutral-contract", @@ -79,7 +79,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "cross-model-failures", @@ -90,7 +90,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "lower-cost-outcomes", @@ -101,7 +101,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "git-worktree-activation", @@ -112,7 +112,7 @@ "readable_evidence": "why-these-steps.md#git-worktree-activation", "implementation": ["../boatstack/runtime_cache.go", "../boatstack/hooks.go"], "verification": ["../boatstack/runtime_cache_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" }, { "id": "visible-updates", @@ -123,7 +123,7 @@ "readable_evidence": "why-these-steps.md#visible-updates", "implementation": ["../boatstack/update.go", "../boatstack/init.go"], "verification": ["../boatstack/update_test.go", "../boatstack/init_test.go", "../boatstack/export_test.go"], - "last_verified_version": "source:4b31ab38875160d6ef71a85c65efcdb4bd7ad91b" + "last_verified_version": "source:7f71d4816931f34405b6c9a6bccca0eb4e29021f" } ] } diff --git a/labs/diagram-json/plan.lock.json b/labs/diagram-json/plan.lock.json index 07f8a2e..7d63daf 100644 --- a/labs/diagram-json/plan.lock.json +++ b/labs/diagram-json/plan.lock.json @@ -6,7 +6,7 @@ "plan_path": "labs/diagram-json/plan.md", "plan_sha256": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "schema_version": 1, - "source_commit": "4b31ab38875160d6ef71a85c65efcdb4bd7ad91b", + "source_commit": "7f71d4816931f34405b6c9a6bccca0eb4e29021f", "source_plan_path": "labs/diagram-json/source-plan.md", "source_plan_sha256": "e10593ddaa7522ab80cc991d0a09399257139799e37f737794cd49d68a39985b", "spec_path": "labs/diagram-json/spec.md", diff --git a/release-notes/2026-07-27-helper-rendered-next-response.md b/release-notes/2026-07-27-helper-rendered-next-response.md new file mode 100644 index 0000000..95f44d2 --- /dev/null +++ b/release-notes/2026-07-27-helper-rendered-next-response.md @@ -0,0 +1,7 @@ +### The next-step response is now rendered by the helper, not re-derived from prose + +Asking Boatstack "what's next" always had a deterministic answer, but the user-facing response — the outcome line and the single next action — was described in the skill instructions as prose, and every agent re-derived it from a written state table on every turn. That table was one more place the guidance could drift from the machine. + +`next-status --format response` now renders the canonical response contract directly: the branded banner, one plain-language outcome line, and exactly one "### Next step" block carrying the exact runnable command whenever one is prescribable — the identical command line `flow next` prints, produced by the same prescription layer, never a copy. The skill instructions now point at the renderer instead of restating the table, and a conformance suite pins the parity, the single-next-step shape, and the rule that machine stage names never leak into status prose. + +This is the first slice of compiling the instruction layer down into the helper: where a rule can be rendered deterministically, the prose now defers to the rendering.