From b04c3706c6d8d8a0092088a37d66244e5a43e33a Mon Sep 17 00:00:00 2001 From: "operator-stack-publisher[bot]" Date: Thu, 30 Jul 2026 11:15:42 +0000 Subject: [PATCH] Sync Boatstack from Intelligence Flow Labs @ 146f50c0f330 --- CONTRIBUTING.md | 2 +- UPSTREAM.json | 29 ++--- boatstack/capture.go | 5 +- boatstack/capture_test.go | 2 +- boatstack/denial.go | 3 + boatstack/pr.go | 25 +++- boatstack/pr_test.go | 115 ++++++++++++++++-- boatstack/references/config-schema.md | 2 +- boatstack/safety.go | 3 + boatstack/visual_evidence.go | 13 ++ docs/configuration.md | 2 +- docs/evidence-engineered-coding.md | 2 +- docs/public-claims.json | 24 ++-- labs/diagram-json/plan.lock.json | 2 +- ...-approved-scenarios-escalate-to-require.md | 3 + 15 files changed, 182 insertions(+), 50 deletions(-) create mode 100644 release-notes/2026-07-30-plan-approved-scenarios-escalate-to-require.md diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 05a828a..265ad25 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,7 +2,7 @@ # Contributing -Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/f42f1a845ecc6e5905d6e38dca863749cc19f0c0/labs/12-product-engineering-loop). +Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/146f50c0f3309e592ed45374b8cddb3d8ed64d85/labs/12-product-engineering-loop). The Boatstack repository receives product/runtime changes through a generated pull request. Review the PR's `UPSTREAM.json`, tests, adapter diff, and context-size change; do not hand-edit generated output on `main`. `.github/workflows` is the exception: it is Boatstack's executable control plane, excluded from scheduled projection and changed only through a separate manually reviewed Boatstack PR. diff --git a/UPSTREAM.json b/UPSTREAM.json index 6a5fa01..a2191ba 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -12,7 +12,7 @@ }, "files": { ".gitignore": "a7079e923a776f14f1bb3a6aa0a11a133a8e1dfb35af020f327623357b7e3957", - "CONTRIBUTING.md": "970762fd2bd13341d06dcef8929099dc3b71b246f3bc52e1e53e9df6957c5c68", + "CONTRIBUTING.md": "c3334d810923b2cf4a745dae0aecc859d5de7cdafc94c0d193fb7d7b08892c53", "README.md": "3ce3e95e511089b44e946a44b8d5f4f81d019ece5336db65b2cab1f9dc4d4dad", "assets/boatstack-journey.svg": "e465befc50c8ce30f3e07e8fd97012931beeb053392c8fbf38ad645023b3cc63", "assets/boatstack-mark.svg": "be1f984da1bfa69fa5d1f986d8343d21f7e20921b71db888c928b4d2e54b09b5", @@ -38,8 +38,8 @@ "boatstack/attach.go": "6a855440fac9acc63be857efef9d76210a4619774728684832dfbb08caf42a30", "boatstack/capability.go": "fc6bccb0ce4df579b81252ecffeb23c68d62f02ff041bfec980d844baf0885a3", "boatstack/capability_test.go": "e8322903a843970d7f0317d2629466532cb05c3ffcb44b9423adf4219faa8021", - "boatstack/capture.go": "0e137d8251a658118bd56812040575d01f02fd2f862c80ef761b253db6b35bbe", - "boatstack/capture_test.go": "31a8afa40a54ae716fea1ea44be3b41705ed171789a181d1ef26432855c05c9e", + "boatstack/capture.go": "98a29fedcd7f6a95fadc9bac8c1bd601fee24c17cd64a6a76892e068f040ef66", + "boatstack/capture_test.go": "7072d0c5f9ab0ea9ef493f9e6dcb01cf0de544514c8201e30b8e4e6a4065505a", "boatstack/changelog.go": "6b06be7cd9738de29ba6e87aa2569f3b027a2e618b04524f5abd7abaa17945bf", "boatstack/changelog_test.go": "ce792f23a7fe1e09fb3096cd1314130a6ab69321d4877b12a8e994027541baf7", "boatstack/cmd/boatstack-helper/coverage_conformance_test.go": "f5a931d2b5cdb0d32af85acbbd499fa0f509ed36d62e772e6b6072e348200aca", @@ -64,7 +64,7 @@ "boatstack/delivery_terminal_conformance_test.go": "50b88bdfb9bdfc9dc50c11f46276e1eb13eed61618e75b4c951dfa2ac6075df0", "boatstack/delivery_test.go": "45c48ff7581c911bcaf821c3e4241d4ae2a9bb4aa682485cc58b6ad8fe1c85bf", "boatstack/deliverycontrol_parity_test.go": "706be08bdc89192cf200f0b2cf889c8f4bb0e666a5d8a557996d7215101ce2ca", - "boatstack/denial.go": "507593c091c7ee80a3a4f0299d247d616a85331bc1c0aaf2d69f8a2e7aaf6f1c", + "boatstack/denial.go": "3b6419412c7672ca2dd9cf80140675b8326d6b54460bbcb650cf3dde48dddd56", "boatstack/denial_escalation_conformance_test.go": "d50f0e1c803c8f5731a46dbe6f82c7935c0ccbd582f513cbe07211f5e902157e", "boatstack/denial_ledger.go": "a35bf8fd8c1f6b9302109cf0087e1b9158e43e07b92b18df5e589a637d58491d", "boatstack/denial_solutions.go": "ae6cb218c01a89c14367242b21e2b96d85e1fda357076b7bffedb56066ef961b", @@ -165,10 +165,10 @@ "boatstack/planning_first_write_conformance_test.go": "873097aa9384b75bf01e74a475f3ec2ac7cca4a28f733e82f2c82959032c6a30", "boatstack/planning_test.go": "06ec7022222d926040c3ae28b84ab50c3d2f804ae6473e61b303804dd992d884", "boatstack/post_publish_prescribe_conformance_test.go": "3c20d359ff84648db7dedb227b4d64e6574d9f41d3cdca0adefec1c60bfbf4ae", - "boatstack/pr.go": "3c7088e47c997a0b6c25e40dad56a1662be0451c025f39460b4b62961cb2a634", + "boatstack/pr.go": "c389bd5fddf788290069eb858e1062433d1ff4dff928855a5211ac470c67ccd7", "boatstack/pr_phase.go": "59f8cbb75b6b538a5345474acd6a725450979579bf8ecf9591956cbbe1cc4737", "boatstack/pr_phase_conformance_test.go": "bc9c834e9c4ed43b35d81abafd7b1bf2a264ea2a8c4a4ec9758ee18d1d438968", - "boatstack/pr_test.go": "89f6f88b1e3eb51efbe6faaff8586bd930d4c868fae5eafd48e1746b919c27a7", + "boatstack/pr_test.go": "5c0ff03eb21e383026a4e9fbc5671b2e316040e4e4c55ab6583b930e7e117dda", "boatstack/provenance.go": "d44dcd5421306269326f1202ba1d52df8c252490550270ef9d022e8ec2b65210", "boatstack/provision.go": "eb7333a73331b011adc93f59a2d97415d850c2e588f0e2bbb5116984e2ef927d", "boatstack/provision_test.go": "214e9edb991a66d5bbb696a7c1b63876d2f799f2cab4e3f40785f4e8f1eac57b", @@ -182,7 +182,7 @@ "boatstack/reexec_unix.go": "ff86157a9aa20c82a56fcd859b70669b7eacf4e0a9f61a4546ef33808437939e", "boatstack/reexec_windows.go": "f5335c8c28cb4e89048b058b1c4d12f78644f99acb4f6167ff60e622dfb9e742", "boatstack/references/artifacts.md": "e362b2663c904beb8daa777c997f9fcb2ed0e70442ed2e7bef3b169895d170f1", - "boatstack/references/config-schema.md": "a6860ec7e1cd155e7e36dc465b8cae79de896714751e1c04dfa68d3f5c4a45e0", + "boatstack/references/config-schema.md": "01a3b7a9269a08a5cd2a86e967940c1e57f78e166db0b7289bff7a98732c8691", "boatstack/references/failure-moves.md": "b65ef72035afa6ad0dce589a0b38f84bc40cde3864c9ecf973f08fc687f001c3", "boatstack/references/host-hook-contracts.md": "2a89d44d0e418a53f2e3b6300fed957cdf878f45ea97ce24b55b66065f0eaa1d", "boatstack/references/irreversible-operation-boundary.md": "e0076f0fea3bf729b2e9bdf353eaeaaf7cdafabfaf26b8d9b27287e5414c2441", @@ -200,7 +200,7 @@ "boatstack/runtime_cache.go": "e026ffc1906f7e1e98b768bae63e6658164d2826c07169c9121ce0f23c73faf8", "boatstack/runtime_cache_test.go": "b981467ddc9f0f562da6bff5de7a80a9fe5a433a0317541d1e48df268546ac85", "boatstack/runtime_provenance_test.go": "1d52f1e6b0691cf4667729cc9b9f3c55c128f0aa3321f3a2843a9aa6fd0e73dc", - "boatstack/safety.go": "2f274b90a2b50798bc5a19a74ddaf4f2eecd66876234d21b3753406f69522170", + "boatstack/safety.go": "643837af3d76d819dc4e639a4ccd7c20934dcb2396636c28bdafa529f7a37b18", "boatstack/safety_corpus_test.go": "824051705dd893338ac2246e2d75231703e36574cc9c23902237f72ab2d35960", "boatstack/safety_test.go": "ddd7a72d4ff50c046aa46fd97b108d629cef27cf15816150b6d54baf5f1c4c36", "boatstack/safety_update_publisher_test.go": "ed3f8187036623694dfe7c395cdae00fdae14609bab6124d1fdfc6fe73fa2196", @@ -217,7 +217,7 @@ "boatstack/update_publication.go": "5c1ac8445345c6546d165b9321a736453b313ced96a0059465b8ad637498bac6", "boatstack/update_publication_test.go": "c5f32578db53be65e35452d5e8b4520884354e4e80a370dcec19860ea644d091", "boatstack/update_test.go": "bf5f19f8499db6dd7356917d867b113b790548ab89bfdea59e2adf4999a82a6a", - "boatstack/visual_evidence.go": "4d69f98adb6087d4830e6703273ae6e03b28ec8b3c496a6e432fd5e0e54d8a2f", + "boatstack/visual_evidence.go": "f9fbde0e89f0ce8d9503939197d9b92cc8da377053007371d28360e29715a84a", "boatstack/visual_evidence_test.go": "0fe8f5154ef4398dfeba5e7f7b387b2d279ed75635ea35d93ff392269f2cc6d0", "boatstack/visual_publisher.go": "ec5e95228b48e4ec20731975606048f5e732c4e09a7fd175770d4b15e447cc88", "boatstack/visual_publisher_test.go": "979f600edae00c77569ae753b80530e8cbf2e3b342995efcd992d351ff382eaf", @@ -230,11 +230,11 @@ "docs/account-recovery-walkthrough.md": "676034974594a7d1a559b24dbed31d7ccc429eb81404b203ca07bbdaa19ec3d3", "docs/benchmark-corpus-audit.md": "f2d206fe8579a514f9da82b2c96c19b343ac004be67617e1bd34f0f8e0e5e6c6", "docs/benchmark-submission-audit.md": "9518abdd17690729c6423f87cab20418ed47b0915b5faa44b9ef975e9e9c3b79", - "docs/configuration.md": "caa95f1dd484a71a9051951652d6457294c213999fe00fb764d05e8a509cc786", - "docs/evidence-engineered-coding.md": "d6d7babc0c4e61030a1e76754dc45f2bc40cdb7df88a9cf312e24aa9c06ce7c7", + "docs/configuration.md": "8c5b0a6a8394333165dd0b2b1a70ca74e9d53f784b478e2ef7f1b12d93bc03b0", + "docs/evidence-engineered-coding.md": "27da8a87d58783dc2ac2ad72cb2265cbdbab456f41fb4e09f261dbc44d90cd12", "docs/generated-files.md": "437791765b0a4015032ae21d1a6618563cad92b7402819e4f963bf5ae16284a3", "docs/getting-started.md": "51c2823f21e35140d31e6d5083dc4b89fddd24721ac6acc474154a4da53ee9f8", - "docs/public-claims.json": "962bde13802724ae3ad87955bec0e48e0d4274385703be509308c43a7e91674f", + "docs/public-claims.json": "3bf81d5a1ca4371fc34a6b25eddd156da6c3e1cda728785cf3a0559a0d367596", "docs/public-surface.md": "713f7a050b5f339cf948299103ef3800417dccfecf2cc1a4166397ea6f978907", "docs/research-and-design.md": "8d78678108f0a6c924e1ff9b32c0f81aae9d1f779e0082843b6f99ad993ae2b6", "docs/safety.md": "7b9b5c515d36e683767ec8d3d9d6d119ac93650b2f629d351deadd4c600ed6a6", @@ -248,7 +248,7 @@ "labs/diagram-json/compiled/evidence.md": "1ba1c989ade070a8ef9a508fbd788d100d7292f2dbacbb2bce895468019f619d", "labs/diagram-json/compiled/tasks.json": "88f60851abf79d851e9fccc754ff3040034ae595306bc87d64784c19eb403e71", "labs/diagram-json/compiled/test-matrix.json": "424657ff505768e50fa113801fd8363364a18269d5297480907a993d44063a39", - "labs/diagram-json/plan.lock.json": "963e46963033dc5911e2cdd4a034f0aff909fb6c7e966a224d854b6fa9d7c447", + "labs/diagram-json/plan.lock.json": "2c861b4174ba2e09d859ff9463f62e0e5ffb4198db121cb426246b206cc697a4", "labs/diagram-json/plan.md": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "labs/diagram-json/questions.md": "74733b015002c8a6777c558e7e997fa48c94850b9bd39054fe9366c97ecf728d", "labs/diagram-json/request.md": "0808fc41c36779c404f4a3a121167da6e76cac56df526e70f9ed6d3e0d4c02ed", @@ -391,12 +391,13 @@ "release-notes/2026-07-28-retromine-recurrence-detector.md": "27790993a02e73f3a2700dce7340d044aeb0e52f785ded68271ff6673add9e25", "release-notes/2026-07-29-readiness-and-journey-control.md": "2411db76974990ad0128fcd90e795c4c1c5c70d6d6f15fccd452f7e1a242310b", "release-notes/2026-07-30-auto-capture-on-ship.md": "49aa077ac52005d380ba2cae9e406c7d0a9941157c40aa4ed44b9e59195b3837", + "release-notes/2026-07-30-plan-approved-scenarios-escalate-to-require.md": "21c3f2be51b834fc2eb7662236db2549e80b1a1a0665a0721e773b7736e2c079", "release-notes/2026-07-30-visual-evidence-survives-preview-commit.md": "71d19e9fabb40e8cb1e939f26bf288911d227979926f43f337046c6cf6b66551" }, "generator": "operatorstack/intelligence-flow:boatstack-distribution", "schema_version": 1, "source": { - "commit": "f42f1a845ecc6e5905d6e38dca863749cc19f0c0", + "commit": "146f50c0f3309e592ed45374b8cddb3d8ed64d85", "path": "labs/12-product-engineering-loop", "repository": "operatorstack/intelligence-flow" } diff --git a/boatstack/capture.go b/boatstack/capture.go index 2c64147..c3481d2 100644 --- a/boatstack/capture.go +++ b/boatstack/capture.go @@ -163,7 +163,10 @@ func CaptureEvidence(options CaptureEvidenceOptions) (PRVisualEvidenceManifest, manifest := PRVisualEvidenceManifest{ Key: key, - Policy: config.Workflow.PRVisualEvidence, + // The manifest records the configured policy verbatim (informational); + // the effective policy — including plan-escalated require semantics — + // is re-derived by resolvePRVisualEvidence at every decode. + Policy: config.Workflow.PRVisualEvidence, Relevance: relevance, RelevanceSource: source, Status: "PASS", diff --git a/boatstack/capture_test.go b/boatstack/capture_test.go index 9b02af4..183e8df 100644 --- a/boatstack/capture_test.go +++ b/boatstack/capture_test.go @@ -112,7 +112,7 @@ func TestCaptureEvidenceProducesManifestTrustedByPRContext(t *testing.T) { if err != nil { t.Fatal(err) } - _, status, count, _, _, _, resolved, err := resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", head, diffHash) + _, status, count, _, _, _, _, resolved, err := resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", head, diffHash) if err != nil { t.Fatal(err) } diff --git a/boatstack/denial.go b/boatstack/denial.go index d661b04..8981cc1 100644 --- a/boatstack/denial.go +++ b/boatstack/denial.go @@ -541,6 +541,9 @@ func denialFor(host string, finding SafetyFinding) Denial { } d.Qualifier = "visual evidence is owed" d.Detail = "PR publication is blocked until required visual evidence is current for " + target + "." + if finding.PolicySource == "plan-escalated" { + d.Detail += " The approved plan declares visual scenarios, so the configured suggest policy ships with require semantics for this feature." + } if reason := strings.TrimSpace(finding.Reason); reason != "" { d.Detail += " Automatic capture reported: " + reason + "." } diff --git a/boatstack/pr.go b/boatstack/pr.go index 9e934d8..c3ef0c2 100644 --- a/boatstack/pr.go +++ b/boatstack/pr.go @@ -64,6 +64,9 @@ type PRContext struct { PRVisualEvidenceFingerprint string `json:"pr_visual_evidence_fingerprint"` PRVisualEvidenceRelevance string `json:"pr_visual_evidence_relevance"` PRVisualEvidenceSource string `json:"pr_visual_evidence_source"` + // PRVisualEvidencePolicySource is "configured", or "plan-escalated" when + // a plan-approved visual decision lifts suggest to require semantics. + PRVisualEvidencePolicySource string `json:"pr_visual_evidence_policy_source,omitempty"` // PRVisualEvidenceCaptureDetail explains why automatic capture could not // produce current evidence. Deliberately outside the context fingerprint: // a flaky harness message must not destabilize preview equality. @@ -183,7 +186,7 @@ func boundedCaptureDetail(detail string) string { return detail } -func resolvePRVisualEvidence(repo string, config ProjectConfig, mode, feature, head, diffHash string) (string, string, int, string, string, string, *PRVisualEvidenceManifest, error) { +func resolvePRVisualEvidence(repo string, config ProjectConfig, mode, feature, head, diffHash string) (string, string, int, string, string, string, string, *PRVisualEvidenceManifest, error) { policy := normalizedPRVisualEvidencePolicy(config.Workflow.PRVisualEvidence) relevance, source := "unresolved", "agent-proposed" var scenarios []PRVisualScenario @@ -191,12 +194,20 @@ func resolvePRVisualEvidence(repo string, config ProjectConfig, mode, feature, h var err error relevance, source, scenarios, err = planVisualDecision(repo, feature) if err != nil { - return "", "", 0, "", "", "", nil, err + return "", "", 0, "", "", "", "", nil, err } } + // The effective policy is what everything downstream (coercion, preview + // frontmatter, the publication block) reads. Escalation keys on the + // plan's own decision, before any manifest overrides it. + policySource := "configured" + if visualEscalationApplies(policy, relevance, len(scenarios)) { + policy = "require" + policySource = "plan-escalated" + } key, err := visualEvidenceKey(mode, feature, head) if err != nil { - return "", "", 0, "", "", "", nil, err + return "", "", 0, "", "", "", "", nil, err } status := "NOT_APPLICABLE" var manifest *PRVisualEvidenceManifest @@ -235,9 +246,9 @@ func resolvePRVisualEvidence(repo string, config ProjectConfig, mode, feature, h } raw, err := MarshalJSON(payload) if err != nil { - return "", "", 0, "", "", "", nil, err + return "", "", 0, "", "", "", "", nil, err } - return policy, status, count, SHA256Bytes(raw), relevance, source, manifest, nil + return policy, status, count, SHA256Bytes(raw), relevance, source, policySource, manifest, nil } type PRPublishOptions struct { @@ -735,7 +746,7 @@ func PreparePRContext(options PRContextOptions) (PRContext, error) { if err != nil { return PRContext{}, err } - visualPolicy, visualStatus, visualCount, visualFingerprint, visualRelevance, visualSource, visualManifest, err := resolvePRVisualEvidence( + visualPolicy, visualStatus, visualCount, visualFingerprint, visualRelevance, visualSource, visualPolicySource, visualManifest, err := resolvePRVisualEvidence( repo, config, mode, options.Feature, head, SHA256Bytes(diff), ) if err != nil { @@ -762,6 +773,7 @@ func PreparePRContext(options PRContextOptions) (PRContext, error) { PRVisualEvidencePolicy: visualPolicy, PRVisualEvidenceStatus: visualStatus, PRVisualEvidenceCount: visualCount, PRVisualEvidenceFingerprint: visualFingerprint, PRVisualEvidenceRelevance: visualRelevance, PRVisualEvidenceSource: visualSource, + PRVisualEvidencePolicySource: visualPolicySource, PRVisualEvidenceCaptureDetail: captureDetail, PRVisualEvidence: visualManifest, PreviewPath: previewPath, @@ -1127,6 +1139,7 @@ func PublishPR(options PRPublishOptions) (string, error) { Source: "publication", BlockingFeature: context.Feature, Reason: context.PRVisualEvidenceCaptureDetail, + PolicySource: context.PRVisualEvidencePolicySource, } return "", fmt.Errorf("%s", denialWithOptions(repo, "", finding).Render(RenderPlain)) } diff --git a/boatstack/pr_test.go b/boatstack/pr_test.go index fe40ea6..10c9cc9 100644 --- a/boatstack/pr_test.go +++ b/boatstack/pr_test.go @@ -620,9 +620,14 @@ func TestManagedPRVisualEvidenceUsesApprovedPlanScenarios(t *testing.T) { if err != nil { t.Fatal(err) } - if context.Mode != "managed" || context.PRVisualEvidenceRelevance != "relevant" || context.PRVisualEvidenceSource != "managed-plan" || context.PRVisualEvidenceStatus != "NOT_VERIFIED" { + // A plan-approved visual decision escalates suggest to require semantics, + // so missing evidence is BLOCKED here, not a shippable NOT_VERIFIED gap. + if context.Mode != "managed" || context.PRVisualEvidenceRelevance != "relevant" || context.PRVisualEvidenceSource != "managed-plan" || context.PRVisualEvidenceStatus != "BLOCKED" { t.Fatalf("managed visual decision was not projected: %#v", context) } + if context.PRVisualEvidencePolicy != "require" || context.PRVisualEvidencePolicySource != "plan-escalated" { + t.Fatalf("plan-approved scenarios did not escalate suggest: %#v", context) + } previewPath := writePreview(t, repo, context, "Expose approved visual review scenario", visualEvidenceBody(managedPRBody(), context.PRVisualEvidenceStatus)) if _, _, err := CheckPRPreview(repo, previewPath); err != nil { t.Fatal(err) @@ -728,15 +733,17 @@ func TestProductDiffChangeInvalidatesPassVisualEvidence(t *testing.T) { t.Fatal(err) } changedDiff := strings.Repeat("c", 64) - _, status, _, _, _, _, _, err := resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", context.HeadBranch, changedDiff) + // The plan declares relevant scenarios, so suggest ships with require + // semantics: stale evidence is BLOCKED, never a shippable gap. + _, status, _, _, _, _, _, _, err := resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", context.HeadBranch, changedDiff) if err != nil { t.Fatal(err) } - if status != "NOT_VERIFIED" { - t.Fatalf("product change did not stale the evidence: %s", status) + if status != "BLOCKED" { + t.Fatalf("product change did not stale the evidence to a block: %s", status) } config.Workflow.PRVisualEvidence = "require" - _, status, _, _, _, _, _, err = resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", context.HeadBranch, changedDiff) + _, status, _, _, _, _, _, _, err = resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", context.HeadBranch, changedDiff) if err != nil { t.Fatal(err) } @@ -745,6 +752,92 @@ func TestProductDiffChangeInvalidatesPassVisualEvidence(t *testing.T) { } } +// Invariant: a plan that promises pixels cannot ship without them — suggest +// escalates to require semantics for plan-approved scenarios even when no +// capture capability is registered, and the denial says why. +func TestSuggestEscalatesToRequireWhenPlanDeclaresScenarios(t *testing.T) { + repo := prTestRepoConfigured(t, func(config *ProjectConfig) { + config.Workflow.PRVisualEvidence = "suggest" + }) + activateManagedFeature(t, repo, "reviewer-ready") + context, err := PreparePRContext(PRContextOptions{Repo: repo, Feature: "reviewer-ready"}) + if err != nil { + t.Fatal(err) + } + if context.PRVisualEvidencePolicy != "require" || context.PRVisualEvidencePolicySource != "plan-escalated" || context.PRVisualEvidenceStatus != "BLOCKED" { + t.Fatalf("plan-approved scenarios did not escalate: %#v", context) + } + previewPath := writePreview(t, repo, context, "Escalate approved visual scenarios", visualEvidenceBody(managedPRBody(), context.PRVisualEvidenceStatus)) + preview, _, err := CheckPRPreview(repo, previewPath) + if err != nil { + t.Fatal(err) + } + _, err = PublishPR(PRPublishOptions{Repo: repo, PreviewPath: previewPath, ExpectedFingerprint: preview.Fingerprint, Action: "open"}) + if err == nil { + t.Fatal("escalated require did not block publication") + } + for _, needle := range []string{"required visual evidence", "require semantics", "capability-register"} { + if !strings.Contains(err.Error(), needle) { + t.Fatalf("escalated denial is missing %q:\n%s", needle, err.Error()) + } + } +} + +// The escalation predicate's full semantics table: `off` is the global +// opt-out, not_relevant the per-feature escape, and configured require needs +// no escalation. Capability availability is deliberately absent — a missing +// harness is a provisioning gap, not a license to ship unverified. +func TestVisualEscalationPredicate(t *testing.T) { + cases := []struct { + policy, relevance string + scenarios int + want bool + }{ + {"suggest", "relevant", 1, true}, + {"suggest", "relevant", 3, true}, + {"suggest", "relevant", 0, false}, + {"suggest", "not_relevant", 0, false}, + {"suggest", "unresolved", 1, false}, + {"require", "relevant", 1, false}, + {"off", "relevant", 1, false}, + } + for _, c := range cases { + if got := visualEscalationApplies(c.policy, c.relevance, c.scenarios); got != c.want { + t.Errorf("visualEscalationApplies(%s, %s, %d) = %v, want %v", c.policy, c.relevance, c.scenarios, got, c.want) + } + } +} + +// Invariant: a not_relevant plan decision (with its reason) keeps the +// configured suggest semantics — the per-feature escape for nonvisual changes. +func TestNotRelevantPlanKeepsSuggestSemantics(t *testing.T) { + repo := prTestRepoConfigured(t, func(config *ProjectConfig) { + config.Workflow.PRVisualEvidence = "suggest" + }) + directory := filepath.Join(repo, ".product-loop", "features", "log-rotation") + if err := os.MkdirAll(directory, 0o755); err != nil { + t.Fatal(err) + } + plan := validPlan() + plan["feature_id"] = "log-rotation" + plan["pr_visual_evidence"] = map[string]any{ + "relevance": "not_relevant", + "reason": "backend log rotation has no reviewer-visible surface", + } + writeMarkdownPlan(t, filepath.Join(directory, "plan.md"), plan, true) + config, _, err := LoadConfig(filepath.Join(repo, ".product-loop", "project.json")) + if err != nil { + t.Fatal(err) + } + policy, status, _, _, _, _, policySource, _, err := resolvePRVisualEvidence(repo, config, "managed", "log-rotation", "feat/log-rotation", strings.Repeat("d", 64)) + if err != nil { + t.Fatal(err) + } + if policy != "suggest" || policySource != "configured" || status != "NOT_APPLICABLE" { + t.Fatalf("not_relevant did not keep suggest semantics: policy=%s source=%s status=%s", policy, policySource, status) + } +} + // Invariant: declared-relevant visual evidence is captured by ship itself, // never prescribed to the agent, and repeat context preparation is a no-op // while the product diff is unchanged. @@ -783,9 +876,9 @@ func TestPreparePRContextAutoCapturesRelevantVisualEvidence(t *testing.T) { } } -// Invariant: a failing harness degrades to exactly today's suggest behavior — -// a visible NOT_VERIFIED gap with a bounded detail, never a context error. -func TestAutoCaptureFailureDegradesToTodayUnderSuggest(t *testing.T) { +// Invariant: a failing harness never errors context preparation — it records +// a bounded detail, and the plan-escalated require semantics hold the block. +func TestAutoCaptureFailureRecordsBoundedDetail(t *testing.T) { repo := prTestRepoConfigured(t, func(config *ProjectConfig) { config.Workflow.PRVisualEvidence = "suggest" config.Project.Commands["visual"] = "repo-owned-harness" @@ -798,8 +891,8 @@ func TestAutoCaptureFailureDegradesToTodayUnderSuggest(t *testing.T) { if err != nil { t.Fatal(err) } - if context.PRVisualEvidenceStatus != "NOT_VERIFIED" { - t.Fatalf("harness failure should leave the recorded gap, got %s", context.PRVisualEvidenceStatus) + if context.PRVisualEvidenceStatus != "BLOCKED" { + t.Fatalf("escalated require should block on a failed capture, got %s", context.PRVisualEvidenceStatus) } if !strings.Contains(context.PRVisualEvidenceCaptureDetail, "dev server unreachable") { t.Fatalf("capture detail lost the harness failure: %q", context.PRVisualEvidenceCaptureDetail) @@ -821,7 +914,7 @@ func TestAutoCaptureUnavailableCapabilityFallsBackToPrescribedPath(t *testing.T) if err != nil { t.Fatal(err) } - if runner.calls != 0 || context.PRVisualEvidenceStatus != "NOT_VERIFIED" { + if runner.calls != 0 || context.PRVisualEvidenceStatus != "BLOCKED" { t.Fatalf("unavailable capability did not fall back cleanly: calls=%d status=%s", runner.calls, context.PRVisualEvidenceStatus) } if !strings.Contains(context.PRVisualEvidenceCaptureDetail, "capability-register") { diff --git a/boatstack/references/config-schema.md b/boatstack/references/config-schema.md index 2d29ce7..e5df486 100644 --- a/boatstack/references/config-schema.md +++ b/boatstack/references/config-schema.md @@ -79,7 +79,7 @@ This is the exhaustive serialization contract, not a list of recommended user ed - `allow_pass_with_gaps` (boolean, optional): Deterministic gate control. `false` rejects `PASS_WITH_GAPS`; `true` preserves explicit gaps. - `maintain_changelog` (boolean, optional): Whether a reader-visible `CHANGELOG.md` entry is required for each delivery slice. - `boundary_analysis` (boolean, optional): Agent-mediated planning guidance that presents local repair versus programmatic enforcement as a material product decision. -- `pr_visual_evidence` (string, optional): `off`, `suggest`, or `require`. Omission is `off`. Relevant PRs use machine-local PNG evidence without committing media to Git; `suggest` records missing evidence as a visible gap and `require` blocks completed publication. +- `pr_visual_evidence` (string, optional): `off`, `suggest`, or `require`. Omission is `off`. Relevant PRs use machine-local PNG evidence without committing media to Git; `suggest` records missing evidence as a visible gap and `require` blocks completed publication. When the approved plan declares `relevance: relevant` with scenarios, `suggest` ships with require semantics for that feature (a plan that promises pixels cannot ship without them) — even when no capture capability is registered yet. The two escapes are `off` (global) and a `not_relevant` plan decision with a reason (per feature, for genuinely nonvisual changes). Boatstack runs a registered capture command (`project.commands.visual`) automatically during ship, so under normal provisioning the escalation is invisible. - `visual_evidence_publish` (object, optional): Agent-mediated publish control for how captured PNG bytes reach the pull-request comment. Omission keeps the default: commit the bytes to a public Boatstack-owned evidence branch and render them inline, but only for a **public** GitHub origin (a private origin falls back to manual attachment). Fields: - `mode` (string, optional): `external-host` opts the repository — including a **private** one — into uploading the exact PNG bytes to an anonymous expiring host so the comment renders inline anywhere. It is **never auto-selected** because it publishes screenshot bytes to a third party; only this explicit value turns it on. Empty keeps the default public-branch behavior. - `host` (string, optional): `litterbox` (default) or `catbox`. Only meaningful when `mode` is `external-host`. `litterbox` auto-expires uploads; `catbox` is permanent. diff --git a/boatstack/safety.go b/boatstack/safety.go index 6d15550..e301fa6 100644 --- a/boatstack/safety.go +++ b/boatstack/safety.go @@ -26,6 +26,9 @@ type SafetyFinding struct { AttemptedPath string `json:"attempted_path,omitempty"` OperationID string `json:"operation_id,omitempty"` OperationState string `json:"operation_state,omitempty"` + // PolicySource explains a policy-derived denial: "plan-escalated" when a + // plan-approved visual decision lifts suggest to require semantics. + PolicySource string `json:"policy_source,omitempty"` AttemptNumber int `json:"attempt_number,omitempty"` ReconciliationRequired bool `json:"reconciliation_required,omitempty"` // RepeatCount is how many times this same denial (category at stage) has diff --git a/boatstack/visual_evidence.go b/boatstack/visual_evidence.go index 1f22727..9fc9bfe 100644 --- a/boatstack/visual_evidence.go +++ b/boatstack/visual_evidence.go @@ -130,6 +130,19 @@ func normalizedPRVisualEvidencePolicy(value string) string { return value } +// visualEscalationApplies decides when the configured suggest policy ships +// with require semantics: the approved plan declares visual relevance with +// concrete scenarios. A plan that promises pixels cannot ship without them. +// The escalation is deliberately independent of capture-capability +// availability — a missing harness is a provisioning gap the publication +// denial names, never a license to ship unverified. `off` remains the global +// opt-out (the predicate never fires) and a not_relevant plan decision (with +// its reason) remains the per-feature escape for genuinely nonvisual changes. +// control-law: plan-approved-scenarios-imply-require +func visualEscalationApplies(configured, relevance string, scenarioCount int) bool { + return configured == "suggest" && relevance == "relevant" && scenarioCount > 0 +} + func visualEvidenceKey(mode, feature, head string) (string, error) { key := feature if mode == "ad-hoc" { diff --git a/docs/configuration.md b/docs/configuration.md index a316d21..ecd7a51 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -52,7 +52,7 @@ failed, or stale results. | Permit visible verification gaps | `workflow.allow_pass_with_gaps` | `false` rejects `PASS_WITH_GAPS` at delivery and PR gates; `true` retains the gaps as evidence. | | Maintain reader-facing history | `workflow.maintain_changelog` | Managed delivery and Boatstack-prepared PRs require a categorized `CHANGELOG.md` entry. | | Check for a systemic boundary | `workflow.boundary_analysis` | Planning guidance asks whether the request is a local symptom before scope expands. | -| Add frontend PR screenshots | `workflow.pr_visual_evidence` | `suggest` exposes missing screenshots as a gap; `require` blocks completed publication. | +| Add frontend PR screenshots | `workflow.pr_visual_evidence` | `suggest` exposes missing screenshots as a gap; `require` blocks completed publication. A plan that approves visual scenarios lifts `suggest` to require semantics for that feature; `off` and a per-feature `not_relevant` decision (with a reason) are the escapes. Boatstack captures registered scenarios automatically during ship. | | Render screenshots inline on a private PR | `workflow.visual_evidence_publish.*` | `mode: external-host` uploads the captured PNGs to an anonymous expiring host so the comment renders inline even on a private repo; opt-in, never automatic. | | Ignore old ambiguous deliveries | `workflow.ignored_deliveries` | Listed feature slugs are excluded from delivery-ambiguity resolution so past work stops blocking new work; new, unlisted ambiguous deliveries still pause. | | Pursue the PR to merge, not just to open | `delivery.terminal` | `merged` keeps the read-only flow advisors naming post-publish steps (watch checks, route corrections) until the PR is observed merged; the default `published` ends the flow when the PR is open, exactly as before. | diff --git a/docs/evidence-engineered-coding.md b/docs/evidence-engineered-coding.md index 0a5605b..2c27596 100644 --- a/docs/evidence-engineered-coding.md +++ b/docs/evidence-engineered-coding.md @@ -146,6 +146,6 @@ Delivery and system improvement also remain separate. A failed task may suggest ## What is evidence-backed -The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`f42f1a845ecc6e5905d6e38dca863749cc19f0c0`](https://github.com/operatorstack/intelligence-flow/tree/f42f1a845ecc6e5905d6e38dca863749cc19f0c0/labs/12-product-engineering-loop). +The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`146f50c0f3309e592ed45374b8cddb3d8ed64d85`](https://github.com/operatorstack/intelligence-flow/tree/146f50c0f3309e592ed45374b8cddb3d8ed64d85/labs/12-product-engineering-loop). The evidence supports specific failure mechanisms and guardrails. It does not establish that Boatstack is optimal, that control-theory notation proves software quality, or that one workflow dominates every team. Those are evaluation questions, so the distribution preserves measurements, provenance, gaps, and negative results. diff --git a/docs/public-claims.json b/docs/public-claims.json index 70ee921..099760a 100644 --- a/docs/public-claims.json +++ b/docs/public-claims.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "source_commit": "f42f1a845ecc6e5905d6e38dca863749cc19f0c0", + "source_commit": "146f50c0f3309e592ed45374b8cddb3d8ed64d85", "statuses": ["verified", "observed", "still_being_evaluated"], "claims": [ { @@ -12,7 +12,7 @@ "readable_evidence": "why-these-steps.md#portable-workflow-and-state", "implementation": ["../boatstack/export.go", "../boatstack/references/artifacts.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "human-decisions", @@ -23,7 +23,7 @@ "readable_evidence": "why-these-steps.md#human-decisions", "implementation": ["../boatstack/references/workflow.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "validation-provenance", @@ -34,7 +34,7 @@ "readable_evidence": "why-these-steps.md#validation-provenance", "implementation": ["validation-and-evidence.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "irreversible-operations", @@ -46,7 +46,7 @@ "readable_evidence": "why-these-steps.md#irreversible-operations", "implementation": ["safety.md", "../boatstack/safety.go", "../boatstack/hooks.go"], "verification": ["../boatstack/safety_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "reviewer-ready-pr", @@ -57,7 +57,7 @@ "readable_evidence": "why-these-steps.md#reviewer-ready-pr", "implementation": ["../boatstack/pr.go", "getting-started.md"], "verification": ["../boatstack/pr_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "phase-scoped-delivery", @@ -68,7 +68,7 @@ "readable_evidence": "why-these-steps.md#phase-scoped-delivery", "implementation": ["../boatstack/delivery.go", "../boatstack/safety.go", "../boatstack/hooks.go", "../boatstack/references/workflow.md"], "verification": ["../boatstack/delivery_test.go", "../boatstack/pr_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "model-neutral-contract", @@ -79,7 +79,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "cross-model-failures", @@ -90,7 +90,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "lower-cost-outcomes", @@ -101,7 +101,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "git-worktree-activation", @@ -112,7 +112,7 @@ "readable_evidence": "why-these-steps.md#git-worktree-activation", "implementation": ["../boatstack/runtime_cache.go", "../boatstack/hooks.go"], "verification": ["../boatstack/runtime_cache_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" }, { "id": "visible-updates", @@ -123,7 +123,7 @@ "readable_evidence": "why-these-steps.md#visible-updates", "implementation": ["../boatstack/update.go", "../boatstack/init.go"], "verification": ["../boatstack/update_test.go", "../boatstack/init_test.go", "../boatstack/export_test.go"], - "last_verified_version": "source:f42f1a845ecc6e5905d6e38dca863749cc19f0c0" + "last_verified_version": "source:146f50c0f3309e592ed45374b8cddb3d8ed64d85" } ] } diff --git a/labs/diagram-json/plan.lock.json b/labs/diagram-json/plan.lock.json index 408fd27..6f71efc 100644 --- a/labs/diagram-json/plan.lock.json +++ b/labs/diagram-json/plan.lock.json @@ -6,7 +6,7 @@ "plan_path": "labs/diagram-json/plan.md", "plan_sha256": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "schema_version": 1, - "source_commit": "f42f1a845ecc6e5905d6e38dca863749cc19f0c0", + "source_commit": "146f50c0f3309e592ed45374b8cddb3d8ed64d85", "source_plan_path": "labs/diagram-json/source-plan.md", "source_plan_sha256": "e10593ddaa7522ab80cc991d0a09399257139799e37f737794cd49d68a39985b", "spec_path": "labs/diagram-json/spec.md", diff --git a/release-notes/2026-07-30-plan-approved-scenarios-escalate-to-require.md b/release-notes/2026-07-30-plan-approved-scenarios-escalate-to-require.md new file mode 100644 index 0000000..27fc32e --- /dev/null +++ b/release-notes/2026-07-30-plan-approved-scenarios-escalate-to-require.md @@ -0,0 +1,3 @@ +### Plan-approved visual scenarios now ship with require semantics + +When the approved plan declares `pr_visual_evidence` relevance `relevant` with scenarios, the configured `suggest` policy escalates to require semantics for that feature: publication blocks until current PASS evidence exists, even when no capture capability is registered yet. A plan that promises pixels can no longer ship with a `NOT_VERIFIED` gap. The escapes stay explicit: `off` globally, or a per-feature `not_relevant` decision with a reason for genuinely nonvisual changes. Because Boatstack now captures registered scenarios automatically during ship, provisioned repositories will not notice the escalation. One-time effect after upgrading: an existing `suggest` preview for a feature with declared scenarios reports a changed context fingerprint — regenerate the preview with `pr-context` before publishing.