diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 870f7a9..c67a85b 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,7 +2,7 @@ # Contributing -Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/5e9cd972c1a75c067b73625cf411ecd290086715/labs/12-product-engineering-loop). +Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/6ec978238627ee8b8f072feeabbada0ccf73420e/labs/12-product-engineering-loop). The Boatstack repository receives product/runtime changes through a generated pull request. Review the PR's `UPSTREAM.json`, tests, adapter diff, and context-size change; do not hand-edit generated output on `main`. `.github/workflows` is the exception: it is Boatstack's executable control plane, excluded from scheduled projection and changed only through a separate manually reviewed Boatstack PR. diff --git a/UPSTREAM.json b/UPSTREAM.json index 6348705..8753b91 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -1,7 +1,7 @@ { "canonical_context": { - "characters": 83659, - "estimated_tokens": 20915, + "characters": 83752, + "estimated_tokens": 20938, "estimator": "ceil(total characters / 4); compactness signal, not provider billing", "files": [ "product-engineering-loop/references/workflow.md", @@ -12,7 +12,7 @@ }, "files": { ".gitignore": "a7079e923a776f14f1bb3a6aa0a11a133a8e1dfb35af020f327623357b7e3957", - "CONTRIBUTING.md": "6833ef30ef884c47136f2bc2e77eec791b723c6cdf561c223cb114f1208dc9a7", + "CONTRIBUTING.md": "34bc79252efdcaff7a67cf46d6eb297c58f2041e5123ce013c3f82dfeb1ddd3a", "README.md": "3ce3e95e511089b44e946a44b8d5f4f81d019ece5336db65b2cab1f9dc4d4dad", "assets/boatstack-journey.svg": "e465befc50c8ce30f3e07e8fd97012931beeb053392c8fbf38ad645023b3cc63", "assets/boatstack-mark.svg": "be1f984da1bfa69fa5d1f986d8343d21f7e20921b71db888c928b4d2e54b09b5", @@ -61,7 +61,9 @@ "boatstack/delivery_reactivation_test.go": "573a2dba0034bc4290478414e3bdd8670b06a326128eb0295d77e748ecc8689e", "boatstack/delivery_test.go": "45c48ff7581c911bcaf821c3e4241d4ae2a9bb4aa682485cc58b6ad8fe1c85bf", "boatstack/deliverycontrol_parity_test.go": "f8662cfc35043395a0e1eef8a87051c2120752f38b09c56b78f82896008f1b65", - "boatstack/denial.go": "fabc517434048c2c7327f0b123baa5e9594507f6d4be460d742f9e8f59d2d0b8", + "boatstack/denial.go": "ce77240edafb2589565ff3821b4e2cf8be3cf5884b1467d606265ca877188d3e", + "boatstack/denial_escalation_conformance_test.go": "d50f0e1c803c8f5731a46dbe6f82c7935c0ccbd582f513cbe07211f5e902157e", + "boatstack/denial_ledger.go": "a35bf8fd8c1f6b9302109cf0087e1b9158e43e07b92b18df5e589a637d58491d", "boatstack/denial_solutions.go": "5e5f110ee62bb08f65e8240b85d34a16cea5cb19c1993e4b39f2e83addfaf95d", "boatstack/denial_solutions_conformance_test.go": "1055048d1d7935dded699abdb7015b30a966e8e1ae54b1253c106dde54aff10e", "boatstack/denial_test.go": "9dc9f0f79328c4947073efaa785479b34e70eb214da57cd72348f39fd672e4fd", @@ -134,7 +136,7 @@ "boatstack/next_test.go": "6b5ec46ecf1a197d7644846cecbb6d99873a06b7c4e5562772b5016fa0a4cb11", "boatstack/operation.go": "073113e1e7b6349417e70b704bd1a342b460604cbd97a7fab06b1a6494604112", "boatstack/operation_test.go": "59d3dc37319aa4d334c0cacbe886e2f757842e6a28dee8781448e528fecbde11", - "boatstack/paths.go": "bc14901f9497bfa87f64eab80ff30cc7c3a31781e3fdb60826df2dc21eb2253b", + "boatstack/paths.go": "f9a615f35e0f439d6f11c48e881f11db26c0029e1eb7dd5978941cf51c74e710", "boatstack/plan.go": "e1e4344f2aae24f56cc767291616947e1bd5ee08fb6e73e60f554215c799590c", "boatstack/plan_test.go": "53477515165a910910b9175bfa33574548cf0d0f3be48e175ec3a775a12d30bb", "boatstack/plan_validation.go": "99e40806fd579ff72de53f391cd6124acc9ff18707ecf9676c5e507b738d87d0", @@ -154,7 +156,7 @@ "boatstack/reexec.go": "fed55416479d7bd3e0c3637057ffe8eb58a032f93fc358f76df906ab7acc677b", "boatstack/reexec_unix.go": "ff86157a9aa20c82a56fcd859b70669b7eacf4e0a9f61a4546ef33808437939e", "boatstack/reexec_windows.go": "f5335c8c28cb4e89048b058b1c4d12f78644f99acb4f6167ff60e622dfb9e742", - "boatstack/references/artifacts.md": "9589849796553dfb433c3d49f2d75f49fd2b50e03068f481c41776f1d4cef066", + "boatstack/references/artifacts.md": "4f27170227ffa2715d632eab4cbdca229726c5fc03c4b0f2af980fa50bb24725", "boatstack/references/config-schema.md": "eff8586850eca9941df1cea29ac7edb779277ed9bd6d938a1980db5028d28cf4", "boatstack/references/failure-moves.md": "b65ef72035afa6ad0dce589a0b38f84bc40cde3864c9ecf973f08fc687f001c3", "boatstack/references/host-hook-contracts.md": "2a89d44d0e418a53f2e3b6300fed957cdf878f45ea97ce24b55b66065f0eaa1d", @@ -170,15 +172,15 @@ "boatstack/runtime_cache.go": "e026ffc1906f7e1e98b768bae63e6658164d2826c07169c9121ce0f23c73faf8", "boatstack/runtime_cache_test.go": "b981467ddc9f0f562da6bff5de7a80a9fe5a433a0317541d1e48df268546ac85", "boatstack/runtime_provenance_test.go": "1d52f1e6b0691cf4667729cc9b9f3c55c128f0aa3321f3a2843a9aa6fd0e73dc", - "boatstack/safety.go": "432e20956789c74dfefc4b174238f320876c9e68ad6c6392286c6acdbdb5ef6d", + "boatstack/safety.go": "2f274b90a2b50798bc5a19a74ddaf4f2eecd66876234d21b3753406f69522170", "boatstack/safety_corpus_test.go": "a43fdc4324abc9db0d93e407336a2b0ca5b755121013ae62ce7d844795a6d8b9", "boatstack/safety_test.go": "ddd7a72d4ff50c046aa46fd97b108d629cef27cf15816150b6d54baf5f1c4c36", "boatstack/safety_update_publisher_test.go": "ed3f8187036623694dfe7c395cdae00fdae14609bab6124d1fdfc6fe73fa2196", "boatstack/skill_frontmatter.go": "73364df463ce828c2d005aab55f72bb92f7a34d99cf3f53d4e0cd5a4da9dbd0e", "boatstack/skill_frontmatter_test.go": "a3ec52e7df357a72265c95dd66db15d9c0effc7e5f90f14ce69c27792ce394eb", "boatstack/solution_closure_conformance_test.go": "f73e6748dac373e2a10bc9269c2f2e060bd220bb4113bd0d5ff66ca4c8e91a54", - "boatstack/statemap.go": "39ff3a7a7254ec5fde8340551fa92a82aac3f00a48bea34c94192823dcb41337", - "boatstack/statemap_conformance_test.go": "38950377d10b97f4223b79cdb16b17c29a6334f8f2a9a4d826ce12cbd292aa57", + "boatstack/statemap.go": "27aabde21c5dbfa9526f1533578c1e512ddd3174606457b67da344c8a6c36733", + "boatstack/statemap_conformance_test.go": "e418f88bd91ab466bc64d0965ff5c404d7810479de509b01ec345826dfef978a", "boatstack/supervisory_control_test.go": "c7ea4bcd678e8ec211dac772c834981c4e21762914be2770a5e181bc24605e06", "boatstack/testdata/reviewer-pr-body.md": "4c64e3788e5d61a377aeb0f797f7fc8d2316ab6e49572d15636eea7ba9e34ac4", "boatstack/testdata/safety/safe_apply.py.txt": "c9ec7fb932cf21b6aa8df597c4d4c54d6ec65e796240e49118d699f583383975", @@ -201,10 +203,10 @@ "docs/benchmark-corpus-audit.md": "f2d206fe8579a514f9da82b2c96c19b343ac004be67617e1bd34f0f8e0e5e6c6", "docs/benchmark-submission-audit.md": "9518abdd17690729c6423f87cab20418ed47b0915b5faa44b9ef975e9e9c3b79", "docs/configuration.md": "060775c73431f28bd16066bdf9e0f89034d2855c7ca0f5544f660d24b91211d0", - "docs/evidence-engineered-coding.md": "f2c0402bd0608ffb0ee40d797d7ca464683fdc20ff3b89cdf62de37fc52ce855", + "docs/evidence-engineered-coding.md": "731c8ea8b4ade606ff5b4293a2ceb3e61ce3e749a1b7ee9803e71bb0f904f607", "docs/generated-files.md": "437791765b0a4015032ae21d1a6618563cad92b7402819e4f963bf5ae16284a3", "docs/getting-started.md": "51c2823f21e35140d31e6d5083dc4b89fddd24721ac6acc474154a4da53ee9f8", - "docs/public-claims.json": "9387bb9b0de3aeba92f1558bb27aefd3faf17a89284cde234857b1aaafdcc6ed", + "docs/public-claims.json": "048dcc89114a78f2ef369bdfb007f336511878ec1ff8deb1317cb180091af673", "docs/public-surface.md": "713f7a050b5f339cf948299103ef3800417dccfecf2cc1a4166397ea6f978907", "docs/research-and-design.md": "8d78678108f0a6c924e1ff9b32c0f81aae9d1f779e0082843b6f99ad993ae2b6", "docs/safety.md": "7b9b5c515d36e683767ec8d3d9d6d119ac93650b2f629d351deadd4c600ed6a6", @@ -218,7 +220,7 @@ "labs/diagram-json/compiled/evidence.md": "1ba1c989ade070a8ef9a508fbd788d100d7292f2dbacbb2bce895468019f619d", "labs/diagram-json/compiled/tasks.json": "88f60851abf79d851e9fccc754ff3040034ae595306bc87d64784c19eb403e71", "labs/diagram-json/compiled/test-matrix.json": "424657ff505768e50fa113801fd8363364a18269d5297480907a993d44063a39", - "labs/diagram-json/plan.lock.json": "96dc2d5722262e62271f1488b158f726f277b0e6d63fffed95f474c2962587ab", + "labs/diagram-json/plan.lock.json": "66f2cf27de629c86726133ec64a9982eb7b1aebad923e652c293c186abe76580", "labs/diagram-json/plan.md": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "labs/diagram-json/questions.md": "74733b015002c8a6777c558e7e997fa48c94850b9bd39054fe9366c97ecf728d", "labs/diagram-json/request.md": "0808fc41c36779c404f4a3a121167da6e76cac56df526e70f9ed6d3e0d4c02ed", @@ -344,13 +346,14 @@ "release-notes/2026-07-27-invalid-delivery-block-actionable.md": "8fac8e3921e2285291703efa46e624b72cb5bac1b8492beca4c4b633abb5ba16", "release-notes/2026-07-27-prescriptive-planning-closure.md": "e544408e1c3cceb0cb1979833ea120853c38020933439b39e7f009f454b9661e", "release-notes/2026-07-27-read-only-inspection-pipelines.md": "0963286371e9a12592915c23a958dd013bf2a35e9fca6921691bc8bb3c3d8dc8", + "release-notes/2026-07-27-repeated-denials-escalate.md": "ee54b597e593ce74d8d9acbc67c74d2d8cb972ff206f618e08a047d4d962d7af", "release-notes/2026-07-27-sandboxed-migration-grading.md": "03cebc372bbdfed37cc70d18f3b6374d1aa5e585bafefa073dbcced58bd0336a", "release-notes/2026-07-27-state-ownership-map.md": "d032547aafc1a4acbeb520f6cbb59d7757de4f33fe824701d7b5ea8cd8c8b9e7" }, "generator": "operatorstack/intelligence-flow:boatstack-distribution", "schema_version": 1, "source": { - "commit": "5e9cd972c1a75c067b73625cf411ecd290086715", + "commit": "6ec978238627ee8b8f072feeabbada0ccf73420e", "path": "labs/12-product-engineering-loop", "repository": "operatorstack/intelligence-flow" } diff --git a/boatstack/denial.go b/boatstack/denial.go index eca7d54..a537745 100644 --- a/boatstack/denial.go +++ b/boatstack/denial.go @@ -70,6 +70,31 @@ type Denial struct { // denials), derived from the state-ownership map. Named, never compiled // into runnable commands — their full arguments are not derivable here. OwnerVerbs []string + // Escalated marks a denial that has repeated past the ledger threshold: + // the rendering lifts the pick-list cap to the full set and prescribes a + // fresh diagnostic. More corrective information, never more severity — + // what is denied never changes. + // control-law: repeated-denials-escalate-to-solutions + Escalated bool + RepeatCount int +} + +// optionTextLimit is the pick-list cap for the text renderings: compact by +// default, the full structured cap once the denial has escalated. +func (d Denial) optionTextLimit() int { + if d.Escalated { + return solutionSetCap + } + return solutionSetTextCap +} + +// escalationLine is the repeat notice with the fresh-probe prescription; empty +// until the denial escalates. +func (d Denial) escalationLine() string { + if !d.Escalated { + return "" + } + return fmt.Sprintf("This denial repeated %d times. Run: boatstack-helper doctor --repo .", d.RepeatCount) } // --- ANSI palette (truecolor; matches the approved mockup) ------------------- @@ -160,10 +185,13 @@ func (d Denial) renderPlain(badge string) string { if len(d.OwnerVerbs) > 0 { b.WriteString("\n\nThis path is owned by: " + strings.Join(d.OwnerVerbs, ", ") + ".") } - if lines := d.optionLines(solutionSetTextCap); len(lines) > 0 { + if lines := d.optionLines(d.optionTextLimit()); len(lines) > 0 { b.WriteString("\n\nYou can:\n") b.WriteString(strings.Join(lines, "\n")) } + if escalation := d.escalationLine(); escalation != "" { + b.WriteString("\n\n" + escalation) + } if d.Hint != "" { b.WriteString("\n\nFalse positive? run: ") b.WriteString(d.Hint) @@ -186,12 +214,15 @@ func (d Denial) renderMarkdown(badge string) string { if len(d.OwnerVerbs) > 0 { b.WriteString("\n\nThis path is owned by: `" + strings.Join(d.OwnerVerbs, "`, `") + "`.") } - if lines := d.optionLines(solutionSetTextCap); len(lines) > 0 { + if lines := d.optionLines(d.optionTextLimit()); len(lines) > 0 { b.WriteString("\n\nYou can:\n") for _, line := range lines { b.WriteString("\n" + line) } } + if escalation := d.escalationLine(); escalation != "" { + b.WriteString("\n\n" + escalation) + } if d.Hint != "" { b.WriteString("\n\nFalse positive? run `" + d.Hint + "`") } @@ -214,12 +245,15 @@ func (d Denial) renderANSI(badge string) string { if len(d.OwnerVerbs) > 0 { b.WriteString("\n" + fgGray + "this path is owned by: " + ansiReset + fgCode + strings.Join(d.OwnerVerbs, ", ") + ansiReset) } - if lines := d.optionLines(solutionSetTextCap); len(lines) > 0 { + if lines := d.optionLines(d.optionTextLimit()); len(lines) > 0 { b.WriteString("\n" + fgGray + "you can:" + ansiReset) for _, line := range lines { b.WriteString("\n" + fgCode + line + ansiReset) } } + if escalation := d.escalationLine(); escalation != "" { + b.WriteString("\n" + fgGray + escalation + ansiReset) + } if d.Hint != "" { b.WriteString("\n" + fgGray + ansiDim + "false positive? run " + ansiReset + fgCode + d.Hint + ansiReset) } @@ -293,6 +327,10 @@ func (d Denial) Structured() map[string]any { if len(d.OwnerVerbs) > 0 { out["owner_verbs"] = d.OwnerVerbs } + if d.Escalated { + out["escalated"] = true + out["repeat_count"] = d.RepeatCount + } return out } @@ -401,6 +439,10 @@ func denialWithOptions(repo, host string, finding SafetyFinding) Denial { if finding.Category == "workflow-state-tamper" { d.OwnerVerbs = tamperOwnerVerbs(repo, finding.AttemptedPath) } + if finding.RepeatCount >= denialEscalationThreshold { + d.Escalated = true + d.RepeatCount = finding.RepeatCount + } return d } diff --git a/boatstack/denial_escalation_conformance_test.go b/boatstack/denial_escalation_conformance_test.go new file mode 100644 index 0000000..72af46f --- /dev/null +++ b/boatstack/denial_escalation_conformance_test.go @@ -0,0 +1,176 @@ +package boatstack + +import ( + "fmt" + "os" + "path/filepath" + "strings" + "testing" +) + +// control-law: repeated-denials-escalate-to-solutions +// +// Repetition of one denial is the signal that the stated law is not reaching +// the model (the sibling harness's paid canary: sixteen identical no-progress +// repair attempts, then a protected-boundary write). A repeat denial escalates +// its corrective INFORMATION — the full solution set plus a fresh diagnostic +// probe — never its severity or its admissibility. Boundaries held here: +// the per-worktree ledger (recordDenial/resetDenialLedger), the HookDecision +// deny/allow wiring, and the escalated rendering. The constitutional corpus +// floor is held separately by safety_corpus_test.go and is untouched: this law +// changes what a denial says, never what is denied. + +func tamperEvent(path string) []byte { + return []byte(fmt.Sprintf( + `{"hook_event_name":"PreToolUse","tool_name":"Write","tool_input":{"file_path":%q,"content":"{}"}}`, path)) +} + +// Positive: the third identical denial escalates — the rendering carries the +// repeat notice and the fresh-probe prescription; the first two do not. +func TestThirdIdenticalDenialEscalates(t *testing.T) { + repo := safetyTestRepo(t) + event := tamperEvent(".git/boatstack/deliveries/demo/state.json") + for attempt := 1; attempt <= denialEscalationThreshold; attempt++ { + output, denied := HookDecision(SafetyHookOptions{Host: "claude", Repo: repo, Input: event}) + if !denied { + t.Fatalf("attempt %d: tamper write must be denied", attempt) + } + escalated := strings.Contains(string(output), "This denial repeated") + if attempt < denialEscalationThreshold && escalated { + t.Fatalf("attempt %d escalated before the threshold:\n%s", attempt, output) + } + if attempt == denialEscalationThreshold { + if !escalated { + t.Fatalf("attempt %d must escalate:\n%s", attempt, output) + } + if !strings.Contains(string(output), fmt.Sprintf("repeated %d times", denialEscalationThreshold)) { + t.Fatalf("escalation must carry the repeat count:\n%s", output) + } + if !strings.Contains(string(output), "boatstack-helper doctor") { + t.Fatalf("escalation must prescribe the fresh diagnostic:\n%s", output) + } + } + } +} + +// Negative/reset: an ALLOWED mutation-capable call is forward progress and +// clears the ledger — the next denial starts unescalated. +func TestAllowedMutationResetsTheLedger(t *testing.T) { + repo := safetyTestRepo(t) + event := tamperEvent(".git/boatstack/deliveries/demo/state.json") + for i := 0; i < denialEscalationThreshold-1; i++ { + if _, denied := HookDecision(SafetyHookOptions{Host: "claude", Repo: repo, Input: event}); !denied { + t.Fatal("tamper write must be denied") + } + } + allowed := []byte(fmt.Sprintf( + `{"hook_event_name":"PreToolUse","tool_name":"Write","tool_input":{"file_path":%q,"content":"export const x = 1"}}`, + filepath.Join(repo, "src", "app.ts"))) + if _, denied := HookDecision(SafetyHookOptions{Host: "claude", Repo: repo, Input: allowed}); denied { + t.Fatal("ordinary product write must be allowed") + } + output, denied := HookDecision(SafetyHookOptions{Host: "claude", Repo: repo, Input: event}) + if !denied { + t.Fatal("tamper write must still be denied after the reset") + } + if strings.Contains(string(output), "This denial repeated") { + t.Fatalf("ledger must reset after allowed progress:\n%s", output) + } +} + +// Relation: different denial keys never cross-escalate, and a read-only +// allowed call does NOT reset the ledger (only mutation progress does). +func TestDenialKeysAreIsolated(t *testing.T) { + repo := safetyTestRepo(t) + tamper := SafetyFinding{Category: "workflow-state-tamper", Source: "delivery-state"} + phase := SafetyFinding{Category: "workflow-phase-bypass", WorkflowStage: "DRAFT_PLAN", Source: "planning-state"} + if got := recordDenial(repo, tamper); got != 1 { + t.Fatalf("first tamper count = %d, want 1", got) + } + if got := recordDenial(repo, tamper); got != 2 { + t.Fatalf("second tamper count = %d, want 2", got) + } + if got := recordDenial(repo, phase); got != 1 { + t.Fatalf("a different category must count from 1, got %d", got) + } + // Same category at a different stage is a different key. + other := SafetyFinding{Category: "workflow-phase-bypass", WorkflowStage: "APPROVED", Source: "planning-state"} + if got := recordDenial(repo, other); got != 1 { + t.Fatalf("same category at a different stage must count from 1, got %d", got) + } + // A read-only allowed call must not reset: HookDecision only resets on + // mutation-capable allows, so the tamper key keeps its count. + readonly := []byte(`{"hook_event_name":"PreToolUse","tool_name":"Bash","tool_input":{"command":"git status"}}`) + if _, denied := HookDecision(SafetyHookOptions{Host: "claude", Repo: repo, Input: readonly}); denied { + t.Fatal("git status must be allowed") + } + if got := recordDenial(repo, tamper); got != 3 { + t.Fatalf("read-only allow must not reset the ledger; tamper count = %d, want 3", got) + } +} + +// Failure-state: bookkeeping degrades, the denial never does. A corrupt ledger +// starts fresh; an unresolvable repository still yields a usable count; the +// escalated rendering is identical in admissibility to the unescalated one. +func TestLedgerFailuresDegradeCalmly(t *testing.T) { + repo := safetyTestRepo(t) + path, ok := denialLedgerPath(repo) + if !ok { + t.Fatal("ledger path must resolve in a git fixture") + } + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte("{not json"), 0o644); err != nil { + t.Fatal(err) + } + finding := SafetyFinding{Category: "workflow-state-tamper", Source: "delivery-state"} + if got := recordDenial(repo, finding); got != 1 { + t.Fatalf("corrupt ledger must start fresh, got %d", got) + } + if got := recordDenial(t.TempDir(), finding); got < 1 { + t.Fatalf("an unresolvable ledger must still return a usable count, got %d", got) + } +} + +// Bypass/bound: the ledger is pruned to its key bound, oldest first, so a +// category-spraying session cannot grow unbounded per-worktree state. +func TestLedgerStaysBounded(t *testing.T) { + repo := safetyTestRepo(t) + for i := 0; i < denialLedgerMaxKeys+8; i++ { + recordDenial(repo, SafetyFinding{Category: fmt.Sprintf("category-%02d", i), Source: "test"}) + } + path, _ := denialLedgerPath(repo) + ledger := loadDenialLedger(path) + if len(ledger.Counts) > denialLedgerMaxKeys { + t.Fatalf("ledger holds %d keys, bound is %d", len(ledger.Counts), denialLedgerMaxKeys) + } +} + +// Rendering: escalation lifts the pick cap to the full structured set and the +// structured payload carries escalated/repeat_count; severity is unchanged. +func TestEscalatedRenderingLiftsTheCapNotTheSeverity(t *testing.T) { + finding := SafetyFinding{ + Category: "workflow-phase-bypass", Source: "planning-state", + WorkflowStage: "DRAFT_PLAN", NextOperation: "plan-gate", BlockingFeature: "demo", + } + calm := denialWithOptions(".", "claude", finding) + finding.RepeatCount = denialEscalationThreshold + escalated := denialWithOptions(".", "claude", finding) + if !escalated.Escalated || escalated.Severity != calm.Severity || escalated.Category != calm.Category { + t.Fatalf("escalation must change information only: %+v vs %+v", escalated, calm) + } + if len(escalated.Options) != len(calm.Options) { + t.Fatalf("the option SET is identical; only the rendered cap lifts") + } + if escalated.optionTextLimit() <= calm.optionTextLimit() && len(calm.Options) > calm.optionTextLimit() { + t.Fatal("escalated rendering must show more of the set") + } + structured := escalated.Structured() + if structured["escalated"] != true || structured["repeat_count"] != denialEscalationThreshold { + t.Fatalf("structured payload must carry escalation: %v", structured) + } + if _, present := calm.Structured()["escalated"]; present { + t.Fatal("unescalated payload must not carry the escalation keys") + } +} diff --git a/boatstack/denial_ledger.go b/boatstack/denial_ledger.go new file mode 100644 index 0000000..b945c45 --- /dev/null +++ b/boatstack/denial_ledger.go @@ -0,0 +1,129 @@ +package boatstack + +import ( + "encoding/json" + "os" + "path/filepath" + "sort" + "time" +) + +// The sibling harness's paid canary recorded the trajectory this law exists to +// stop: an agent under repair pressure repeated the same denied move sixteen +// times and then escalated into a protected-boundary write. Repetition of one +// denial is the signal that the stated law is not reaching the model — so a +// repeat denial escalates its corrective information (the full solution set +// plus a fresh diagnostic probe), never its severity. The law is pure +// optimization: it changes WHAT a denial says, never what is denied or +// allowed, and a broken ledger degrades to the unescalated rendering — a +// denial must never turn into a crash or an allow because bookkeeping failed. +// control-law: repeated-denials-escalate-to-solutions + +const ( + // denialEscalationThreshold is the identical-denial count at which the + // rendering escalates (matches the fresh-probe discipline: two repeats of + // the same failure without a new probe is thrash). + denialEscalationThreshold = 3 + // denialLedgerMaxKeys bounds the ledger; the oldest keys are pruned. + denialLedgerMaxKeys = 32 +) + +type denialLedgerEntry struct { + Count int `json:"count"` + LastUnix int64 `json:"last_unix"` +} + +type denialLedger struct { + Counts map[string]denialLedgerEntry `json:"counts"` +} + +// denialLedgerKey identifies "the same denial": the category at the workflow +// stage it fired. Two different categories never cross-escalate. +func denialLedgerKey(finding SafetyFinding) string { + return finding.Category + "\x00" + finding.WorkflowStage +} + +func denialLedgerPath(repo string) (string, bool) { + w, err := ResolveWorkspaceContext(repo) + if err != nil { + return "", false + } + base, err := w.GuardDir() + if err != nil { + return "", false + } + return filepath.Join(base, "denials.json"), true +} + +func loadDenialLedger(path string) denialLedger { + ledger := denialLedger{Counts: map[string]denialLedgerEntry{}} + value, err := os.ReadFile(path) + if err != nil { + return ledger + } + // An unreadable or corrupt ledger starts fresh — fail-calm, never fail-open + // or crash. + var parsed denialLedger + if json.Unmarshal(value, &parsed) == nil && parsed.Counts != nil { + ledger = parsed + } + if ledger.Counts == nil { + ledger.Counts = map[string]denialLedgerEntry{} + } + return ledger +} + +func saveDenialLedger(path string, ledger denialLedger) { + // Prune to the bound, dropping the oldest keys first. + if len(ledger.Counts) > denialLedgerMaxKeys { + type aged struct { + key string + last int64 + } + entries := make([]aged, 0, len(ledger.Counts)) + for key, entry := range ledger.Counts { + entries = append(entries, aged{key, entry.LastUnix}) + } + sort.Slice(entries, func(i, j int) bool { return entries[i].last > entries[j].last }) + for _, stale := range entries[denialLedgerMaxKeys:] { + delete(ledger.Counts, stale.key) + } + } + value, err := json.Marshal(ledger) + if err != nil { + return + } + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + return // fail-calm: persistence is best-effort + } + _ = os.WriteFile(path, value, 0o644) +} + +// recordDenial bumps the finding's repeat count in the per-worktree ledger and +// returns the new count. Every failure mode returns a usable count of at least +// 1 — bookkeeping can degrade the escalation, never the denial itself. +func recordDenial(repo string, finding SafetyFinding) int { + path, ok := denialLedgerPath(repo) + if !ok { + return 1 + } + ledger := loadDenialLedger(path) + key := denialLedgerKey(finding) + entry := ledger.Counts[key] + entry.Count++ + entry.LastUnix = time.Now().Unix() + ledger.Counts[key] = entry + saveDenialLedger(path, ledger) + return entry.Count +} + +// resetDenialLedger clears the ledger. Called when the guard ALLOWS a +// mutation-capable call: forward progress means the agent is no longer stuck, +// so stale history must not escalate the next unrelated denial. +func resetDenialLedger(repo string) { + path, ok := denialLedgerPath(repo) + if !ok { + return + } + _ = os.Remove(path) +} diff --git a/boatstack/paths.go b/boatstack/paths.go index 584b6fe..250cb03 100644 --- a/boatstack/paths.go +++ b/boatstack/paths.go @@ -211,6 +211,18 @@ func (w WorkspaceContext) FlowDir() (string, error) { return filepath.Join(base, "flow"), nil } +// GuardDir holds the per-worktree guard bookkeeping (the denial ledger). It is +// worktree-partitioned like the delivery state: one worktree's denial history +// must never escalate a sibling's denials. +// control-law: repeated-denials-escalate-to-solutions +func (w WorkspaceContext) GuardDir() (string, error) { + base, err := w.worktreeControlDir() + if err != nil { + return "", err + } + return filepath.Join(base, "guard"), nil +} + // RuntimeDir holds the shared, version-namespaced runtime binary for the current // platform. version and sourceCommit are validated as single safe path segments. func (w WorkspaceContext) RuntimeDir(version, sourceCommit string) (string, error) { diff --git a/boatstack/references/artifacts.md b/boatstack/references/artifacts.md index d29c492..9e61e0a 100644 --- a/boatstack/references/artifacts.md +++ b/boatstack/references/artifacts.md @@ -133,6 +133,7 @@ clone, `external` outside the repository (Detached Supervision). | delivery-state | runtime-worktree | per-worktree | delivery transitions | | operation-ledger | runtime-worktree | per-worktree | run-preflight, publishers | | flow-logs | runtime-worktree | per-worktree | flow | +| guard-denial-ledger | runtime-worktree | per-worktree | safety-hook, ambient-safety-hook | | runtime-slots | runtime-shared | git-common | init, update, hydrate-runtime | | mutation-receipts | runtime-shared | git-common | activate-plan, undo | | update-previews | runtime-shared | git-common | prepare-update-pr, publish-update-pr | diff --git a/boatstack/safety.go b/boatstack/safety.go index d0f85b6..6d15550 100644 --- a/boatstack/safety.go +++ b/boatstack/safety.go @@ -28,6 +28,12 @@ type SafetyFinding struct { OperationState string `json:"operation_state,omitempty"` AttemptNumber int `json:"attempt_number,omitempty"` ReconciliationRequired bool `json:"reconciliation_required,omitempty"` + // RepeatCount is how many times this same denial (category at stage) has + // fired consecutively in this worktree, from the guard's denial ledger. + // At the escalation threshold the rendering lifts its solution-set cap and + // prescribes a fresh diagnostic — more information, never more severity. + // control-law: repeated-denials-escalate-to-solutions + RepeatCount int `json:"repeat_count,omitempty"` } type SafetyReport struct { @@ -1400,13 +1406,22 @@ func HookDecision(options SafetyHookOptions) ([]byte, bool) { findings := ClassifyTool(repo, name, input) if len(findings) == 0 { if finding := superviseToolAttempt(repo, host, name, input, options.Input); finding != nil { + finding.RepeatCount = recordDenial(repo, *finding) value, _ := contract.deny(repo, *finding) return value, true } + // An allowed mutation-capable call is forward progress: stale denial + // history must not escalate the next unrelated denial. + // control-law: repeated-denials-escalate-to-solutions + if mutationCapableTool(name, input) { + resetDenialLedger(repo) + } value, _ := contract.allow() return value, false } - value, _ := contract.deny(repo, findings[0]) + finding := findings[0] + finding.RepeatCount = recordDenial(repo, finding) + value, _ := contract.deny(repo, finding) return value, true } diff --git a/boatstack/statemap.go b/boatstack/statemap.go index 7f1d078..bf46d3d 100644 --- a/boatstack/statemap.go +++ b/boatstack/statemap.go @@ -186,6 +186,20 @@ func StateRegistry() []StateEntry { return filepath.Join(base, "trajectory.jsonl"), nil }, }, + { + // The denial ledger backing repeated-denials-escalate-to-solutions. + // Written only by the hook's own deny path; per-worktree so one + // worktree's denial history never escalates a sibling's denials. + Name: "guard-denial-ledger", Class: ClassRuntimeWorktree, Partition: "per-worktree", Gitignored: true, GuardProtected: true, + OwnerVerbs: []string{"safety-hook", "ambient-safety-hook"}, + Sample: func(w WorkspaceContext) (string, error) { + base, err := w.GuardDir() + if err != nil { + return "", err + } + return filepath.Join(base, "denials.json"), nil + }, + }, { Name: "runtime-slots", Class: ClassRuntimeShared, Partition: "git-common", Gitignored: true, GuardProtected: true, OwnerVerbs: []string{"init", "update", "hydrate-runtime"}, diff --git a/boatstack/statemap_conformance_test.go b/boatstack/statemap_conformance_test.go index dad8864..1755a1e 100644 --- a/boatstack/statemap_conformance_test.go +++ b/boatstack/statemap_conformance_test.go @@ -67,6 +67,7 @@ func TestEveryWorkspaceResolverIsDeclared(t *testing.T) { "DeliveryDir": {resolve(w.DeliveryDir), ClassRuntimeWorktree}, "OperationDir": {resolve(w.OperationDir), ClassRuntimeWorktree}, "FlowDir": {resolve(w.FlowDir), ClassRuntimeWorktree}, + "GuardDir": {resolve(w.GuardDir), ClassRuntimeWorktree}, "RuntimeDir": {resolve(func() (string, error) { return w.RuntimeDir("v0.0.0", "0000000") }), ClassRuntimeShared}, diff --git a/docs/evidence-engineered-coding.md b/docs/evidence-engineered-coding.md index a4cb151..8ff7398 100644 --- a/docs/evidence-engineered-coding.md +++ b/docs/evidence-engineered-coding.md @@ -96,7 +96,7 @@ subject to acceptance criteria pass approval is current ``` -That is why context trimming is not automatically an optimization. If removing state increases rework or false acceptance, total cost rises. The canonical runtime references are approximately **20915 estimated tokens**, while host adapters point to one operation at a time. +That is why context trimming is not automatically an optimization. If removing state increases rework or false acceptance, total cost rises. The canonical runtime references are approximately **20938 estimated tokens**, while host adapters point to one operation at a time. ## Control appears at transitions @@ -146,6 +146,6 @@ Delivery and system improvement also remain separate. A failed task may suggest ## What is evidence-backed -The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`5e9cd972c1a75c067b73625cf411ecd290086715`](https://github.com/operatorstack/intelligence-flow/tree/5e9cd972c1a75c067b73625cf411ecd290086715/labs/12-product-engineering-loop). +The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`6ec978238627ee8b8f072feeabbada0ccf73420e`](https://github.com/operatorstack/intelligence-flow/tree/6ec978238627ee8b8f072feeabbada0ccf73420e/labs/12-product-engineering-loop). The evidence supports specific failure mechanisms and guardrails. It does not establish that Boatstack is optimal, that control-theory notation proves software quality, or that one workflow dominates every team. Those are evaluation questions, so the distribution preserves measurements, provenance, gaps, and negative results. diff --git a/docs/public-claims.json b/docs/public-claims.json index 5139a4c..665f0a2 100644 --- a/docs/public-claims.json +++ b/docs/public-claims.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "source_commit": "5e9cd972c1a75c067b73625cf411ecd290086715", + "source_commit": "6ec978238627ee8b8f072feeabbada0ccf73420e", "statuses": ["verified", "observed", "still_being_evaluated"], "claims": [ { @@ -12,7 +12,7 @@ "readable_evidence": "why-these-steps.md#portable-workflow-and-state", "implementation": ["../boatstack/export.go", "../boatstack/references/artifacts.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "human-decisions", @@ -23,7 +23,7 @@ "readable_evidence": "why-these-steps.md#human-decisions", "implementation": ["../boatstack/references/workflow.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "validation-provenance", @@ -34,7 +34,7 @@ "readable_evidence": "why-these-steps.md#validation-provenance", "implementation": ["validation-and-evidence.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "irreversible-operations", @@ -46,7 +46,7 @@ "readable_evidence": "why-these-steps.md#irreversible-operations", "implementation": ["safety.md", "../boatstack/safety.go", "../boatstack/hooks.go"], "verification": ["../boatstack/safety_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "reviewer-ready-pr", @@ -57,7 +57,7 @@ "readable_evidence": "why-these-steps.md#reviewer-ready-pr", "implementation": ["../boatstack/pr.go", "getting-started.md"], "verification": ["../boatstack/pr_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "phase-scoped-delivery", @@ -68,7 +68,7 @@ "readable_evidence": "why-these-steps.md#phase-scoped-delivery", "implementation": ["../boatstack/delivery.go", "../boatstack/safety.go", "../boatstack/hooks.go", "../boatstack/references/workflow.md"], "verification": ["../boatstack/delivery_test.go", "../boatstack/pr_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "model-neutral-contract", @@ -79,7 +79,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "cross-model-failures", @@ -90,7 +90,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "lower-cost-outcomes", @@ -101,7 +101,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "git-worktree-activation", @@ -112,7 +112,7 @@ "readable_evidence": "why-these-steps.md#git-worktree-activation", "implementation": ["../boatstack/runtime_cache.go", "../boatstack/hooks.go"], "verification": ["../boatstack/runtime_cache_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" }, { "id": "visible-updates", @@ -123,7 +123,7 @@ "readable_evidence": "why-these-steps.md#visible-updates", "implementation": ["../boatstack/update.go", "../boatstack/init.go"], "verification": ["../boatstack/update_test.go", "../boatstack/init_test.go", "../boatstack/export_test.go"], - "last_verified_version": "source:5e9cd972c1a75c067b73625cf411ecd290086715" + "last_verified_version": "source:6ec978238627ee8b8f072feeabbada0ccf73420e" } ] } diff --git a/labs/diagram-json/plan.lock.json b/labs/diagram-json/plan.lock.json index 08c9853..48c4dd8 100644 --- a/labs/diagram-json/plan.lock.json +++ b/labs/diagram-json/plan.lock.json @@ -6,7 +6,7 @@ "plan_path": "labs/diagram-json/plan.md", "plan_sha256": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "schema_version": 1, - "source_commit": "5e9cd972c1a75c067b73625cf411ecd290086715", + "source_commit": "6ec978238627ee8b8f072feeabbada0ccf73420e", "source_plan_path": "labs/diagram-json/source-plan.md", "source_plan_sha256": "e10593ddaa7522ab80cc991d0a09399257139799e37f737794cd49d68a39985b", "spec_path": "labs/diagram-json/spec.md", diff --git a/release-notes/2026-07-27-repeated-denials-escalate.md b/release-notes/2026-07-27-repeated-denials-escalate.md new file mode 100644 index 0000000..f1b24b9 --- /dev/null +++ b/release-notes/2026-07-27-repeated-denials-escalate.md @@ -0,0 +1,5 @@ +### A denial that repeats escalates its help, not its severity + +An agent that hits the same guardrail again and again is telling you the stated rule is not reaching it — and under that pressure agents drift toward worse moves, not better ones. The guard now keeps a small per-worktree count of identical denials. From the third identical denial, the message escalates its corrective information: the pick list expands from three commands to the full legal set, and the denial prescribes a fresh diagnostic (`boatstack-helper doctor`) instead of leaving the loop to continue. + +Nothing about admissibility changes — what was denied stays denied, what was allowed stays allowed, and the constitutional destruction floor is untouched. Any allowed mutating call counts as forward progress and clears the history, so an old denial streak never escalates an unrelated message. The ledger lives with the rest of the worktree's control state, is bounded, and degrades silently: if its file is corrupt or unwritable, the denial simply renders unescalated.