diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f194d83..f389712 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,7 +2,7 @@ # Contributing -Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/3e44eb72add2a33fb84bbc5a0142af9c1828feb1/labs/12-product-engineering-loop). +Boatstack is a generated content distribution. Propose changes to workflow semantics, templates, evidence rules, or generated presentation in [Intelligence Flow](https://github.com/operatorstack/intelligence-flow/tree/5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d/labs/12-product-engineering-loop). The Boatstack repository receives product/runtime changes through a generated pull request. Review the PR's `UPSTREAM.json`, tests, adapter diff, and context-size change; do not hand-edit generated output on `main`. `.github/workflows` is the exception: it is Boatstack's executable control plane, excluded from scheduled projection and changed only through a separate manually reviewed Boatstack PR. diff --git a/UPSTREAM.json b/UPSTREAM.json index aab8d5e..d8f44d8 100644 --- a/UPSTREAM.json +++ b/UPSTREAM.json @@ -12,7 +12,7 @@ }, "files": { ".gitignore": "a7079e923a776f14f1bb3a6aa0a11a133a8e1dfb35af020f327623357b7e3957", - "CONTRIBUTING.md": "cb76f4adb6c3ea2b4b1287ca6099bf0b38f56552649389fc8b42fd95320cb792", + "CONTRIBUTING.md": "ae6641051824b5bbc7588d7d70a2657f091e0579d5db4616a8aed5b4f18e1d69", "README.md": "125b47671a68556df382f19756fb61fa18925606cbbaf54d6bc9df8872b36870", "assets/boatstack-journey.svg": "e465befc50c8ce30f3e07e8fd97012931beeb053392c8fbf38ad645023b3cc63", "assets/boatstack-mark.svg": "be1f984da1bfa69fa5d1f986d8343d21f7e20921b71db888c928b4d2e54b09b5", @@ -34,9 +34,13 @@ "boatstack/assets/templates/test-plan.md": "6db8a9f27dd171fb80222a501cae50eb051e7278c04703fa43b5ff86dd4d2df4", "boatstack/atomic_unix.go": "89f2723361591de2bb8bd22ce7e34ec529d3278509f0df78fd5c4a7d4140fbe9", "boatstack/atomic_windows.go": "cefd775cbe7e7c3bd8a3f5673b11cdd784c6d3ebd6de7dcb8f39406b0bee511f", + "boatstack/capability.go": "9d9a75086e88bb1d9d4170821436c686002fe3a125e6ee7d23eea5779aac5cfe", + "boatstack/capability_test.go": "e8322903a843970d7f0317d2629466532cb05c3ffcb44b9423adf4219faa8021", + "boatstack/capture.go": "e32dd7096ccc4844d37c9ec0aea8284adee58bc84cd5b6264516b6661b348bfc", + "boatstack/capture_test.go": "63fa1177738081f1e862364d7a4257f5e259f8e9c36276ba1775b8085b277105", "boatstack/changelog.go": "c5e1f31440b44d61e6037ad27af0333540af3545d655e35819a0241cbbebd8ec", "boatstack/changelog_test.go": "ce792f23a7fe1e09fb3096cd1314130a6ab69321d4877b12a8e994027541baf7", - "boatstack/cmd/boatstack-helper/main.go": "43f070dc71f924ea6466b4190c43811d7884cece7d9d44df433c9beeb6982053", + "boatstack/cmd/boatstack-helper/main.go": "5aff755d60b2f5c06dab3ed5d30df026d28c6fb90d891b3be38e955aa200621d", "boatstack/cmd/boatstack-helper/main_test.go": "ff73003b6a5157202fa09ddf1129fb13c3d79702b2e05a8721ce5a11bf5ab779", "boatstack/command.go": "94d2117c6e390d5a644afc5cd90f7e712e3f8b1032f8c9e3b253cc134524c28a", "boatstack/command_test.go": "9f707abba3640add81c3e97ba7e72fedbf98f3394b1c060a9ca4b4a28e919968", @@ -46,7 +50,7 @@ "boatstack/delivery.go": "ea53af0e702ec3668a563a5f786dcac2e095362285ca6093b7ed71ec495a0a48", "boatstack/delivery_test.go": "564ad2029a8412de7953967b1acdf377e6c87b6f6e3465d1f8741a4813f2949f", "boatstack/evidence.go": "497a31e6ff632cb1d7c3adfc9f269af3f6aa84e948dd5d417c162767542a27df", - "boatstack/export.go": "abccdfc04b7b49a17e8531c0332f80262e693273f7974b646528fe00af1d6ae6", + "boatstack/export.go": "388c676e4fb25442a15e51e31bfc5931b6264d7a95a3422ecec0c1f78ab613d9", "boatstack/export_test.go": "67eb890728d20630925d6e4e90d2a97ec025195ba5c54994c1b098ab72721dca", "boatstack/go.mod": "6086ef1b2a83f5696190dca692c653925f27b61f652f659fd3fca43ed54a1641", "boatstack/go.sum": "26c315c867b11b886f3c9402fce7f341f6a9115a5d61f54afbb5e1b1fb5f6017", @@ -72,6 +76,8 @@ "boatstack/planning_test.go": "c105a9c78c342be06614bf54d0bc1b661b0f7af64d63b79e43bd1fcc2769edd5", "boatstack/pr.go": "bdb066acb329b6b772cb880db2e590b381cae909f270085cfacb8ef2fbfda651", "boatstack/pr_test.go": "7f82954d94c1ceae848a581dda25e58af92251d78a5a94ed2d672bedf5a0349e", + "boatstack/provision.go": "4882d49681f99b11ba9d182ca13772131b7f9a11a6c2b560800654ca14f5111e", + "boatstack/provision_test.go": "214e9edb991a66d5bbb696a7c1b63876d2f799f2cab4e3f40785f4e8f1eac57b", "boatstack/recovery.go": "dd816b18b54a0085b8d8276a93ee98d2b1e90099059a0d85cf6e24edf6f37d5b", "boatstack/recovery_test.go": "29490e7477ba602491330036a491289dd9117b99ff862f66dae421ba17e04c9f", "boatstack/references/artifacts.md": "5fa888ac519085d65cee1d04df5902761651bcf2d7af81711fa0f8ecd1fc0f59", @@ -99,7 +105,7 @@ "boatstack/update_publication.go": "c8b7bd38019b1cbf8c523652e9e20e8c971c7e48b2f3972a0ac638648326fe63", "boatstack/update_publication_test.go": "c5f32578db53be65e35452d5e8b4520884354e4e80a370dcec19860ea644d091", "boatstack/update_test.go": "bf5f19f8499db6dd7356917d867b113b790548ab89bfdea59e2adf4999a82a6a", - "boatstack/visual_evidence.go": "90a68d554e10ff4fb7afa45000912cdedd4cdf93b3d279055b50844401924f01", + "boatstack/visual_evidence.go": "4d69f98adb6087d4830e6703273ae6e03b28ec8b3c496a6e432fd5e0e54d8a2f", "boatstack/visual_evidence_test.go": "0fe8f5154ef4398dfeba5e7f7b387b2d279ed75635ea35d93ff392269f2cc6d0", "boatstack/workspace.go": "91b343400b3506a6f516c28fabc3f1575f22024a5b19f934a020be660a20482e", "boatstack/workspace_sync.go": "0cc2f03fd1aa57c66b3d603d5ef30aa820f14ab60771a795cf10c46f3c9a71b8", @@ -109,10 +115,10 @@ "docs/benchmark-corpus-audit.md": "f2d206fe8579a514f9da82b2c96c19b343ac004be67617e1bd34f0f8e0e5e6c6", "docs/benchmark-submission-audit.md": "9518abdd17690729c6423f87cab20418ed47b0915b5faa44b9ef975e9e9c3b79", "docs/configuration.md": "4d8f207b415a5a1e3b9b1698ee7bb1221aa0e5496a061bb8054294df2f347ad1", - "docs/evidence-engineered-coding.md": "5d3732ff40bec63ff8ed89f60621347520e3607c00438115b9051d85368a2c82", + "docs/evidence-engineered-coding.md": "58205ac01839ab8c29d76090e2744afa827a623d7be6de6b6e468f7b3730c047", "docs/generated-files.md": "437791765b0a4015032ae21d1a6618563cad92b7402819e4f963bf5ae16284a3", "docs/getting-started.md": "f314270c5ed1a55bbef5f3ddbcb5596693dbee9374e5f0d3df8838cefbd68052", - "docs/public-claims.json": "c4ed79593ff0acd86e4c1602bc269e8ccc9fe6400f8c798d714c78d599aa47a3", + "docs/public-claims.json": "abbf9221262970d74e60e5cd54f69b1dad93ce453fa75e9516f86ee4ef12b2f8", "docs/public-surface.md": "713f7a050b5f339cf948299103ef3800417dccfecf2cc1a4166397ea6f978907", "docs/research-and-design.md": "8d78678108f0a6c924e1ff9b32c0f81aae9d1f779e0082843b6f99ad993ae2b6", "docs/safety.md": "7b9b5c515d36e683767ec8d3d9d6d119ac93650b2f629d351deadd4c600ed6a6", @@ -126,7 +132,7 @@ "labs/diagram-json/compiled/evidence.md": "1ba1c989ade070a8ef9a508fbd788d100d7292f2dbacbb2bce895468019f619d", "labs/diagram-json/compiled/tasks.json": "88f60851abf79d851e9fccc754ff3040034ae595306bc87d64784c19eb403e71", "labs/diagram-json/compiled/test-matrix.json": "424657ff505768e50fa113801fd8363364a18269d5297480907a993d44063a39", - "labs/diagram-json/plan.lock.json": "24f89e5c01a81b707c8de9b815b891210841a444d868056306009e43803ca37f", + "labs/diagram-json/plan.lock.json": "40ec1d86a0c3e7186546f6f26bc4023cbea0020b04553537515b3a85d5ae37fe", "labs/diagram-json/plan.md": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "labs/diagram-json/questions.md": "74733b015002c8a6777c558e7e997fa48c94850b9bd39054fe9366c97ecf728d", "labs/diagram-json/request.md": "0808fc41c36779c404f4a3a121167da6e76cac56df526e70f9ed6d3e0d4c02ed", @@ -186,9 +192,13 @@ "release-notes/2026-07-22-value-translation-boundary.md": "9cf168ff7caaf3906b78533935bdfb2c86e753984ed1e5ec8204cb373d083390", "release-notes/2026-07-23-bootstrap-safe-update-repair.md": "d8e66c46ae05e3d228dfea45879f4d1166d5e3a253ac24bababd4c3e396c214b", "release-notes/2026-07-23-canonical-update-ownership.md": "7f34f890b252493797519389b23f1b56ec7ec16db7547aac8ea196f75f3b8c2c", + "release-notes/2026-07-23-capability-provisioning.md": "86fefcc4212212b0522b2a9b732663ee98cc41cec96b9842f19bb4433277a6a4", + "release-notes/2026-07-23-capture-evidence-orchestration.md": "13e0dfa65898a4ba63a7866fc1db4d32df2f5610db0e8301734d47404f0e3da9", + "release-notes/2026-07-23-evidence-capability-substrate.md": "a3e6e364536f826f299243932494096c27458e2621a5ce55465c5db280b53758", "release-notes/2026-07-23-explicit-source-plan.md": "ec1f97434f9263f6db4bc3b83ae83213053cd2bc6b7465e4fb2d67cbea5cf731", "release-notes/2026-07-23-ignore-ambiguous-deliveries.md": "9b1b9fd48db340b91fcced1c282297de0ecad8744723f2b1fdd94033927a88c4", "release-notes/2026-07-23-labkit-publishing-pipeline.md": "b3bfc5f28baf3961fb187380ff66e5383186c05a933bc6ca162416ce61149218", + "release-notes/2026-07-23-labkit-standalone-package.md": "50ae9ba89fdada6e04a9200ea26d53a2a04484a1759cdea663687cc9b42fb93c", "release-notes/2026-07-23-managed-pr-task-graph-layout.md": "e2d67f15cc6a1eb200d6f13f81d891b30eae1bd51182d768dda561fcfdc69435", "release-notes/2026-07-23-recoverable-repository-sync.md": "3afc4f6220ae76df3bd6dcd15fc180135274c808729e9c512a2c53462ec690c2", "release-notes/2026-07-23-shipped-feature-candidate-resolution.md": "bd8ee8e7f3f216b356b121a83ef10cb0c8131a90b9ab803edebf23b882d9cf89" @@ -196,7 +206,7 @@ "generator": "operatorstack/intelligence-flow:boatstack-distribution", "schema_version": 1, "source": { - "commit": "3e44eb72add2a33fb84bbc5a0142af9c1828feb1", + "commit": "5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d", "path": "labs/12-product-engineering-loop", "repository": "operatorstack/intelligence-flow" } diff --git a/boatstack/capability.go b/boatstack/capability.go new file mode 100644 index 0000000..b192c5e --- /dev/null +++ b/boatstack/capability.go @@ -0,0 +1,70 @@ +package boatstack + +import ( + "fmt" + "strings" +) + +// Capability describes a named producer of PR evidence. It is the generic spine +// that concrete evidence types register on: detection ("can this repository +// produce the evidence?"), capture orchestration ("what command runs it, and +// when?"), and provisioning ("if it is unavailable, how do we help?") all read +// this registry instead of hard-coding a single evidence type. +// +// Boatstack ships the contract in this registry, not the capture harness. The +// harness is authored in the user's repository and invoked through the resolved +// repository command; Boatstack only records what the command must satisfy. +type Capability struct { + // Name is the canonical capability identifier, e.g. "visual". + Name string + // CommandAliases are the project.commands keys that satisfy the capability, + // in priority order. The first non-empty command wins. + CommandAliases []string + // AdmittedStages are the delivery stages in which a capture attempt may run. + AdmittedStages []string + // RetryClass is the operation retry class recorded for a capture attempt. + RetryClass string +} + +// CapabilityResolution is the portable capability cut for one capability: it +// reports whether the repository owns a command that produces this evidence. +type CapabilityResolution struct { + Name string `json:"name"` + Kind string `json:"kind"` // repository-command | unavailable + Command string `json:"command,omitempty"` +} + +// capabilityRegistry holds every evidence capability Boatstack knows about. +// Adding a provider is a single entry here plus its tenant-specific manifest +// contract (see visual_evidence.go for the first tenant). +var capabilityRegistry = map[string]Capability{ + "visual": { + Name: "visual", + CommandAliases: []string{"visual", "screenshot", "e2e"}, + AdmittedStages: []string{"BUILD", "TEST_PASSED"}, + RetryClass: "IDEMPOTENT_EXTERNAL", + }, +} + +// LookupCapability returns the registered capability metadata for a name. +func LookupCapability(name string) (Capability, bool) { + capability, ok := capabilityRegistry[strings.TrimSpace(name)] + return capability, ok +} + +// ResolveCapability performs the repository-owned capability cut: it selects the +// first project command alias that is set. It is the generic detection primitive +// shared by capture orchestration and provisioning. Kind is "repository-command" +// when the repository owns a command, otherwise "unavailable". +func ResolveCapability(name string, config ProjectConfig) (CapabilityResolution, error) { + capability, ok := LookupCapability(name) + if !ok { + return CapabilityResolution{}, fmt.Errorf("unknown evidence capability %q", name) + } + for _, alias := range capability.CommandAliases { + if command := strings.TrimSpace(config.Project.Commands[alias]); command != "" { + return CapabilityResolution{Name: capability.Name, Kind: "repository-command", Command: command}, nil + } + } + return CapabilityResolution{Name: capability.Name, Kind: "unavailable"}, nil +} diff --git a/boatstack/capability_test.go b/boatstack/capability_test.go new file mode 100644 index 0000000..8e30e72 --- /dev/null +++ b/boatstack/capability_test.go @@ -0,0 +1,61 @@ +package boatstack + +import "testing" + +func TestResolveCapabilitySelectsRepositoryCommandByAliasPriority(t *testing.T) { + config := testConfig() + delete(config.Project.Commands, "visual") + delete(config.Project.Commands, "screenshot") + delete(config.Project.Commands, "e2e") + + // Lowest-priority alias still resolves the repository-owned cut. + config.Project.Commands["e2e"] = "npm run e2e" + resolution, err := ResolveCapability("visual", config) + if err != nil { + t.Fatalf("resolve visual: %v", err) + } + if resolution.Kind != "repository-command" || resolution.Command != "npm run e2e" || resolution.Name != "visual" { + t.Fatalf("alias fallback did not resolve: %#v", resolution) + } + + // A higher-priority alias wins over a lower one. + config.Project.Commands["visual"] = "npm run capture:visual" + resolution, err = ResolveCapability("visual", config) + if err != nil { + t.Fatalf("resolve visual: %v", err) + } + if resolution.Command != "npm run capture:visual" { + t.Fatalf("higher-priority alias did not win: %#v", resolution) + } +} + +func TestResolveCapabilityReportsUnavailableWithoutCommand(t *testing.T) { + config := testConfig() + delete(config.Project.Commands, "visual") + delete(config.Project.Commands, "screenshot") + delete(config.Project.Commands, "e2e") + + resolution, err := ResolveCapability("visual", config) + if err != nil { + t.Fatalf("resolve visual: %v", err) + } + if resolution.Kind != "unavailable" || resolution.Command != "" { + t.Fatalf("expected unavailable resolution: %#v", resolution) + } +} + +func TestResolveCapabilityRejectsUnknownCapability(t *testing.T) { + if _, err := ResolveCapability("does-not-exist", testConfig()); err == nil { + t.Fatal("expected error for unknown capability") + } +} + +func TestLookupCapabilityExposesRegisteredMetadata(t *testing.T) { + capability, ok := LookupCapability("visual") + if !ok { + t.Fatal("visual capability is not registered") + } + if len(capability.CommandAliases) == 0 || capability.RetryClass == "" || len(capability.AdmittedStages) == 0 { + t.Fatalf("visual capability metadata is incomplete: %#v", capability) + } +} diff --git a/boatstack/capture.go b/boatstack/capture.go new file mode 100644 index 0000000..4fb5148 --- /dev/null +++ b/boatstack/capture.go @@ -0,0 +1,306 @@ +package boatstack + +import ( + "bytes" + "fmt" + "image/png" + "os" + "os/exec" + "path/filepath" + "strings" + "time" +) + +const captureMaxAttempts = 3 + +// CaptureRequest is the framework-agnostic contract Boatstack passes to the +// repository-owned capture harness for one scenario. Boatstack ships this +// contract; the harness (authored in the user's repository) satisfies it by +// writing exactly one PNG to OutputPath. The contract is surfaced to the harness +// as environment variables (see execCaptureRunner). +type CaptureRequest struct { + Repo string + Capability string + Command string + Scenario PRVisualScenario + OutputPath string +} + +// CaptureRunner runs one scenario's repository capture command. It must produce +// a valid PNG at request.OutputPath. It is an interface so tests can drive +// capture deterministically without a real browser or dev server. +type CaptureRunner interface { + Run(request CaptureRequest) error +} + +// execCaptureRunner invokes the repository-owned command through the shell, +// exposing the capture contract as environment variables. The command is taken +// from trusted project configuration. +type execCaptureRunner struct{} + +func (execCaptureRunner) Run(request CaptureRequest) error { + command := exec.Command("sh", "-c", request.Command) + command.Dir = request.Repo + command.Env = append(os.Environ(), + "BOATSTACK_CAPTURE_CAPABILITY="+request.Capability, + "BOATSTACK_CAPTURE_SCENARIO_ID="+request.Scenario.ID, + "BOATSTACK_CAPTURE_ENTRY="+request.Scenario.Entry, + "BOATSTACK_CAPTURE_STATE="+request.Scenario.State, + "BOATSTACK_CAPTURE_VIEWPORT="+request.Scenario.Viewport, + "BOATSTACK_CAPTURE_OUTPUT="+request.OutputPath, + ) + // The harness's authoritative output is the PNG on disk, not stdout; only + // stderr is retained, as bounded diagnostics for a failed capture. + var diagnostics bytes.Buffer + command.Stdout = nil + command.Stderr = &diagnostics + if err := command.Run(); err != nil { + return fmt.Errorf("capture command failed: %w: %s", err, boundedObservation(diagnostics.String())) + } + return nil +} + +// CaptureEvidenceOptions configures a managed capture run. +type CaptureEvidenceOptions struct { + Repo string + Capability string + Feature string + Base string + Runner CaptureRunner +} + +// CaptureEvidence orchestrates capture for a managed feature. It resolves the +// repository-owned capability command, reads the plan-declared scenarios, runs +// each scenario as a supervised operation, stamps the manifest to the current +// head commit and product diff, and ingests it through SavePRVisualEvidence. +// The manifest is trusted only if it conforms; a non-conformant manifest is a +// blocking error, never a silent PASS. +func CaptureEvidence(options CaptureEvidenceOptions) (PRVisualEvidenceManifest, error) { + repo, err := ResolveRepository(options.Repo) + if err != nil { + return PRVisualEvidenceManifest{}, err + } + name := strings.TrimSpace(options.Capability) + if name == "" { + name = "visual" + } + capability, ok := LookupCapability(name) + if !ok { + return PRVisualEvidenceManifest{}, fmt.Errorf("unknown evidence capability %q", name) + } + feature := strings.TrimSpace(options.Feature) + if feature == "" { + return PRVisualEvidenceManifest{}, fmt.Errorf("capture requires a managed --feature") + } + config, _, err := LoadConfig(filepath.Join(repo, ".product-loop", "project.json")) + if err != nil { + return PRVisualEvidenceManifest{}, fmt.Errorf("capture requires a valid Boatstack project configuration: %w", err) + } + resolution, err := ResolveCapability(name, config) + if err != nil { + return PRVisualEvidenceManifest{}, err + } + if resolution.Kind != "repository-command" { + return PRVisualEvidenceManifest{}, fmt.Errorf("evidence capability %q is unavailable: register a repository command (project.commands) or provision it first", name) + } + + relevance, source, scenarios, err := planVisualDecision(repo, feature) + if err != nil { + return PRVisualEvidenceManifest{}, err + } + if relevance == "not_relevant" { + return PRVisualEvidenceManifest{}, fmt.Errorf("%s evidence is marked not_relevant for %q; nothing to capture", name, feature) + } + if len(scenarios) == 0 { + return PRVisualEvidenceManifest{}, fmt.Errorf("no %s scenarios declared in the plan (pr_visual_evidence.scenarios)", name) + } + + head, err := gitCommand(repo, "rev-parse", "--abbrev-ref", "HEAD") + if err != nil { + return PRVisualEvidenceManifest{}, err + } + base := strings.TrimSpace(options.Base) + if base == "" { + base = strings.TrimSpace(config.Project.DefaultBranch) + } + if base == "" { + base = defaultPRBase(repo) + } + headCommit, diffHash, err := captureProductDiff(repo, base, feature, head) + if err != nil { + return PRVisualEvidenceManifest{}, err + } + + key, err := visualEvidenceKey("managed", feature, head) + if err != nil { + return PRVisualEvidenceManifest{}, err + } + stagingDir, err := captureStagingDirectory(repo, key) + if err != nil { + return PRVisualEvidenceManifest{}, err + } + runner := options.Runner + if runner == nil { + runner = execCaptureRunner{} + } + + items := make([]PRVisualEvidenceItem, 0, len(scenarios)) + for _, scenario := range scenarios { + outputPath := filepath.Join(stagingDir, scenario.ID+".png") + if err := captureScenario(repo, capability, resolution.Command, scenario, outputPath, feature, head, headCommit, diffHash, runner); err != nil { + return PRVisualEvidenceManifest{}, err + } + items = append(items, PRVisualEvidenceItem{ + ScenarioID: scenario.ID, + Path: outputPath, + MIMEType: "image/png", + Viewport: scenario.Viewport, + CapturedAt: time.Now().UTC().Truncate(time.Second).Format(time.RFC3339), + Status: "captured", + PrivacyStatus: "clean", + }) + } + + manifest := PRVisualEvidenceManifest{ + Key: key, + Policy: config.Workflow.PRVisualEvidence, + Relevance: relevance, + RelevanceSource: source, + Status: "PASS", + SourceCommit: headCommit, + ProductDiffSHA256: diffHash, + Scenarios: scenarios, + Items: items, + } + saved, err := SavePRVisualEvidence(repo, manifest) + if err != nil { + return PRVisualEvidenceManifest{}, fmt.Errorf("captured %s evidence is non-conformant (BLOCKED): %w", name, err) + } + return saved, nil +} + +// captureProductDiff reproduces the pr-context product-diff fingerprint so a +// captured manifest is trusted (PASS) by resolvePRVisualEvidence: same head +// commit and same product diff. +func captureProductDiff(repo, base, feature, head string) (headCommit, diffHash string, err error) { + baseCommit, err := resolveBaseCommit(repo, base) + if err != nil { + return "", "", err + } + mergeBaseCommit, err := gitCommand(repo, "merge-base", baseCommit, "HEAD") + if err != nil || mergeBaseCommit == "" { + return "", "", fmt.Errorf("cannot determine the merge base between %s and %s", base, head) + } + headCommit, err = gitCommand(repo, "rev-parse", "HEAD") + if err != nil { + return "", "", err + } + previewPath, err := expectedPRPreviewPath("managed", feature, head) + if err != nil { + return "", "", err + } + diff, _, err := productDiff(repo, mergeBaseCommit, previewPath) + if err != nil { + return "", "", err + } + return headCommit, SHA256Bytes(diff), nil +} + +func captureStagingDirectory(repo, key string) (string, error) { + directory, err := visualEvidenceDirectory(repo, key) + if err != nil { + return "", err + } + staging := filepath.Join(directory, "capture-staging") + if err := os.MkdirAll(staging, 0o700); err != nil { + return "", err + } + return staging, nil +} + +// captureScenario runs one scenario as a supervised operation with a bounded +// retry budget. The fingerprint is stable for a given command, scenario, and +// product diff, so a successful capture on the same commit is reused rather than +// re-run. +func captureScenario(repo string, capability Capability, command string, scenario PRVisualScenario, outputPath, feature, head, headCommit, diffHash string, runner CaptureRunner) error { + fingerprint := SHA256Bytes([]byte(strings.Join([]string{ + command, scenario.ID, scenario.Viewport, scenario.Entry, scenario.State, headCommit, diffHash, + }, "\x00"))) + kind := "capture:" + capability.Name + postcondition := fmt.Sprintf("valid PNG captured for scenario %s (%s)", scenario.ID, scenario.Viewport) + + prepared, err := PrepareOperation(OperationPrepareOptions{ + Repo: repo, + Kind: kind, + Scope: OperationScope{Feature: feature, HeadBranch: head}, + Target: outputPath, + PackageFingerprint: fingerprint, + AuthorizationFingerprint: fingerprint, + RetryClass: capability.RetryClass, + MaxAttempts: captureMaxAttempts, + ExpectedPostcondition: postcondition, + }) + if err != nil { + return fmt.Errorf("prepare capture of %s: %w", scenario.ID, err) + } + if prepared.State == OperationSucceeded { + if verifyCapturedPNG(outputPath) == nil { + return nil + } + return fmt.Errorf("capture of %s already succeeded but its artifact is missing; run operation-status and reconcile", scenario.ID) + } + + var lastErr error + for attempt := 0; attempt < captureMaxAttempts; attempt++ { + begun, beginErr := BeginOperation(repo, prepared.OperationID, fmt.Sprintf("%s@%d", scenario.ID, attempt), kind) + if beginErr != nil { + return fmt.Errorf("begin capture of %s: %w", scenario.ID, beginErr) + } + if begun.Receipt.State == OperationSucceeded { + if verifyCapturedPNG(outputPath) == nil { + return nil + } + return fmt.Errorf("capture of %s reports success but its artifact is missing", scenario.ID) + } + runErr := runner.Run(CaptureRequest{ + Repo: repo, Capability: capability.Name, Command: command, Scenario: scenario, OutputPath: outputPath, + }) + if runErr == nil { + runErr = verifyCapturedPNG(outputPath) + } + if runErr == nil { + if _, err := CompleteOperation(repo, prepared.OperationID, begun.LeaseToken, "SUCCEEDED", "captured "+scenario.ID, outputPath); err != nil { + return fmt.Errorf("record capture success for %s: %w", scenario.ID, err) + } + return nil + } + lastErr = runErr + receipt, completeErr := CompleteOperation(repo, prepared.OperationID, begun.LeaseToken, "RETRYABLE", runErr.Error(), "") + if completeErr != nil { + return fmt.Errorf("record capture retry for %s: %w", scenario.ID, completeErr) + } + if receipt.State == OperationFailedFinal { + break + } + } + return fmt.Errorf("capture of scenario %s failed after %d attempts: %w", scenario.ID, captureMaxAttempts, lastErr) +} + +func verifyCapturedPNG(path string) error { + info, err := os.Lstat(path) + if err != nil || !info.Mode().IsRegular() || info.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("capture output is missing or unsafe: %s", path) + } + value, err := os.ReadFile(path) + if err != nil { + return err + } + configuration, err := png.DecodeConfig(bytes.NewReader(value)) + if err != nil { + return fmt.Errorf("capture output is not a valid PNG: %w", err) + } + if configuration.Width < 1 || configuration.Height < 1 { + return fmt.Errorf("capture output has no pixels: %s", path) + } + return nil +} diff --git a/boatstack/capture_test.go b/boatstack/capture_test.go new file mode 100644 index 0000000..17ce972 --- /dev/null +++ b/boatstack/capture_test.go @@ -0,0 +1,181 @@ +package boatstack + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +// stubCaptureRunner drives capture deterministically without a browser or dev +// server. write decides what (if anything) lands at the requested output path. +type stubCaptureRunner struct { + calls int + write func(request CaptureRequest) error +} + +func (runner *stubCaptureRunner) Run(request CaptureRequest) error { + runner.calls++ + return runner.write(request) +} + +// captureTestRepo builds a managed feature whose plan declares one relevant +// visual scenario and registers a visual capability command. The command itself +// is never executed in tests — the injected runner stands in for the repo-owned +// harness. +func captureTestRepo(t *testing.T, feature string) string { + t.Helper() + repo := t.TempDir() + runGit(t, repo, "init", "-b", "main") + runGit(t, repo, "config", "user.name", "Boatstack Test") + runGit(t, repo, "config", "user.email", "boatstack@example.invalid") + + config := testConfig() + config.Project.DefaultBranch = "main" + config.Workflow.PRVisualEvidence = "require" + config.Project.Commands["visual"] = "exit 1" // proves the runner, not the shell command, drives capture + value, err := MarshalJSON(config) + if err != nil { + t.Fatal(err) + } + if err := os.MkdirAll(filepath.Join(repo, ".product-loop"), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(repo, ".product-loop", "project.json"), value, 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(repo, "README.md"), []byte("# Fixture\n"), 0o644); err != nil { + t.Fatal(err) + } + runGit(t, repo, "add", ".") + runGit(t, repo, "commit", "-m", "base") + + runGit(t, repo, "switch", "-c", "feat/"+feature) + directory := filepath.Join(repo, ".product-loop", "features", feature) + if err := os.MkdirAll(directory, 0o755); err != nil { + t.Fatal(err) + } + plan := validPlan() + plan["feature_id"] = feature + plan["pr_visual_evidence"] = map[string]any{ + "relevance": "relevant", + "scenarios": []any{map[string]any{ + "id": "warning", "entry": "/onboarding", "state": "picker open", "viewport": "1440x900", + "expected": []any{"warning visible"}, + }}, + } + writeMarkdownPlan(t, filepath.Join(directory, "plan.md"), plan, true) + if err := os.WriteFile(filepath.Join(repo, "feature.go"), []byte("package fixture\n"), 0o644); err != nil { + t.Fatal(err) + } + runGit(t, repo, "add", ".") + runGit(t, repo, "commit", "-m", "feature work") + return repo +} + +func TestCaptureEvidenceProducesManifestTrustedByPRContext(t *testing.T) { + repo := captureTestRepo(t, "reviewer-ready") + runner := &stubCaptureRunner{write: func(request CaptureRequest) error { + if request.Capability != "visual" || request.Scenario.ID != "warning" || request.OutputPath == "" { + t.Fatalf("runner received an ill-formed capture request: %#v", request) + } + writeTestPNG(t, request.OutputPath) + return nil + }} + + manifest, err := CaptureEvidence(CaptureEvidenceOptions{ + Repo: repo, Capability: "visual", Feature: "reviewer-ready", Runner: runner, + }) + if err != nil { + t.Fatalf("capture failed: %v", err) + } + if manifest.Status != "PASS" || len(manifest.Items) != 1 || manifest.Items[0].PrivacyStatus != "clean" { + t.Fatalf("capture did not produce a conformant PASS manifest: %#v", manifest) + } + if !strings.Contains(manifest.Items[0].Path, filepath.Join("boatstack", "visual-evidence")) { + t.Fatalf("captured PNG was not ingested into Boatstack state: %s", manifest.Items[0].Path) + } + + // Capture must leave the product tree untouched (evidence lives in Git-common state). + if status := runGit(t, repo, "status", "--short"); status != "" { + t.Fatalf("capture mutated the product tree: %s", status) + } + + // The manifest must be trusted by the same resolver pr-context uses: identical + // head commit and product diff → status is the manifest's PASS, not NOT_VERIFIED. + head := runGit(t, repo, "rev-parse", "--abbrev-ref", "HEAD") + headCommit, diffHash, err := captureProductDiff(repo, "main", "reviewer-ready", head) + if err != nil { + t.Fatal(err) + } + config, _, err := LoadConfig(filepath.Join(repo, ".product-loop", "project.json")) + if err != nil { + t.Fatal(err) + } + _, status, count, _, _, _, resolved, err := resolvePRVisualEvidence(repo, config, "managed", "reviewer-ready", head, headCommit, diffHash) + if err != nil { + t.Fatal(err) + } + if status != "PASS" || count != 1 || resolved == nil { + t.Fatalf("captured evidence was not trusted by pr-context: status=%s count=%d", status, count) + } + + // A second capture on the same commit is idempotent: it reuses the supervised + // operation's successful artifact instead of re-running the harness. + priorCalls := runner.calls + if _, err := CaptureEvidence(CaptureEvidenceOptions{ + Repo: repo, Capability: "visual", Feature: "reviewer-ready", Runner: runner, + }); err != nil { + t.Fatalf("idempotent re-capture failed: %v", err) + } + if runner.calls != priorCalls { + t.Fatalf("re-capture re-ran the harness (%d extra calls) instead of reusing the receipt", runner.calls-priorCalls) + } +} + +func TestCaptureEvidenceFailsClosedOnNonConformantOutput(t *testing.T) { + repo := captureTestRepo(t, "reviewer-ready") + runner := &stubCaptureRunner{write: func(request CaptureRequest) error { + // The harness reports success but writes bytes that are not a valid PNG. + return os.WriteFile(request.OutputPath, []byte("not a png"), 0o600) + }} + + _, err := CaptureEvidence(CaptureEvidenceOptions{ + Repo: repo, Capability: "visual", Feature: "reviewer-ready", Runner: runner, + }) + if err == nil { + t.Fatal("capture accepted a non-conformant artifact instead of failing closed") + } + if runner.calls != captureMaxAttempts { + t.Fatalf("capture did not exhaust the supervised retry budget: %d attempts", runner.calls) + } + // Nothing may be persisted: a failed capture must not leave trusted evidence. + if _, err := LoadPRVisualEvidence(repo, "reviewer-ready"); err == nil { + t.Fatal("a failed capture persisted a manifest") + } +} + +func TestCaptureEvidenceRequiresAResolvedCapabilityCommand(t *testing.T) { + repo := captureTestRepo(t, "reviewer-ready") + // Remove every command alias so the capability resolves to unavailable. + config, _, err := LoadConfig(filepath.Join(repo, ".product-loop", "project.json")) + if err != nil { + t.Fatal(err) + } + delete(config.Project.Commands, "visual") + value, err := MarshalJSON(config) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(repo, ".product-loop", "project.json"), value, 0o644); err != nil { + t.Fatal(err) + } + + _, err = CaptureEvidence(CaptureEvidenceOptions{ + Repo: repo, Capability: "visual", Feature: "reviewer-ready", + Runner: &stubCaptureRunner{write: func(CaptureRequest) error { return nil }}, + }) + if err == nil || !strings.Contains(err.Error(), "unavailable") { + t.Fatalf("capture ran without a resolved repository command: %v", err) + } +} diff --git a/boatstack/cmd/boatstack-helper/main.go b/boatstack/cmd/boatstack-helper/main.go index 556293b..863a275 100644 --- a/boatstack/cmd/boatstack-helper/main.go +++ b/boatstack/cmd/boatstack-helper/main.go @@ -444,6 +444,74 @@ func recordPRVisualEvidenceCommand(arguments []string) int { return 0 } +func captureEvidenceCommand(arguments []string) int { + flags := flag.NewFlagSet("capture-evidence", flag.ContinueOnError) + repo := flags.String("repo", ".", "repository whose Git-common state owns the evidence") + capability := flags.String("capability", "visual", "evidence capability to capture") + feature := flags.String("feature", "", "managed Boatstack feature slug") + base := flags.String("base", "", "base branch for the product diff (defaults to the project default branch)") + if err := flags.Parse(arguments); err != nil { + return 2 + } + if *feature == "" { + return fail(fmt.Errorf("capture-evidence requires --feature")) + } + captured, err := boatstack.CaptureEvidence(boatstack.CaptureEvidenceOptions{ + Repo: *repo, Capability: *capability, Feature: *feature, Base: *base, + }) + if err != nil { + return fail(err) + } + value, err := boatstack.MarshalJSON(captured) + if err != nil { + return fail(err) + } + fmt.Print(string(value)) + return 0 +} + +func provisionCapabilityCommand(arguments []string) int { + flags := flag.NewFlagSet("provision-capability", flag.ContinueOnError) + repo := flags.String("repo", ".", "repository to inspect for evidence-capability provisioning") + capability := flags.String("capability", "visual", "evidence capability to provision") + if err := flags.Parse(arguments); err != nil { + return 2 + } + guide, err := boatstack.CapabilityProvisionGuide(*repo, *capability) + if err != nil { + return fail(err) + } + value, err := boatstack.MarshalJSON(guide) + if err != nil { + return fail(err) + } + fmt.Print(string(value)) + return 0 +} + +func capabilityRegisterCommand(arguments []string) int { + flags := flag.NewFlagSet("capability-register", flag.ContinueOnError) + repo := flags.String("repo", ".", "repository whose Boatstack configuration owns the command") + capability := flags.String("capability", "visual", "evidence capability to register a command for") + command := flags.String("command", "", "repository command that produces the evidence") + if err := flags.Parse(arguments); err != nil { + return 2 + } + if *command == "" { + return fail(fmt.Errorf("capability-register requires --command")) + } + registered, err := boatstack.RegisterCapabilityCommand(*repo, *capability, *command) + if err != nil { + return fail(err) + } + value, err := boatstack.MarshalJSON(registered) + if err != nil { + return fail(err) + } + fmt.Print(string(value)) + return 0 +} + func recordPRVisualPublicationCommand(arguments []string) int { flags := flag.NewFlagSet("record-pr-visual-publication", flag.ContinueOnError) repo := flags.String("repo", ".", "repository whose Git-common state owns the evidence") @@ -977,7 +1045,7 @@ func workspaceSyncCommand(arguments []string) int { func run() int { if len(os.Args) < 2 { - fmt.Fprintln(os.Stderr, "usage: boatstack-helper ") + fmt.Fprintln(os.Stderr, "usage: boatstack-helper ") return 2 } switch os.Args[1] { @@ -1027,6 +1095,12 @@ func run() int { return recordDeliveryGateCommand(os.Args[2:]) case "record-pr-visual-evidence": return recordPRVisualEvidenceCommand(os.Args[2:]) + case "capture-evidence": + return captureEvidenceCommand(os.Args[2:]) + case "provision-capability": + return provisionCapabilityCommand(os.Args[2:]) + case "capability-register": + return capabilityRegisterCommand(os.Args[2:]) case "record-pr-visual-publication": return recordPRVisualPublicationCommand(os.Args[2:]) case "pr-context": diff --git a/boatstack/export.go b/boatstack/export.go index 00d10a5..80f924c 100644 --- a/boatstack/export.go +++ b/boatstack/export.go @@ -268,7 +268,7 @@ func BuildExportBundle(configPath string, config ProjectConfig, rawConfig []byte operations := map[string]string{ "boatstack-next": "Run the project-local helper next-status --repo . --json. This operation is strictly read-only: do not run the reported operation, edit artifacts, contact GitHub beyond the helper's bounded published-PR inspection, or advance a gate. Translate the structured result into the canonical response contract. Show the verified feature and active slice when present. Distinguish NOT_STARTED, whose next operation is auto-plan run with the plan path via --plan, from PUBLISHED, which responds PR published and makes reviewing its checks the one action, and FEATURE_COMPLETE, which is reserved for a verified merged PR and responds Feature complete with No action required. If verification_status is BLOCKED, name the ambiguity or invalid evidence and make its safe restoration the one action; never clear artifacts. Conversation, terminal, worktree, or process observations may be included as clearly labeled context only and must never override the repository-backed result. Otherwise make the returned next_operation the one next action.", "boatstack-run": "First run the read-only next-status --repo . --json and operation-status --repo . --json. If an operation is executing, wait and report it instead of launching it again; if reconciliation is required, verify its exact postcondition before retrying. If NOT_STARTED, respond Start a Boatstack feature and ask the user for the plan produced in the host conversation, then execute auto-plan with its path via --plan (Boatstack does not scan directories for plans) without Git preflight, pausing at its normal decision or approval boundary; do not fetch or require a feature branch. If PUBLISHED, report that the PR is awaiting or lacks verified completion and make reviewing its checks the one next action; do not claim completion. If FEATURE_COMPLETE, respond Feature complete with No action required. Stop on UNVERIFIED, BLOCKED, ambiguous, stale, or invalid state. Before executing the first delivery-stage next_operation (build, repair, test-gate, review-gate, or ship-gate), run the project-local helper run-preflight --repo . --json; planning and plan-gate do not require it. Stop on a blocked preflight; never merge, rebase, force-push, discard changes, switch branches, or create a constrained delivery branch to repair freshness. Then execute exactly the verified next_operation using the canonical operation semantics, verify the resulting repository state, and resolve again. Continue across every declared delivery slice. Pause for the exact plan approval reply a, any material product decision, and the exact PR publication reply o or u; after a valid reply in the current host session, automatically continue the run. A run request never supplies approval or publication authority. For a same-intent test or review failure, use repair, record the observation, and retry from the returned stage. The delivery state's durable repair_attempt is the budget; stop after three complete automated repair-and-gate cycles even across new turns, host restarts, or async notifications. Stop immediately on an amendment, ambiguity, unsafe or destructive capability, stale evidence, branch mismatch, unsupported recovery, or exhausted repair budget. If Cursor reports MainThreadShellExec not initialized, explain that Cursor failed before the Boatstack hook started and make Developer: Reload Window the one recovery action; do not recommend reinstall unless Boatstack reports a missing, drifted, unsafe, or checksum-invalid runtime. Do not use conversation as workflow evidence. Durable operation receipts store execution facts and retry budgets, never autonomous workflow intent. Report the feature, active slice, stages completed, completion or pause reason, durable repair-cycle count, and exactly one next action. Ship means publishing every declared slice PR for review; never merge or deploy.", - "auto-plan": "Take the plan produced in the host conversation, supplied explicitly via --plan (Boatstack never scans directories for plans), and refine it into a Markdown-only draft feature package whose canonical structured artifact is plan.md. Run check-plan read-only. If workflow.boundary_analysis is true, evaluate if the change is a symptom of a missing systemic boundary and perform a rapid codebase scan for other vulnerabilities. Present this as a material product decision with tiered paths: [1a] Symptom Patch or [1b] Programmatic Enforcement (Slice 1 for the boundary, Slice 2 for the feature). When workflow.pr_visual_evidence is suggest or require, record a structural pr_visual_evidence decision: relevant with one to three entry/state/viewport/expected scenarios, or not_relevant with a reason. Discover existing visual tooling but never require a frontend framework or add repository tooling during planning. Record affected_paths and structured side_effects for external writes; use an immutable target identity, transactional or fix-forward recovery, and destructive=false. When workflow.maintain_changelog is true, include CHANGELOG.md in every delivery slice's affected paths. Keep internal phases as tasks in one delivery slice. Only when the accepted outcome explicitly needs multiple PRs, declare ordered delivery_slices and assign every task exactly once; plan approval never authorizes publication. Do not implement, create JSON or locks, or imply acceptance. If ready, respond with Plan ready and make Run /plan-gate the one next action. If decisions remain, respond with I need your input and ask only 1-3 material questions.", + "auto-plan": "Take the plan produced in the host conversation, supplied explicitly via --plan (Boatstack never scans directories for plans), and refine it into a Markdown-only draft feature package whose canonical structured artifact is plan.md. Run check-plan read-only. If workflow.boundary_analysis is true, evaluate if the change is a symptom of a missing systemic boundary and perform a rapid codebase scan for other vulnerabilities. Present this as a material product decision with tiered paths: [1a] Symptom Patch or [1b] Programmatic Enforcement (Slice 1 for the boundary, Slice 2 for the feature). When workflow.pr_visual_evidence is suggest or require, record a structural pr_visual_evidence decision: relevant with one to three entry/state/viewport/expected scenarios, or not_relevant with a reason. Discover existing visual tooling but never require a frontend framework or add repository tooling during planning. When a scenario is relevant but no capability command resolves, surface a material provisioning decision with tiered paths: [1a] provision the capture capability now as its own ordered delivery slice, [1b] bundle the capture harness into the feature slice, or [1c] record the gap and defer; this is a surfaced choice, never an imposed framework. Record affected_paths and structured side_effects for external writes; use an immutable target identity, transactional or fix-forward recovery, and destructive=false. When workflow.maintain_changelog is true, include CHANGELOG.md in every delivery slice's affected paths. Keep internal phases as tasks in one delivery slice. Only when the accepted outcome explicitly needs multiple PRs, declare ordered delivery_slices and assign every task exactly once; plan approval never authorizes publication. Do not implement, create JSON or locks, or imply acceptance. If ready, respond with Plan ready and make Run /plan-gate the one next action. If decisions remain, respond with I need your input and ask only 1-3 material questions.", "plan-gate": "Run check-plan read-only and present its plan fingerprint, baseline product diff fingerprint, changed paths, exact baseline diff when non-empty, and all open decisions. If workflow.human_plan_approval is true, require explicit human approval. While plan approval is pending, the normal user action is the exact standalone reply a. Trim surrounding whitespace and match a case-insensitively; do not treat [a] or an a embedded in other text as approval. Continue accepting the full reply approve for compatibility, but do not advertise it in the user-facing response. Resolve approved_by from an explicit supplied identity, otherwise from the authenticated GitHub login when available; ask one short identity follow-up only when neither exists, and never infer it from a filesystem username, commit history, or agent identity. On approval invoke record-approval with the displayed baseline fingerprint, omitting it only when the baseline is clean, so it writes only approval.md. While pending respond Ready for your approval and render: Reply `a` to approve. After recording respond Approved — ready to build. If human_plan_approval is false, do not request approval or create approval.md; state that Build will create a fingerprinted policy-activation lock. In either mode Remain in Plan mode, do not compile, and make entering execution mode and running /build the next action once ready.", "build": "First confirm the host is in an execution-capable mode. If the mode transition is rejected or product-code writes remain unavailable, return READY_FOR_BUILD internally without activating the plan, compiling JSON, or writing a lock. Only then locate plan.md and, when workflow.human_plan_approval is true, approval.md; run activate-plan before the first product-code edit and omit --approval for policy activation. Stop if it reports BLOCKED. Read delivery-status and implement only the active delivery slice task_ids. When workflow.maintain_changelog is true, add a concise entry grounded in the active slice's actual changes under the current CHANGELOG.md Unreleased heading before recording test evidence. Use only the one allowed category needed by the entry and do not add empty category headings. If the file is absent, create the documented minimal skeleton with ## [Unreleased] - YYYY-MM-DD and the first categorized entry; if it exists, add to the current file without rewriting its history or layout. Run the internal repository safety check after operational or high-risk edits; a destructive capability blocks execution and gate progression but does not block reviewable source editing. Implementation tactics remain open inside the authorized boundary, but push and PR mutation are never build tactics and are denied while managed delivery is active. On success respond Build complete and make Run /test-gate the one next action. When a new product decision blocks work, respond Build needs a decision and ask only that question.", "repair": "First run recovery-status --repo . with the user's exact free-form requested change, its observed source stage, bounded evidence when available, and --json. This resolver covers both active and current-branch published deliveries. On repair_active, read delivery-status, the current plan lock and acceptance criteria, the actual diff, and current receipts; classify the request and invoke record-change before any product edit. On draft_corrective_child, invoke record-change on the published parent, preserve its lock, receipts, slices, and publication evidence, and automatically prepare the suggested one-slice child plan with parent_delivery, exact correction, inherited intent, observed failure, returned existing_diff_sha256 and existing_changed_paths, verification requirements, and the resolved PR destination. Lead with The PR needs a corrective delivery. I prepared it for your approval. Then pause at the normal fingerprinted plan approval boundary; never reuse the parent's approval. An open PR reuses its verified head branch and is updated after fresh gates and publication confirmation. A merged or closed PR uses a fresh branch and PR; when a fingerprinted correction diff already exists, leave the original worktree untouched and transfer that exact reviewed diff into the fresh child only after approval. PUBLISHED_UNKNOWN may be drafted but its destination remains blocking at publication. Stop on BLOCKED and ask one targeted feature question using the returned blockers. If no managed target exists, continue ordinary conversation. Never discard pre-existing correction edits, edit runtime state directly, or bypass test, review, and ship gates. Never ask the user to repeat a denied push or PR mutation. If Cursor reports MainThreadShellExec not initialized, make Developer: Reload Window the one recovery action because Boatstack's hook did not start; reserve reinstall guidance for Boatstack runtime integrity errors.", diff --git a/boatstack/provision.go b/boatstack/provision.go new file mode 100644 index 0000000..08ee211 --- /dev/null +++ b/boatstack/provision.go @@ -0,0 +1,278 @@ +package boatstack + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" + "strings" +) + +// FrontendStack are the detected facts about a repository's frontend tooling. It +// is the input to a context-aware provisioning guide: it is discovered, never +// imposed, so a repository without a frontend framework is reported honestly +// rather than forced to adopt one. +type FrontendStack struct { + Framework string `json:"framework"` // next | vite | react | vue | svelte | angular | none + PackageManager string `json:"package_manager"` // npm | pnpm | yarn | bun | none + DevCommand string `json:"dev_command,omitempty"` + HasPlaywright bool `json:"has_playwright"` + HasCypress bool `json:"has_cypress"` + HasStorybook bool `json:"has_storybook"` +} + +// detectFrontendStack reads package.json, lockfiles, and known config files to +// report a repository's frontend facts. It mirrors detectTestCommand's idiom: +// evidence-driven, no network, and silent (all-zero) when nothing is found. +func detectFrontendStack(repo string) FrontendStack { + stack := FrontendStack{Framework: "none", PackageManager: "none"} + var manifest struct { + Scripts map[string]string `json:"scripts"` + Dependencies map[string]string `json:"dependencies"` + DevDependencies map[string]string `json:"devDependencies"` + } + value, err := os.ReadFile(filepath.Join(repo, "package.json")) + if err != nil || json.Unmarshal(value, &manifest) != nil { + return stack + } + + stack.PackageManager = detectPackageManager(repo) + dependency := func(name string) bool { + _, direct := manifest.Dependencies[name] + _, dev := manifest.DevDependencies[name] + return direct || dev + } + switch { + case dependency("next"): + stack.Framework = "next" + case dependency("@angular/core"): + stack.Framework = "angular" + case dependency("svelte"): + stack.Framework = "svelte" + case dependency("vue"): + stack.Framework = "vue" + case dependency("vite"): + stack.Framework = "vite" + case dependency("react"): + stack.Framework = "react" + } + for _, name := range []string{"dev", "start", "serve"} { + if strings.TrimSpace(manifest.Scripts[name]) != "" { + stack.DevCommand = packageManagerRun(stack.PackageManager, name) + break + } + } + stack.HasPlaywright = dependency("@playwright/test") || dependency("playwright") || + fileExists(filepath.Join(repo, "playwright.config.ts")) || fileExists(filepath.Join(repo, "playwright.config.js")) + stack.HasCypress = dependency("cypress") || + fileExists(filepath.Join(repo, "cypress.config.ts")) || fileExists(filepath.Join(repo, "cypress.config.js")) + stack.HasStorybook = dependency("storybook") || dependency("@storybook/react") || + dirExists(filepath.Join(repo, ".storybook")) + return stack +} + +func detectPackageManager(repo string) string { + switch { + case fileExists(filepath.Join(repo, "pnpm-lock.yaml")): + return "pnpm" + case fileExists(filepath.Join(repo, "yarn.lock")): + return "yarn" + case fileExists(filepath.Join(repo, "bun.lock")), fileExists(filepath.Join(repo, "bun.lockb")): + return "bun" + default: + return "npm" + } +} + +func packageManagerRun(manager, script string) string { + if manager == "" || manager == "none" { + manager = "npm" + } + return manager + " run " + script +} + +func dirExists(path string) bool { + info, err := os.Stat(path) + return err == nil && info.IsDir() +} + +// ProvisionGuide is the context-aware answer to "this repository cannot yet +// produce evidence — how do we help?". Boatstack ships the contract +// the in-repository harness must satisfy plus stack-tailored steps; the harness +// itself is authored in the user's repository, never shipped by Boatstack. +type ProvisionGuide struct { + Capability string `json:"capability"` + Tier string `json:"tier"` // available | provision | unsupported + Available bool `json:"available"` + ResolvedCommand string `json:"resolved_command,omitempty"` + Stack FrontendStack `json:"stack"` + SuggestedAlias string `json:"suggested_alias,omitempty"` + SuggestedCommand string `json:"suggested_command,omitempty"` + Contract []string `json:"contract"` + Steps []string `json:"steps"` +} + +// captureContract is the framework-agnostic contract every capture harness must +// satisfy, expressed as the environment the harness is invoked with. It mirrors +// execCaptureRunner so the guide and the runtime never drift. +var captureContract = []string{ + "Boatstack invokes the registered command once per scenario.", + "BOATSTACK_CAPTURE_CAPABILITY — the capability being captured (e.g. visual).", + "BOATSTACK_CAPTURE_SCENARIO_ID — the plan scenario id.", + "BOATSTACK_CAPTURE_ENTRY — the scenario entry point (route or component).", + "BOATSTACK_CAPTURE_STATE — the scenario state to render.", + "BOATSTACK_CAPTURE_VIEWPORT — the required viewport, e.g. 1440x900.", + "BOATSTACK_CAPTURE_OUTPUT — the absolute path the harness must write exactly one PNG to.", + "Render fixture or mock data only; never production secrets or PII.", +} + +// CapabilityProvisionGuide composes capability detection with frontend-stack +// facts to produce a context-aware provisioning guide for a repository. +func CapabilityProvisionGuide(repo, name string) (ProvisionGuide, error) { + resolved, err := ResolveRepository(repo) + if err != nil { + return ProvisionGuide{}, err + } + if strings.TrimSpace(name) == "" { + name = "visual" + } + capability, ok := LookupCapability(name) + if !ok { + return ProvisionGuide{}, fmt.Errorf("unknown evidence capability %q", name) + } + config, _, err := LoadConfig(filepath.Join(resolved, ".product-loop", "project.json")) + if err != nil { + return ProvisionGuide{}, fmt.Errorf("provisioning requires a valid Boatstack project configuration: %w", err) + } + resolution, err := ResolveCapability(name, config) + if err != nil { + return ProvisionGuide{}, err + } + + guide := ProvisionGuide{Capability: capability.Name, Contract: captureContract, Stack: detectFrontendStack(resolved)} + if resolution.Kind == "repository-command" { + guide.Tier = "available" + guide.Available = true + guide.ResolvedCommand = resolution.Command + guide.Steps = []string{ + fmt.Sprintf("%s evidence is already wired: project.commands resolves %q.", capability.Name, resolution.Command), + "Run capture-evidence to generate evidence for the active feature.", + } + return guide, nil + } + + guide.SuggestedAlias = capability.Name + if guide.Stack.Framework == "none" { + guide.Tier = "unsupported" + guide.Steps = []string{ + "No frontend framework was detected, so visual evidence cannot be captured yet.", + "Boatstack never adds a framework for you: if this change is user-visible, set one up (or opt out by recording the scenario as not_relevant).", + "Once a framework and dev command exist, re-run provisioning to get a stack-tailored harness guide.", + } + return guide, nil + } + + guide.Tier = "provision" + guide.SuggestedCommand = packageManagerRun(guide.Stack.PackageManager, "capture:"+capability.Name) + guide.Steps = provisionSteps(guide.Stack, guide.SuggestedAlias, guide.SuggestedCommand) + return guide, nil +} + +func provisionSteps(stack FrontendStack, alias, suggestedCommand string) []string { + steps := []string{ + fmt.Sprintf("Detected a %s app using %s.", stack.Framework, stack.PackageManager), + } + if stack.HasPlaywright { + steps = append(steps, "Reuse the existing Playwright setup: add a script that reads the BOATSTACK_CAPTURE_* env and screenshots the scenario to BOATSTACK_CAPTURE_OUTPUT.") + } else { + steps = append(steps, "Add a headless rasterizer (Playwright is the least-effort fit) that renders one scenario and writes it to BOATSTACK_CAPTURE_OUTPUT.") + } + if stack.HasStorybook { + steps = append(steps, "Storybook is present: map each scenario id to a story so the harness renders isolated component state with fixture data.") + } + steps = append(steps, + fmt.Sprintf("Expose the harness as a project command, then register it: capability-register --capability %s --command \"%s\".", alias, suggestedCommand), + "Confirm registration with provision-capability; its tier should flip to available. Then run capture-evidence.", + ) + return steps +} + +// RegisteredCapability reports the outcome of registering a capability command. +type RegisteredCapability struct { + Capability string `json:"capability"` + Alias string `json:"alias"` + Command string `json:"command"` + Source string `json:"source"` // source-and-export | generated-only +} + +// RegisterCapabilityCommand records a repository-owned command for a capability. +// When a canonical .boatstack-project.json source exists it mutates the source +// and regenerates the full export so the source and every generated file stay in +// sync; otherwise it round-trips the generated project.json alone, matching the +// IgnoreDelivery idiom. +func RegisterCapabilityCommand(repo, name, command string) (RegisteredCapability, error) { + resolved, err := ResolveRepository(repo) + if err != nil { + return RegisteredCapability{}, err + } + capability, ok := LookupCapability(strings.TrimSpace(name)) + if !ok { + return RegisteredCapability{}, fmt.Errorf("unknown evidence capability %q", name) + } + command = strings.TrimSpace(command) + if command == "" { + return RegisteredCapability{}, fmt.Errorf("capability-register requires a non-empty --command") + } + + sourcePath := filepath.Join(resolved, ".boatstack-project.json") + if fileExists(sourcePath) { + config, _, err := LoadConfig(sourcePath) + if err != nil { + return RegisteredCapability{}, err + } + setCapabilityCommand(&config, capability.Name, command) + if err := ValidateConfig(config); err != nil { + return RegisteredCapability{}, err + } + rawConfig, err := MarshalJSON(config) + if err != nil { + return RegisteredCapability{}, err + } + bundle, err := BuildExportBundle(sourcePath, config, rawConfig, "boatstack") + if err != nil { + return RegisteredCapability{}, err + } + if err := WriteExport(resolved, bundle.Files); err != nil { + return RegisteredCapability{}, err + } + if err := atomicWriteMode(sourcePath, rawConfig, 0o644); err != nil { + return RegisteredCapability{}, err + } + return RegisteredCapability{Capability: capability.Name, Alias: capability.Name, Command: command, Source: "source-and-export"}, nil + } + + configPath := filepath.Join(resolved, ".product-loop", "project.json") + config, _, err := LoadConfig(configPath) + if err != nil { + return RegisteredCapability{}, err + } + setCapabilityCommand(&config, capability.Name, command) + if err := ValidateConfig(config); err != nil { + return RegisteredCapability{}, err + } + value, err := GeneratedJSON(config) + if err != nil { + return RegisteredCapability{}, err + } + if err := atomicWriteMode(configPath, value, 0o644); err != nil { + return RegisteredCapability{}, err + } + return RegisteredCapability{Capability: capability.Name, Alias: capability.Name, Command: command, Source: "generated-only"}, nil +} + +func setCapabilityCommand(config *ProjectConfig, alias, command string) { + if config.Project.Commands == nil { + config.Project.Commands = map[string]string{} + } + config.Project.Commands[alias] = command +} diff --git a/boatstack/provision_test.go b/boatstack/provision_test.go new file mode 100644 index 0000000..77bdc83 --- /dev/null +++ b/boatstack/provision_test.go @@ -0,0 +1,170 @@ +package boatstack + +import ( + "os" + "path/filepath" + "testing" +) + +func writeProjectConfig(t *testing.T, repo string, mutate func(*ProjectConfig)) { + t.Helper() + runGit(t, repo, "init", "-b", "main") + config := testConfig() + config.Project.DefaultBranch = "main" + if mutate != nil { + mutate(&config) + } + value, err := MarshalJSON(config) + if err != nil { + t.Fatal(err) + } + if err := os.MkdirAll(filepath.Join(repo, ".product-loop"), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(repo, ".product-loop", "project.json"), value, 0o644); err != nil { + t.Fatal(err) + } +} + +func TestDetectFrontendStackReadsManifestAndTooling(t *testing.T) { + repo := t.TempDir() + manifest := `{ + "scripts": {"dev": "vite", "test": "vitest"}, + "dependencies": {"react": "^18.0.0", "vite": "^5.0.0"}, + "devDependencies": {"@playwright/test": "^1.40.0"} + }` + if err := os.WriteFile(filepath.Join(repo, "package.json"), []byte(manifest), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(repo, "pnpm-lock.yaml"), []byte("lockfileVersion: '9.0'\n"), 0o644); err != nil { + t.Fatal(err) + } + stack := detectFrontendStack(repo) + if stack.Framework != "vite" || stack.PackageManager != "pnpm" || stack.DevCommand != "pnpm run dev" || !stack.HasPlaywright { + t.Fatalf("unexpected stack facts: %#v", stack) + } +} + +func TestDetectFrontendStackReportsNoneWithoutManifest(t *testing.T) { + stack := detectFrontendStack(t.TempDir()) + if stack.Framework != "none" || stack.PackageManager != "none" || stack.DevCommand != "" { + t.Fatalf("empty repo should report no frontend stack: %#v", stack) + } +} + +func TestProvisionGuideReportsAvailableWhenCommandResolves(t *testing.T) { + repo := t.TempDir() + writeProjectConfig(t, repo, func(config *ProjectConfig) { + config.Project.Commands["visual"] = "pnpm run capture:visual" + }) + guide, err := CapabilityProvisionGuide(repo, "visual") + if err != nil { + t.Fatal(err) + } + if guide.Tier != "available" || !guide.Available || guide.ResolvedCommand != "pnpm run capture:visual" { + t.Fatalf("resolved capability should be available: %#v", guide) + } +} + +func TestProvisionGuideTailorsStepsToDetectedStack(t *testing.T) { + repo := t.TempDir() + writeProjectConfig(t, repo, nil) // no visual command → must provision + manifest := `{"scripts": {"dev": "next dev", "test": "jest"}, "dependencies": {"next": "^14.0.0"}}` + if err := os.WriteFile(filepath.Join(repo, "package.json"), []byte(manifest), 0o644); err != nil { + t.Fatal(err) + } + guide, err := CapabilityProvisionGuide(repo, "visual") + if err != nil { + t.Fatal(err) + } + if guide.Tier != "provision" || guide.Available { + t.Fatalf("missing command should require provisioning: %#v", guide) + } + if guide.Stack.Framework != "next" || guide.SuggestedCommand != "npm run capture:visual" { + t.Fatalf("guide did not reflect the detected stack: %#v", guide) + } + if len(guide.Contract) == 0 || len(guide.Steps) == 0 { + t.Fatalf("provision guide must ship a contract and steps: %#v", guide) + } +} + +func TestProvisionGuideReportsUnsupportedWithoutFrontend(t *testing.T) { + repo := t.TempDir() + writeProjectConfig(t, repo, nil) // no command, no package.json + guide, err := CapabilityProvisionGuide(repo, "visual") + if err != nil { + t.Fatal(err) + } + if guide.Tier != "unsupported" || guide.SuggestedCommand != "" { + t.Fatalf("a backend-only repo should be unsupported, never forced: %#v", guide) + } +} + +func TestRegisterCapabilityCommandSyncsSourceAndExport(t *testing.T) { + repo := t.TempDir() + runGit(t, repo, "init", "-b", "main") + config := testConfig() + config.Project.DefaultBranch = "main" + raw, err := MarshalJSON(config) + if err != nil { + t.Fatal(err) + } + // Seed a canonical source plus its generated export, as a real install would. + if err := os.WriteFile(filepath.Join(repo, ".boatstack-project.json"), raw, 0o644); err != nil { + t.Fatal(err) + } + bundle, err := BuildExportBundle(".boatstack-project.json", config, raw, "boatstack") + if err != nil { + t.Fatal(err) + } + if err := WriteExport(repo, bundle.Files); err != nil { + t.Fatal(err) + } + + result, err := RegisterCapabilityCommand(repo, "visual", "pnpm run capture:visual") + if err != nil { + t.Fatal(err) + } + if result.Source != "source-and-export" || result.Alias != "visual" { + t.Fatalf("unexpected registration outcome: %#v", result) + } + // Both the source and the generated export must now resolve the command. + source, _, err := LoadConfig(filepath.Join(repo, ".boatstack-project.json")) + if err != nil { + t.Fatal(err) + } + generated, _, err := LoadConfig(filepath.Join(repo, ".product-loop", "project.json")) + if err != nil { + t.Fatal(err) + } + if source.Project.Commands["visual"] != "pnpm run capture:visual" || generated.Project.Commands["visual"] != "pnpm run capture:visual" { + t.Fatalf("source and export drifted: source=%q export=%q", source.Project.Commands["visual"], generated.Project.Commands["visual"]) + } + // And provisioning must now report the capability as available. + guide, err := CapabilityProvisionGuide(repo, "visual") + if err != nil { + t.Fatal(err) + } + if guide.Tier != "available" { + t.Fatalf("registration did not make the capability available: %#v", guide) + } +} + +func TestRegisterCapabilityCommandFallsBackToGeneratedConfig(t *testing.T) { + repo := t.TempDir() + writeProjectConfig(t, repo, nil) // generated project.json only, no source + result, err := RegisterCapabilityCommand(repo, "visual", "npm run capture:visual") + if err != nil { + t.Fatal(err) + } + if result.Source != "generated-only" { + t.Fatalf("expected generated-only registration: %#v", result) + } + generated, _, err := LoadConfig(filepath.Join(repo, ".product-loop", "project.json")) + if err != nil { + t.Fatal(err) + } + if generated.Project.Commands["visual"] != "npm run capture:visual" { + t.Fatalf("command was not persisted: %#v", generated.Project.Commands) + } +} diff --git a/boatstack/visual_evidence.go b/boatstack/visual_evidence.go index ca4de7e..1f22727 100644 --- a/boatstack/visual_evidence.go +++ b/boatstack/visual_evidence.go @@ -77,13 +77,17 @@ type PRVisualCaptureCapability struct { Command string `json:"command,omitempty"` } -// ResolvePRVisualCaptureCapability implements the portable capability cut. It -// selects repository-owned tooling before host or machine-local capabilities. +// ResolvePRVisualCaptureCapability implements the portable capability cut for the +// visual capability. It selects repository-owned tooling (via the generic +// ResolveCapability spine) before host or machine-local capabilities. The +// browser-specific fallbacks below the repository cut are visual-only. func ResolvePRVisualCaptureCapability(repo string, config ProjectConfig, hostBrowser bool, suppliedLaunch string, expectedReceipt PRVisualCapabilityReceipt) (PRVisualCaptureCapability, error) { - for _, name := range []string{"visual", "screenshot", "e2e"} { - if command := strings.TrimSpace(config.Project.Commands[name]); command != "" { - return PRVisualCaptureCapability{Kind: "repository-command", Command: command}, nil - } + resolution, err := ResolveCapability("visual", config) + if err != nil { + return PRVisualCaptureCapability{}, err + } + if resolution.Kind == "repository-command" { + return PRVisualCaptureCapability{Kind: "repository-command", Command: resolution.Command}, nil } if hostBrowser { return PRVisualCaptureCapability{Kind: "host-browser"}, nil diff --git a/docs/evidence-engineered-coding.md b/docs/evidence-engineered-coding.md index 6a033b8..b69a4c2 100644 --- a/docs/evidence-engineered-coding.md +++ b/docs/evidence-engineered-coding.md @@ -146,6 +146,6 @@ Delivery and system improvement also remain separate. A failed task may suggest ## What is evidence-backed -The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`3e44eb72add2a33fb84bbc5a0142af9c1828feb1`](https://github.com/operatorstack/intelligence-flow/tree/3e44eb72add2a33fb84bbc5a0142af9c1828feb1/labs/12-product-engineering-loop). +The current moves were derived from the Intelligence Flow benchmark corpus and product-repository studies. The generated source commit is [`5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d`](https://github.com/operatorstack/intelligence-flow/tree/5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d/labs/12-product-engineering-loop). The evidence supports specific failure mechanisms and guardrails. It does not establish that Boatstack is optimal, that control-theory notation proves software quality, or that one workflow dominates every team. Those are evaluation questions, so the distribution preserves measurements, provenance, gaps, and negative results. diff --git a/docs/public-claims.json b/docs/public-claims.json index 564f1e8..e03a90b 100644 --- a/docs/public-claims.json +++ b/docs/public-claims.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "source_commit": "3e44eb72add2a33fb84bbc5a0142af9c1828feb1", + "source_commit": "5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d", "statuses": ["verified", "observed", "still_being_evaluated"], "claims": [ { @@ -12,7 +12,7 @@ "readable_evidence": "why-these-steps.md#portable-workflow-and-state", "implementation": ["../boatstack/export.go", "../boatstack/references/artifacts.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "human-decisions", @@ -23,7 +23,7 @@ "readable_evidence": "why-these-steps.md#human-decisions", "implementation": ["../boatstack/references/workflow.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "validation-provenance", @@ -34,7 +34,7 @@ "readable_evidence": "why-these-steps.md#validation-provenance", "implementation": ["validation-and-evidence.md", "../boatstack/plan.go"], "verification": ["../boatstack/plan_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "irreversible-operations", @@ -46,7 +46,7 @@ "readable_evidence": "why-these-steps.md#irreversible-operations", "implementation": ["safety.md", "../boatstack/safety.go", "../boatstack/hooks.go"], "verification": ["../boatstack/safety_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "reviewer-ready-pr", @@ -57,7 +57,7 @@ "readable_evidence": "why-these-steps.md#reviewer-ready-pr", "implementation": ["../boatstack/pr.go", "getting-started.md"], "verification": ["../boatstack/pr_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "phase-scoped-delivery", @@ -68,7 +68,7 @@ "readable_evidence": "why-these-steps.md#phase-scoped-delivery", "implementation": ["../boatstack/delivery.go", "../boatstack/safety.go", "../boatstack/hooks.go", "../boatstack/references/workflow.md"], "verification": ["../boatstack/delivery_test.go", "../boatstack/pr_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "model-neutral-contract", @@ -79,7 +79,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md", "../boatstack/references/workflow.md"], "verification": ["../boatstack/export_test.go", "../boatstack/planning_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "cross-model-failures", @@ -90,7 +90,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "lower-cost-outcomes", @@ -101,7 +101,7 @@ "readable_evidence": "why-these-steps.md#model-choice-and-budget", "implementation": ["research-and-design.md"], "verification": ["benchmark-corpus-audit.md", "benchmark-submission-audit.md"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "git-worktree-activation", @@ -112,7 +112,7 @@ "readable_evidence": "why-these-steps.md#git-worktree-activation", "implementation": ["../boatstack/runtime_cache.go", "../boatstack/hooks.go"], "verification": ["../boatstack/runtime_cache_test.go", "../boatstack/hooks_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" }, { "id": "visible-updates", @@ -123,7 +123,7 @@ "readable_evidence": "why-these-steps.md#visible-updates", "implementation": ["../boatstack/update.go", "../boatstack/init.go"], "verification": ["../boatstack/update_test.go", "../boatstack/init_test.go", "../boatstack/export_test.go"], - "last_verified_version": "source:3e44eb72add2a33fb84bbc5a0142af9c1828feb1" + "last_verified_version": "source:5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d" } ] } diff --git a/labs/diagram-json/plan.lock.json b/labs/diagram-json/plan.lock.json index e12145a..f805554 100644 --- a/labs/diagram-json/plan.lock.json +++ b/labs/diagram-json/plan.lock.json @@ -6,7 +6,7 @@ "plan_path": "labs/diagram-json/plan.md", "plan_sha256": "3cc4f533b8d69386deff16b3a594a3ba09d4c0c3db636cccd8c4380084ce6a51", "schema_version": 1, - "source_commit": "3e44eb72add2a33fb84bbc5a0142af9c1828feb1", + "source_commit": "5f4dda1585d8032019dd0ed2c3b8ea264abfcf4d", "source_plan_path": "labs/diagram-json/source-plan.md", "source_plan_sha256": "e10593ddaa7522ab80cc991d0a09399257139799e37f737794cd49d68a39985b", "spec_path": "labs/diagram-json/spec.md", diff --git a/release-notes/2026-07-23-capability-provisioning.md b/release-notes/2026-07-23-capability-provisioning.md new file mode 100644 index 0000000..80ffdb7 --- /dev/null +++ b/release-notes/2026-07-23-capability-provisioning.md @@ -0,0 +1,21 @@ +### Boatstack helps repositories that cannot yet produce evidence + +When a feature needs visual evidence but the repository has no command to produce +it, Boatstack now guides you to set one up instead of silently leaving a gap. A +new `provision-capability` operation detects the repository's frontend stack +(framework, package manager, and whether Playwright, Cypress, or Storybook are +present) and returns a context-aware guide: the framework-agnostic contract the +in-repository capture harness must satisfy, plus stack-tailored steps to build +it. Boatstack ships the contract and the guide, never the harness itself. + +`capability-register` then records the repository-owned command in one step, +keeping the canonical `.boatstack-project.json` source and every generated file +in sync (or round-tripping the generated configuration alone when there is no +source). Once registered, provisioning reports the capability as available and +`capture-evidence` can run. + +Planning is aware of this too: when a visual scenario is relevant but no capture +command resolves, auto-plan surfaces a material provisioning decision with tiered +paths — provision now as its own delivery slice, bundle the harness into the +feature slice, or record the gap and defer. It remains a surfaced choice; +Boatstack never imposes a frontend framework. diff --git a/release-notes/2026-07-23-capture-evidence-orchestration.md b/release-notes/2026-07-23-capture-evidence-orchestration.md new file mode 100644 index 0000000..01dbcb2 --- /dev/null +++ b/release-notes/2026-07-23-capture-evidence-orchestration.md @@ -0,0 +1,19 @@ +### Boatstack can now capture PR visual evidence for you + +A new `capture-evidence` operation turns a repository's declared visual scenarios +into trusted PR evidence without manual screenshotting. It resolves the +repository-owned capability command, reads the scenarios recorded in the feature +plan, and runs each one as a supervised, fingerprinted operation — retrying a +flaky capture within a bounded budget and reusing a successful capture on the +same commit instead of re-running it. + +Each screenshot the harness produces is conformance-checked before it is +ingested: the manifest is stamped to the current head commit and product diff, so +`pr-context` trusts it as PASS only while it still matches the change under +review. A harness that reports success but produces a non-conformant artifact +fails closed — capture never records evidence it cannot stand behind. + +Boatstack ships the capture contract, not the harness. The repository owns a +`visual` (or `screenshot`/`e2e`) command that renders one PNG per scenario, and +Boatstack invokes it through a stable environment-variable contract +(`BOATSTACK_CAPTURE_SCENARIO_ID`, `_ENTRY`, `_STATE`, `_VIEWPORT`, `_OUTPUT`). diff --git a/release-notes/2026-07-23-evidence-capability-substrate.md b/release-notes/2026-07-23-evidence-capability-substrate.md new file mode 100644 index 0000000..4840198 --- /dev/null +++ b/release-notes/2026-07-23-evidence-capability-substrate.md @@ -0,0 +1,9 @@ +### Evidence capabilities are now a shared, extensible substrate + +Boatstack's PR evidence detection is no longer hard-wired to a single evidence +type. A generic evidence-capability registry now backs the "can this repository +produce the evidence?" cut, and visual evidence is its first tenant. Nothing +changes in how visual evidence behaves today — the same repository commands, +statuses, and fingerprints — but the shared spine means future evidence types +(and the upcoming capture and provisioning flows) plug in through one registry +instead of a bespoke path each time. diff --git a/release-notes/2026-07-23-labkit-standalone-package.md b/release-notes/2026-07-23-labkit-standalone-package.md new file mode 100644 index 0000000..c669130 --- /dev/null +++ b/release-notes/2026-07-23-labkit-standalone-package.md @@ -0,0 +1,8 @@ +### Publishing toolkit moved to a standalone package + +The `labkit` toolkit that projects Boatstack into this repository moved from a +submodule of the monorepo's Python package to a standalone top-level `labkit/` +project, invoked as `python -m labkit`. This only changes where the publishing +tooling lives and how the sync workflow installs it; the projected files are +byte-for-byte identical (asserted by the golden reproduction test), with no +effect on Boatstack's runtime, CLI, skill, or public contract.