From ebdba81108d4f85ec1c4d84468e9e2454f60dddd Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 18:50:46 -0500 Subject: [PATCH 1/9] refactor(toolchain): give the resolved go binary one owner internal/gobuildrunner resolved `go` to an absolute path and internal/commandscope is about to need the same answer, and the two must agree: a scope resolved by one toolchain and a build made by another would compare two compilers' answers and the disagreement would look like a property of the module under test. The resolution moves to internal/gotoolchain, unchanged, with the reasoning that earned it -- PATH decides which compiler runs unless GOROOT is preferred (go:S4036, CWE-426) -- and a test that pins both branches. Its mutants move with it, so mutantsPerReleaseOnThisRepository is unchanged at 873. --- internal/gobuildrunner/gobuildrunner.go | 44 +--------- internal/gobuildrunner/module_scope.go | 3 +- internal/gotoolchain/gotoolchain.go | 50 +++++++++++ internal/gotoolchain/gotoolchain_test.go | 101 +++++++++++++++++++++++ 4 files changed, 155 insertions(+), 43 deletions(-) create mode 100644 internal/gotoolchain/gotoolchain.go create mode 100644 internal/gotoolchain/gotoolchain_test.go diff --git a/internal/gobuildrunner/gobuildrunner.go b/internal/gobuildrunner/gobuildrunner.go index 5645775..8ae4521 100644 --- a/internal/gobuildrunner/gobuildrunner.go +++ b/internal/gobuildrunner/gobuildrunner.go @@ -10,16 +10,15 @@ package gobuildrunner import ( "errors" - "go/build" "os" "os/exec" "path/filepath" - "runtime" "strconv" "strings" "github.com/Disble/ditto/internal/cmdtestrunner" "github.com/Disble/ditto/internal/ditto" + "github.com/Disble/ditto/internal/gotoolchain" "github.com/Disble/ditto/internal/result" ) @@ -46,52 +45,13 @@ type GoBuildRunner struct { } func New(packagePath string) *GoBuildRunner { - return &GoBuildRunner{packagePath: packagePath, toolchain: goToolchain()} + return &GoBuildRunner{packagePath: packagePath, toolchain: gotoolchain.Path()} } // Toolchain is the absolute path of the `go` binary this runner builds with, or // empty when none could be found. func (r *GoBuildRunner) Toolchain() string { return r.toolchain } -// goToolchain resolves the compiler once, to an absolute path, instead of naming -// it and letting the operating system search. -// -// Two reasons, pointing the same way. `exec.Command("go", …)` reads PATH at -// every call, so a directory an attacker can write to — or prepend — decides -// which compiler runs. That is SonarQube's go:S4036, and it is the same shape as -// git's inherited addressing that `environment` below strips: something ambient -// deciding what a subprocess really is. -// -// The other reason matters more here. Ditto exists to compare verdicts, and -// building a mutant's tests with a different toolchain from the one running the -// suite would make a disagreement that is nobody's mutation look like one that -// is. GOROOT is the toolchain that built this binary, so it is preferred, and -// PATH is the fallback for a GOROOT that is not on disk. -// -// It resolves to empty rather than panicking. A build that cannot happen is an -// answer the caller already knows how to take: Built stays false and the file -// falls back to the path ditto has always taken. -func goToolchain() string { - name := "go" - if runtime.GOOS == "windows" { - name += ".exe" - } - - if root := build.Default.GOROOT; root != "" { - candidate := filepath.Join(root, "bin", name) - if info, err := os.Stat(candidate); err == nil && !info.IsDir() { - return candidate - } - } - - found, err := exec.LookPath("go") - if err != nil { - return "" - } - - return found -} - // Select is which mutant the next run asks the binary for. func (r *GoBuildRunner) Select(mutant int) { r.mutant = mutant } diff --git a/internal/gobuildrunner/module_scope.go b/internal/gobuildrunner/module_scope.go index 6d565f5..edc10ca 100644 --- a/internal/gobuildrunner/module_scope.go +++ b/internal/gobuildrunner/module_scope.go @@ -17,6 +17,7 @@ import ( "github.com/Disble/ditto/internal/cmdtestrunner" "github.com/Disble/ditto/internal/ditto" + "github.com/Disble/ditto/internal/gotoolchain" "github.com/Disble/ditto/internal/result" ) @@ -82,7 +83,7 @@ var errEmptyModuleScope = errors.New("ditto: module scope discovered no packages // NewModuleScope returns a runner for the exact default ./... Go test scope. func NewModuleScope() *ModuleScopeRunner { - return &ModuleScopeRunner{toolchain: goToolchain()} + return &ModuleScopeRunner{toolchain: gotoolchain.Path()} } // Toolchain is the resolved Go executable, or empty when none could be found. diff --git a/internal/gotoolchain/gotoolchain.go b/internal/gotoolchain/gotoolchain.go new file mode 100644 index 0000000..cb07c8b --- /dev/null +++ b/internal/gotoolchain/gotoolchain.go @@ -0,0 +1,50 @@ +// Package gotoolchain resolves the Go toolchain once, to an absolute path. +// +// Two packages ask it, and they have to agree. internal/gobuildrunner compiles a +// mutant's tests with it; internal/commandscope asks it which packages those +// tests compile. A scope resolved by one toolchain and a build made by another +// would compare answers from two different compilers, and the disagreement would +// look like a property of the module under test. +// +// Resolving is not tidiness. `exec.Command("go", …)` reads PATH at every call, +// so a directory an attacker can write to — or prepend — decides which compiler +// runs: SonarQube's go:S4036, CWE-426. GOROOT is the toolchain that built this +// binary, so it is preferred; PATH is the fallback for a GOROOT that is not on +// disk. +// +// It answers empty rather than failing. A caller that cannot build already knows +// what to do with that — `gobuildrunner` reports that the binary was never +// built, and a scope that cannot be resolved refuses nothing — and a panic here +// would turn a missing toolchain into a defect that looks like ditto's. +package gotoolchain + +import ( + "go/build" + "os" + "os/exec" + "path/filepath" + "runtime" +) + +// Path is the absolute path of the `go` binary to run, or empty when none could +// be found. +func Path() string { + name := "go" + if runtime.GOOS == "windows" { + name += ".exe" + } + + if root := build.Default.GOROOT; root != "" { + candidate := filepath.Join(root, "bin", name) + if info, err := os.Stat(candidate); err == nil && !info.IsDir() { + return candidate + } + } + + found, err := exec.LookPath("go") + if err != nil { + return "" + } + + return found +} diff --git a/internal/gotoolchain/gotoolchain_test.go b/internal/gotoolchain/gotoolchain_test.go new file mode 100644 index 0000000..05ce130 --- /dev/null +++ b/internal/gotoolchain/gotoolchain_test.go @@ -0,0 +1,101 @@ +package gotoolchain_test + +import ( + "go/build" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "github.com/Disble/ditto/internal/gotoolchain" +) + +// TestPathPrefersGOROOT covers the resolution CI and a local run both depend on: +// the toolchain that built this binary is the one that compiles a mutant's +// tests. It is also the half that keeps PATH — writable by whoever can prepend a +// directory to it — from deciding which compiler runs (go:S4036, CWE-426). +func TestPathPrefersGOROOT(t *testing.T) { + root := t.TempDir() + + // Both names, because the lookup appends .exe on Windows and does not + // anywhere else: a test that created one of them would be silent about the + // other platform, and this repository's worst bugs have been platform-shaped. + writeFile(t, filepath.Join(root, "bin", "go")) + writeFile(t, filepath.Join(root, "bin", "go.exe")) + + restore := setGOROOT(t, root) + defer restore() + + got := gotoolchain.Path() + if !strings.HasPrefix(got, filepath.Clean(root)) { + t.Fatalf("Path() = %q, which is not inside the GOROOT it was given (%q)", got, root) + } +} + +// TestPathRefusesADirectoryNamedGo is the control for the guard inside the +// GOROOT branch: a path that exists but is not a file is not a toolchain. A +// directory there would otherwise be handed to exec.Command and fail at every +// call, which reads as a broken module rather than a broken GOROOT. +func TestPathRefusesADirectoryNamedGo(t *testing.T) { + root := t.TempDir() + + for _, name := range []string{"go", "go.exe"} { + if err := os.MkdirAll(filepath.Join(root, "bin", name), 0o750); err != nil { + t.Fatalf("creating the decoy directory: %v", err) + } + } + + restore := setGOROOT(t, root) + defer restore() + + got := gotoolchain.Path() + if got == "" { + t.Skip("no go binary on PATH to fall back to") + } + + if strings.HasPrefix(got, filepath.Clean(root)) { + t.Fatalf("Path() = %q, which is the directory it must refuse", got) + } +} + +// TestPathFallsBackToPath is the other half of the contract, and it is asserted +// against LookPath rather than against a literal: whatever PATH resolves `go` to +// is what a caller gets when GOROOT has no binary to offer. +func TestPathFallsBackToPath(t *testing.T) { + restore := setGOROOT(t, filepath.Join(t.TempDir(), "no-toolchain-here")) + defer restore() + + want, err := exec.LookPath("go") + if err != nil { + t.Skip("no go binary on PATH, so there is no fallback to assert") + } + + if got := gotoolchain.Path(); got != want { + t.Fatalf("Path() = %q, want the PATH answer %q", got, want) + } +} + +func writeFile(t *testing.T, path string) { + t.Helper() + + if err := os.MkdirAll(filepath.Dir(path), 0o750); err != nil { + t.Fatalf("creating %s: %v", filepath.Dir(path), err) + } + + if err := os.WriteFile(path, []byte("#!/bin/sh\n"), 0o600); err != nil { + t.Fatalf("writing %s: %v", path, err) + } +} + +// setGOROOT points the resolution at a directory instead of the running +// toolchain's, and returns the restore. The value is asked for at every call, so +// nothing caches the real one. +func setGOROOT(t *testing.T, root string) func() { + t.Helper() + + previous := build.Default.GOROOT + build.Default.GOROOT = root + + return func() { build.Default.GOROOT = previous } +} From fd72e593c8f69afb2e6b5ff2c259fb6afa6f5355 Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 19:43:32 -0500 Subject: [PATCH 2/9] feat(scope): read which packages one test command executes Ditto scopes mutants to a change and judges them with one configured command, and those are two different questions: a staged change spanning five packages, judged by a command naming one of them, produces mutants nothing it runs can kill. They survive, and the score mixes 'your tests missed this' with 'your command cannot see this'. internal/commandscope answers the second question without parsing the command. `go test -json` names every package it executes -- including the ones with no test files, which it reports as a skip -- and `go list -deps -test` names every package those compile in. The closure is the load-bearing half: a mutant in a dependency of a named package IS killable by that package's tests, so a set comparison against the names alone would accuse a correct run. -test is what makes a test-only import count. A command that emits no stream -- make, gotestsum, a wrapper -- is not guessed at: unknown refuses nothing, because a missing accusation costs a number somebody chose an opaque command to get and a wrong one fails a run that was correct. One process per scope, measured at 0.28-0.31s over this repository's 64 executed packages, and nothing at all for a command ditto cannot read. Reading it needs the baseline's own output, and result.Output answered only for an Ok: a green suite is an Err, and a green suite is exactly the run whose stream carries the scope. Widened to the command's output whichever way it ended. --- internal/commandscope/commandscope.go | 293 ++++++++++++++++++ .../commandscope_internal_test.go | 154 +++++++++ internal/commandscope/commandscope_test.go | 192 ++++++++++++ internal/result/result.go | 32 +- internal/result/result_test.go | 27 ++ perf/baseline.json | 4 +- 6 files changed, 688 insertions(+), 14 deletions(-) create mode 100644 internal/commandscope/commandscope.go create mode 100644 internal/commandscope/commandscope_internal_test.go create mode 100644 internal/commandscope/commandscope_test.go create mode 100644 internal/result/result_test.go diff --git a/internal/commandscope/commandscope.go b/internal/commandscope/commandscope.go new file mode 100644 index 0000000..48b0cf3 --- /dev/null +++ b/internal/commandscope/commandscope.go @@ -0,0 +1,293 @@ +// Package commandscope answers which packages one test command executes. +// +// Ditto scopes mutants to a change and judges them with a single configured test +// command, and those are two different questions. A staged change that spans +// five packages, judged by a command that names one of them, produces mutants in +// packages the command never builds: nothing it runs can kill them, so they +// survive, and the score that comes back mixes "your tests missed this" with +// "your command cannot see this". Measured on the report that produced this +// package: 43 mutants, 23 killed, 20 survived, and 17 of those survivors in a +// package the command never executes — a run whose ceiling was 0.605, printed as +// 0.53 against a bar of 0.80. docs/reports/ditto-mutation-scope.md. +// +// Nothing here parses the command. Ditto asks the toolchain instead, twice: +// +// - `go test -json` names every package it executes, and that stream is +// already in hand: the baseline run every release makes once produces it. +// - `go list -deps -test` names every package those compile in, which is the +// side that decides whether a mutant could be killed at all. A mutant in a +// dependency of a named package IS killable by that package's tests, so +// naming the package is too narrow an answer; `-test` is what makes a +// test-only import count. +// +// A command that emits no such stream — make, gotestsum, a wrapper script — is +// not guessed at: its scope is unknown, and unknown refuses nothing. A missing +// accusation costs a number somebody configured an opaque command to get; a +// wrong one fails a run that was correct, which is the direction this repository +// does not fail in. +package commandscope + +import ( + "bytes" + "encoding/json" + "fmt" + "os" + "os/exec" + "path" + "runtime" + "slices" + "strings" + "sync" + + "github.com/Disble/ditto/internal/gotoolchain" +) + +// Runner runs one process, so the toolchain call can be pinned in a test. +type Runner interface { + Output(dir, name string, args ...string) ([]byte, error) +} + +// Scope is what one test command compiles, read from the stream it produced. +type Scope struct { + root string + executed []string + runner Runner + + mutex sync.Mutex + resolved bool + compiled map[string]bool +} + +// New reads the packages a `go test -json` stream named, and answers the +// toolchain's own question about the rest when it is asked. +// +// root is the directory the command ran in: the mutated tree, not the checkout +// it came from. Every answer is relative to it. +func New(output, root string, runner Runner) *Scope { + return &Scope{ + root: root, + executed: executedPackages(output), + runner: runner, + } +} + +// Executes reports whether the command compiles the package that owns a +// repository-relative source path. +// +// The answer is per package and not per file. A file the package does not +// compile on this platform — one behind a build tag for another operating system +// — is inside a package the command executes, and its mutants stay survivors. +// That is the boundary of this check: it separates "the command never builds +// this code" from "nothing ran this code", and only the first is a scope +// mismatch. +func (s *Scope) Executes(relativePath string) bool { + if len(s.executed) == 0 { + // The command told us nothing: no `-json`, no stream, or a command that + // is not `go test` at all. + return true + } + + s.mutex.Lock() + defer s.mutex.Unlock() + + if !s.resolved { + s.resolved = true + + s.compiled = s.compiledDirectories() + } + + if s.compiled == nil { + // The toolchain could not answer — no `go` binary, a module it cannot + // read. Unknown again, and unknown refuses nothing. + return true + } + + return s.compiled[normalizePath(path.Dir(relativePath), runtime.GOOS)] +} + +// compiledDirectories asks the toolchain which directories the executed packages +// compile into a test binary. +// +// One process per release, and only for a command that emitted a stream: +// measured at 0.28-0.31s over this repository's 64 executed packages, against a +// release that pays one suite run per mutant. It runs after the baseline, which +// has already built the same tree, so the module graph and the build cache are +// warm. +func (s *Scope) compiledDirectories() map[string]bool { + binary := gotoolchain.Path() + if binary == "" { + return nil + } + + arguments := append([]string{"list", "-deps", "-test", "-e", "-f", "{{.Dir}}"}, s.executed...) + + output, err := s.runner.Output(s.root, binary, arguments...) + if err != nil { + return nil + } + + if len(bytes.TrimSpace(output)) == 0 { + // A toolchain that answered nothing has not answered. An empty set would + // be read as "this command compiles nothing", which accuses every mutant + // in the scope. + return nil + } + + directories := make(map[string]bool) + + for line := range strings.SplitSeq(string(output), "\n") { + relative := relativeDirectory(s.root, strings.TrimSpace(line), runtime.GOOS) + if relative == "" { + continue + } + + directories[normalizePath(relative, runtime.GOOS)] = true + } + + return directories +} + +// event is the part of `go test -json` this reads. The rest is ignored on +// purpose: a field ditto does not read is a field that cannot make it wrong. +// Same shape and same reason as the struct in internal/verdict. +type event struct { + Package string `json:"Package"` //nolint:tagliatelle // go test -json emits these names + Action string `json:"Action"` //nolint:tagliatelle // go test -json emits these names +} + +// executedPackages reads the packages a stream named, in the order it named +// them. +// +// A stream names every package the command executes, and a package with no test +// files is one of them: measured on go1.27, `go test -json ./...` emits a start, +// an output line reading `[no test files]` and a skip for it. What it does not +// name is that package's dependencies, which never have a test binary of their +// own — which is why the closure below exists rather than a set comparison. +func executedPackages(output string) []string { + seen := make(map[string]bool) + names := []string{} + + for line := range strings.SplitSeq(output, "\n") { + line = strings.TrimSpace(line) + if !strings.HasPrefix(line, "{") { + // Compiler output and prose share the stream. A line that is not an + // event is not an error, and refusing the scope over one would throw + // away an answer a reader is about to act on. + continue + } + + var decoded event + if err := json.Unmarshal([]byte(line), &decoded); err != nil { + continue + } + + if decoded.Package == "" || decoded.Action == "" || seen[decoded.Package] { + continue + } + + seen[decoded.Package] = true + names = append(names, decoded.Package) + } + + return names +} + +// relativeDirectory turns one absolute directory from the toolchain into a +// repository-relative one, and answers empty for everything outside the root: +// the closure carries the standard library and every dependency, and none of +// those can hold a mutant. +func relativeDirectory(root, directory, goos string) string { + if directory == "" { + return "" + } + + normalizedRoot := normalizePath(root, goos) + normalizedDirectory := normalizePath(directory, goos) + + if normalizedDirectory == normalizedRoot { + return "." + } + + // The separator is part of the prefix, so /treehouse is not under /tree. + prefix := normalizedRoot + "/" + if !strings.HasPrefix(normalizedDirectory, prefix) { + return "" + } + + return normalizedDirectory[len(prefix):] +} + +// normalizePath is the form two directories are compared in, and the identity +// one package is filed under. +// +// Two things are folded here and both of them have produced a defect in this +// repository already. Separators: a path from the toolchain and a path ditto +// built are the same directory spelled two ways. Case: Windows compares paths +// case-insensitively, so the fold is not optional there — it is the difference +// between clearing a mutant and wrongfully accusing it, and a wrong accusation +// fails a run that was correct. +// +// The system is a parameter rather than a read of runtime.GOOS, the way the +// target OS is in the batch planner's binary name: the folding half only exists +// on Windows, CI runs on Linux, and a test that passes "windows" is the only way +// this host covers it. +func normalizePath(value, goos string) string { + slashed := path.Clean(strings.ReplaceAll(value, `\`, "/")) + if goos == windows { + return strings.ToLower(slashed) + } + + return slashed +} + +// windows is the one GOOS whose paths fold case, named so the fold can be +// exercised where it does not run. +const windows = "windows" + +// gitEnvironment is what git exports to a hook, and in a linked worktree it +// exports them as absolute paths. Everything spawned below inherits them, so a +// command meant for the sandbox addresses the hook's repository instead — and +// then succeeds, which is the whole problem. +// +// Measured twice in this repository before the list existed: once leaving a +// stray commit on a live branch, once writing core.bare and an identity into a +// shared config. The toolchain does not read git here, and this list is not an +// argument that it might: it is the rule every process ditto starts follows. +var gitEnvironment = []string{ //nolint:gochecknoglobals // one fixed list, read only + "GIT_DIR", + "GIT_INDEX_FILE", + "GIT_WORK_TREE", + "GIT_OBJECT_DIRECTORY", + "GIT_COMMON_DIR", +} + +// OSRunner runs processes with the inherited git addressing removed. +type OSRunner struct{} + +func (OSRunner) Output(dir, name string, args ...string) ([]byte, error) { + command := exec.Command(name, args...) //nolint:noctx // there is no cancellation contract here: the query is bounded by go list's own work + command.Dir = dir + command.Env = withoutGitEnvironment(os.Environ()) + + output, err := command.CombinedOutput() + if err != nil { + return nil, fmt.Errorf("%s %s: %w: %s", name, strings.Join(args, " "), err, bytes.TrimSpace(output)) + } + + return output, nil +} + +func withoutGitEnvironment(environment []string) []string { + kept := make([]string, 0, len(environment)) + + for _, variable := range environment { + name, _, _ := strings.Cut(variable, "=") + if slices.Contains(gitEnvironment, name) { + continue + } + + kept = append(kept, variable) + } + + return kept +} diff --git a/internal/commandscope/commandscope_internal_test.go b/internal/commandscope/commandscope_internal_test.go new file mode 100644 index 0000000..fef73f3 --- /dev/null +++ b/internal/commandscope/commandscope_internal_test.go @@ -0,0 +1,154 @@ +package commandscope + +import ( + "reflect" + "testing" +) + +// A stream names every package the command executes, and a package with no test +// files is one of them: go test reports it as a skip rather than staying silent. +// Measured on go1.27 — this is the case that decides whether a package nobody +// wrote tests for is read as executed or as absent. +func TestExecutedPackagesReadsAPackageWithNoTestFiles(t *testing.T) { + t.Parallel() + + stream := `{"Action":"start","Package":"probe/internal/noTests"} +{"Action":"output","Package":"probe/internal/noTests","Output":"? \tprobe/internal/noTests\t[no test files]\n"} +{"Action":"skip","Package":"probe/internal/noTests","Elapsed":0} +{"Action":"start","Package":"probe/internal/withTests"} +{"Action":"run","Package":"probe/internal/withTests","Test":"TestSum"} +{"Action":"pass","Package":"probe/internal/withTests","Elapsed":0.3} +` + + got := executedPackages(stream) + want := []string{"probe/internal/noTests", "probe/internal/withTests"} + + if !reflect.DeepEqual(got, want) { + t.Fatalf("executedPackages = %v, want %v", got, want) + } +} + +// Compiler output, prose and mangled JSON share the stream, and none of them is +// an error: the scope is an answer a reader acts on, and refusing it over one +// unreadable line would trade it for nothing. +func TestExecutedPackagesIgnoresWhatIsNotAnEvent(t *testing.T) { + t.Parallel() + + stream := "? \tprobe/internal/x\t[no test files]\n" + + "# probe/internal/x\n" + + "x.go:3:2: undefined: Missing\n" + + "{\"Action\":\"start\"}\n" + + "{\"Package\":\"probe/internal/x\"}\n" + + "{not json}\n" + + "{\"Action\":\"start\",\"Package\":\"probe/internal/x\"}\n" + + got := executedPackages(stream) + want := []string{"probe/internal/x"} + + if !reflect.DeepEqual(got, want) { + t.Fatalf("executedPackages = %v, want %v", got, want) + } +} + +// A command that is not `go test -json` prints none of this, and the answer has +// to be "nothing was named" rather than a guess: make, gotestsum and a wrapper +// script all read as unknown here. +func TestExecutedPackagesIsEmptyForACommandThatNamedNothing(t *testing.T) { + t.Parallel() + + for _, stream := range []string{"", "ok \tprobe/internal/x\t0.2s\n", "BUILD FAILED\n"} { + if got := executedPackages(stream); len(got) != 0 { + t.Fatalf("executedPackages(%q) = %v, want nothing", stream, got) + } + } +} + +func TestRelativeDirectoryRefusesWhatIsOutsideTheRoot(t *testing.T) { + t.Parallel() + + cases := []struct { + name string + root string + directory string + want string + }{ + {name: "inside", root: "/tree", directory: "/tree/internal/desktop", want: "internal/desktop"}, + {name: "the root itself", root: "/tree", directory: "/tree", want: "."}, + {name: "a sibling with the same prefix", root: "/tree", directory: "/treehouse/x", want: ""}, + {name: "elsewhere", root: "/tree", directory: "/other/x", want: ""}, + {name: "the standard library", root: "/tree", directory: "/usr/local/go/src/fmt", want: ""}, + {name: "no directory at all", root: "/tree", directory: "", want: ""}, + } + + for _, testcase := range cases { + t.Run(testcase.name, func(t *testing.T) { + t.Parallel() + + if got := relativeDirectory(testcase.root, testcase.directory, "linux"); got != testcase.want { + t.Fatalf("relativeDirectory(%q, %q) = %q, want %q", + testcase.root, testcase.directory, got, testcase.want) + } + }) + } +} + +// The case-fold is the half CI cannot run, so it is exercised here with the +// target named. Windows spells one directory two ways and they are one package; +// reporting them as two fails to clear a mutant, which is a wrong accusation +// rather than a missing one. +func TestNormalizePathFoldsOnlyOnWindows(t *testing.T) { + t.Parallel() + + cases := []struct { + name string + goos string + path string + want string + }{ + {name: "windows folds case", goos: "windows", path: `Internal\Desktop`, want: "internal/desktop"}, + {name: "windows folds the drive letter", goos: "windows", path: `C:\Tree\Pkg`, want: "c:/tree/pkg"}, + {name: "windows drops a trailing separator", goos: "windows", path: `C:\Tree\Pkg\`, want: "c:/tree/pkg"}, + {name: "linux folds nothing", goos: "linux", path: "Internal/Desktop", want: "Internal/Desktop"}, + } + + for _, testcase := range cases { + t.Run(testcase.name, func(t *testing.T) { + t.Parallel() + + if got := normalizePath(testcase.path, testcase.goos); got != testcase.want { + t.Fatalf("normalizePath(%q, %q) = %q, want %q", testcase.path, testcase.goos, got, testcase.want) + } + }) + } +} + +// And the two spellings of one Windows directory have to reach the same key, or +// the fold is decorative: that is the failure this test exists for. +func TestTheTwoSpellingsOfOneDirectoryAgree(t *testing.T) { + t.Parallel() + + root := `C:\Users\Dev\Temp\ditto-1` + fromToolchain := normalizePath(relativeDirectory(root, root+`\internal\desktop`, "windows"), "windows") + fromDitto := normalizePath("internal/desktop", "windows") + + if fromToolchain != fromDitto { + t.Fatalf("one directory produced two keys: %q and %q", fromToolchain, fromDitto) + } +} + +// The inherited git addressing is removed from every process ditto starts, and +// a list is not a guard: this is the half that would notice the list being +// dropped from the command above. +func TestWithoutGitEnvironmentKeepsEverythingElse(t *testing.T) { + t.Parallel() + + kept := withoutGitEnvironment([]string{ + "PATH=/bin", "GIT_DIR=/somewhere/.git", "HOME=/home/x", + "GIT_INDEX_FILE=/i", "GIT_WORK_TREE=/w", "GIT_OBJECT_DIRECTORY=/o", "GIT_COMMON_DIR=/c", + }) + want := []string{"PATH=/bin", "HOME=/home/x"} + + if !reflect.DeepEqual(kept, want) { + t.Fatalf("withoutGitEnvironment = %v, want %v", kept, want) + } +} diff --git a/internal/commandscope/commandscope_test.go b/internal/commandscope/commandscope_test.go new file mode 100644 index 0000000..2a5611c --- /dev/null +++ b/internal/commandscope/commandscope_test.go @@ -0,0 +1,192 @@ +package commandscope_test + +import ( + "errors" + "path/filepath" + "slices" + "strings" + "testing" + + "github.com/Disble/ditto/internal/commandscope" +) + +// scriptedToolchain is the toolchain's answer, pinned. It also counts how often +// it was asked, because "one process per release" is part of what the scope +// costs and a test is the only place that can hold it. +type scriptedToolchain struct { + output []byte + err error + asked int + lastRun []string +} + +func (s *scriptedToolchain) Output(_, name string, args ...string) ([]byte, error) { + s.asked++ + s.lastRun = append([]string{name}, args...) + + return s.output, s.err +} + +// streamOf is the smallest `go test -json` stream that names packages: the +// scope reads the names, and nothing else in an event is its business. +func streamOf(packages ...string) string { + lines := make([]string, 0, len(packages)) + for _, name := range packages { + lines = append(lines, `{"Action":"start","Package":"`+name+`"}`) + } + + return strings.Join(lines, "\n") + "\n" +} + +// TestExecutesReadsTheClosureAndNotJustTheNames is the case that decides whether +// the check is usable at all. +// +// A mutant in a dependency of a package the command names IS killable by that +// package's tests — the dependency is compiled into its test binary — so a set +// comparison against the names would accuse a run that was correct. Measured on +// the report's own change: the command named syncdiag and the change also touched +// packages it imports. +func TestExecutesReadsTheClosureAndNotJustTheNames(t *testing.T) { + t.Parallel() + + root := t.TempDir() + + toolchain := &scriptedToolchain{output: []byte(strings.Join([]string{ + filepath.Join(root, "internal", "observability", "readcap"), + filepath.Join(root, "internal", "observability", "syncdiag"), + filepath.Join(root, "testdata", "ignored"), + "/usr/local/go/src/fmt", + "", + }, "\n"))} + + scope := commandscope.New(streamOf("example.com/mod/internal/observability/syncdiag"), root, toolchain) + + named := "internal/observability/syncdiag/reader.go" + if !scope.Executes(named) { + t.Fatalf("the package the command names is not reported as executed: %q", named) + } + + inClosure := "internal/observability/readcap/reader.go" + if !scope.Executes(inClosure) { + t.Fatalf("a package compiled into the named package's test binary is not reported as executed: %q", inClosure) + } + + // The report's own case: staged, mutated, and never built by the command. + unreachable := "internal/desktop/app_runtime_services.go" + if scope.Executes(unreachable) { + t.Fatalf("a package the command does not compile is reported as executed: %q", unreachable) + } + + // Something the closure named that is not in this tree cannot be a mutant, + // and the answer for one is not what clears the answer for the rest. + if scope.Executes("fmt/print.go") { + t.Fatal("a directory outside the tree was read as executed") + } +} + +// TestExecutesAsksOncePerScope holds the cost: one process per release, however +// many files ask. It is the difference between a check that is free in the +// common case and one that pays 0.28s per file. +func TestExecutesAsksOncePerScope(t *testing.T) { + t.Parallel() + + root := t.TempDir() + toolchain := &scriptedToolchain{output: []byte(filepath.Join(root, "calc"))} + + scope := commandscope.New(streamOf("example.com/mod/calc"), root, toolchain) + + for i := range 5 { + scope.Executes("calc/calc.go") + scope.Executes("other/other.go") + + if toolchain.asked != 1 { + t.Fatalf("the toolchain was asked %d times after %d rounds; one process per release is the cost", toolchain.asked, i+1) + } + } + + // And the query has to be the closure query. -deps is what reaches a + // dependency, and -test is what reaches one that only a test file imports — + // the case module_scope.go already measured for the gated path. + for _, flag := range []string{"-deps", "-test"} { + if !slices.Contains(toolchain.lastRun, flag) { + t.Fatalf("the toolchain was asked without %s: %v", flag, toolchain.lastRun) + } + } +} + +// TestExecutesRefusesNothingWhenTheCommandToldItNothing is the fail-open rule, +// and it is the answer for every command that is not `go test -json`: make, +// gotestsum, a wrapper. A missing accusation leaves a number somebody chose an +// opaque command to get; a wrong one fails a run that was correct. +func TestExecutesRefusesNothingWhenTheCommandToldItNothing(t *testing.T) { + t.Parallel() + + toolchain := &scriptedToolchain{output: []byte(filepath.Join(t.TempDir(), "calc"))} + + for _, stream := range []string{"", "ok \texample.com/mod/calc\t0.2s\n", "make: Nothing to be done.\n"} { + scope := commandscope.New(stream, t.TempDir(), toolchain) + + if !scope.Executes("internal/desktop/app.go") { + t.Fatalf("a command that named no package refused one: %q", stream) + } + } + + if toolchain.asked != 0 { + t.Fatalf("the toolchain was asked %d times for commands ditto cannot read; the check costs nothing there", toolchain.asked) + } +} + +// TestExecutesRefusesNothingWhenTheToolchainCannotAnswer is the other half of +// fail-open: a module the toolchain cannot read, no `go` binary, a query that +// fails. Unknown is not evidence of a mismatch. +func TestExecutesRefusesNothingWhenTheToolchainCannotAnswer(t *testing.T) { + t.Parallel() + + cases := []struct { + name string + toolchain *scriptedToolchain + }{ + {name: "the query failed", toolchain: &scriptedToolchain{err: errors.New("no go binary")}}, + {name: "the query answered nothing", toolchain: &scriptedToolchain{output: []byte(" \n")}}, + } + + for _, testcase := range cases { + t.Run(testcase.name, func(t *testing.T) { + t.Parallel() + + scope := commandscope.New(streamOf("example.com/mod/calc"), t.TempDir(), testcase.toolchain) + + if !scope.Executes("internal/desktop/app.go") { + t.Fatal("a scope that could not be resolved refused a mutant") + } + }) + } +} + +// The inherited git addressing is removed from every process ditto starts, and +// this is the guard for it: cmdtestrunner has the version that measures the +// damage, and the incident behind both is in AGENTS.md — a stray commit on a +// live branch, from a command that was meant for a temporary directory. +func TestOSRunnerStripsTheInheritedGitAddressing(t *testing.T) { + decoy := t.TempDir() + + for variable, value := range map[string]string{ + "GIT_DIR": filepath.Join(decoy, ".git"), + "GIT_INDEX_FILE": filepath.Join(decoy, ".git", "index"), + "GIT_WORK_TREE": decoy, + "GIT_OBJECT_DIRECTORY": filepath.Join(decoy, "objects"), + "GIT_COMMON_DIR": filepath.Join(decoy, "common"), + } { + t.Setenv(variable, value) + } + + output, err := commandscope.OSRunner{}.Output(t.TempDir(), "sh", "-c", + `printf '%s|%s|%s|%s|%s' "$GIT_DIR" "$GIT_INDEX_FILE" "$GIT_WORK_TREE" "$GIT_OBJECT_DIRECTORY" "$GIT_COMMON_DIR"`) + if err != nil { + t.Fatalf("running the probe: %v", err) + } + + if got := strings.TrimSpace(string(output)); got != "||||" { + t.Fatalf("the subprocess inherited git's addressing: %q", got) + } +} diff --git a/internal/result/result.go b/internal/result/result.go index 5499575..6c1a79d 100644 --- a/internal/result/result.go +++ b/internal/result/result.go @@ -14,20 +14,28 @@ func Err[Type any](errorMessage string) Result[Type] { return err[Type]{errorMessage} } -// Output is the value an Ok carries, and empty for an Err. +// Output is what the command printed, whichever way it ended. // -// It exists because a refusal has to be able to show what it refused over. The -// laboratory reads a failing test command as a killed mutant, so when nothing is -// mutated that same failure is a red baseline — and a refusal that does not -// print the command's own output leaves the reader guessing which of a hundred -// reasons a suite might be red. Measured the hard way: an embedded directory -// missing from a sandbox produced a refusal that named neither the file nor the -// pattern, and finding it took four measurements that should have been none. +// It answered only for an Ok until the command's own package scope had to be +// read: that scope is announced by `go test -json` on the stream of a GREEN +// suite, and a green suite is an Err — so the one run whose output the scope +// check needs was the one this refused to answer for. The name and the contract +// now say the same thing: the output belongs to the command, not to the verdict. +// +// What the Ok side is for is unchanged, and it was measured the hard way: the +// laboratory reads a failing command as a killed mutant, so a refusal to score a +// red baseline has to show the command's own words — the first version named +// neither the file nor the pattern, and finding the embedded directory behind it +// took four measurements that should have been none. func Output(res Result[string]) string { - value, isOk := res.(ok[string]) - if !isOk { - return "" + if ended, isOk := res.(ok[string]); isOk { + return ended.value + } + + if ended, isErr := res.(err[string]); isErr { + return ended.errorMessage } - return value.value + // Unreachable: the seal keeps every implementation inside this package. + return "" } diff --git a/internal/result/result_test.go b/internal/result/result_test.go new file mode 100644 index 0000000..ac6bf70 --- /dev/null +++ b/internal/result/result_test.go @@ -0,0 +1,27 @@ +package result_test + +import ( + "testing" + + "github.com/Disble/ditto/internal/result" +) + +// TestOutputAnswersForBothEndings pins the widened contract, and it is a test +// rather than a sentence because the narrower one was load-bearing: the package +// scope a release reads is announced by a GREEN suite, and a green suite is an +// Err. Answering empty there is what made the scope unreadable. +func TestOutputAnswersForBothEndings(t *testing.T) { + t.Parallel() + + // Ok is a command that failed — a killed mutant — and its output is what a + // refusal prints. + if got := result.Output(result.Ok("boom")); got != "boom" { + t.Fatalf("Output(Ok) = %q, want the command's own words", got) + } + + // Err is a command that succeeded, and its stream is the only place the + // package scope of a passing suite exists. + if got := result.Output(result.Err[string]("{\"Action\":\"skip\"}")); got != "{\"Action\":\"skip\"}" { + t.Fatalf("Output(Err) = %q, want the stream a passing suite produced", got) + } +} diff --git a/perf/baseline.json b/perf/baseline.json index 7382943..a62b051 100644 --- a/perf/baseline.json +++ b/perf/baseline.json @@ -17,7 +17,7 @@ "laboratoryRunsForOneChangedFunction": 4, "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": 8, "testCommandInvocationsPerReleaseWholeFixture": 49, - "mutantsPerReleaseOnThisRepository": 873 + "mutantsPerReleaseOnThisRepository": 904 }, "targets": { "sourceParsesPerReleaseWithThreeViruses": "Reached: 4, one parse per source file, down from 12. GoSourceFile.Incubate now takes the whole mutator set and parses once for all of them. With the default 14 mutators this is 14 parses per file reduced to 1.", @@ -28,6 +28,6 @@ "laboratoryRunsForOneChangedFunction": "4, the mutators that fire on one changed line and nothing else in the repository. This is what WithChangedRanges buys: without it the same fixture charges 48.", "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": "8, exactly twice the single-file number. The two ranges name offsets that exist in both files, because every fixture file has the same byte layout, so a scope holding one flat set of ranges would charge 16 and grow as the square of the file count. Keeping the ranges beside their file makes that impossible rather than merely unlikely.", "testCommandInvocationsPerReleaseWholeFixture": "49 for the 48 mutants of the whole fixture, plus one. That one is the baseline: the laboratory runs the suite once on unmutated code before scoring anything, because a test command that fails before it compiles fails for every mutant too, and ditto recognises a killed mutant by exactly that. Measured on ditto's own gate before the guard existed: 431 of 431 killed in 5.46 seconds, a perfect score for a run that compiled nothing. Every other laboratory counter here goes through a stand-in and cannot see a run the laboratory makes on its own, which is why this one exists — a cost nobody records is one that grows unnoticed, the mirror of the unrecorded gain this file already refuses. It must not grow: one baseline per release, never one per mutant.", - "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022." + "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md." } } From 2603fde397513db665a9e2cb2b0f73d179d6a0d0 Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 19:46:54 -0500 Subject: [PATCH 3/9] feat(laboratory): read the command's package scope from its own baseline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The laboratory is the only place that can learn what the configured test command executes without paying for it: the baseline run it already makes once per release prints the `go test -json` stream that names every package the command executes, and the sandbox it ran in is the tree the rest is resolved against. So Executes(repository, path) answers whether the command compiles the package that owns a file, and the first call is what pays for the baseline on a release that never asks otherwise. There is no second cost: the same sandbox, the same sync.Once. Held by count, not by argument — one baseline, one toolchain query, however many files ask. It answers true when the question cannot be answered: a command that is not `go test -json` has no readable package scope, and neither does one whose toolchain cannot be asked. A missing accusation leaves a number somebody configured an opaque command to get; a wrong one fails a run that was correct. --- internal/laboratory/laboratory.go | 70 +++++++++++++++++ internal/laboratory/laboratory_test.go | 100 +++++++++++++++++++++++++ perf/baseline.json | 4 +- 3 files changed, 172 insertions(+), 2 deletions(-) diff --git a/internal/laboratory/laboratory.go b/internal/laboratory/laboratory.go index 28e6fae..9c7e5fe 100644 --- a/internal/laboratory/laboratory.go +++ b/internal/laboratory/laboratory.go @@ -5,6 +5,7 @@ import ( "time" "github.com/Disble/ditto/internal/color" + "github.com/Disble/ditto/internal/commandscope" "github.com/Disble/ditto/internal/ditto" "github.com/Disble/ditto/internal/future" "github.com/Disble/ditto/internal/gomutatedfile" @@ -32,6 +33,12 @@ type TemporaryDirectory interface { // which is what ditto is built for — only ever builds one. The number alive at // any instant is therefore the same as when each mutant built its own, which // matters because that number is also what an interrupted run leaves behind. +// +// It is also the only place that can learn what the configured test command +// executes, without paying for it: the baseline run it already makes once per +// release prints the `go test -json` stream that names every package the command +// executes, and the sandbox it ran in is the tree to resolve the rest against. +// See Executes and internal/commandscope. type Laboratory struct { logger ditto.Logger testRunner TestRunner @@ -41,6 +48,24 @@ type Laboratory struct { idle []ditto.TemporaryRepository baseline sync.Once + scope *commandscope.Scope +} + +// scopeRunner is the process seam under the one `go list` a scope costs. +// +// A package-level variable with a Set for tests rather than a constructor +// parameter, because it is not something a caller configures: it is how ditto +// asks the toolchain which packages the command compiles, and the only thing +// that ever varies is a test's answer to it. +var scopeRunner commandscope.Runner = commandscope.OSRunner{} //nolint:gochecknoglobals // one seam, with the restore below + +// SetScopeRunnerForTest replaces the process the scope resolution would start, +// and returns the restore. +func SetScopeRunnerForTest(runner commandscope.Runner) func() { + previous := scopeRunner + scopeRunner = runner + + return func() { scopeRunner = previous } } func New(logger ditto.Logger, testRunner TestRunner, temporaryDirectory TemporaryDirectory) *Laboratory { @@ -65,6 +90,38 @@ func (l *Laboratory) Test( return future.Resolved(l.testRunner.Test(sandbox)) } +// Executes reports whether the configured test command can execute the package +// that owns a repository-relative source path. +// +// This is the one question the report cannot answer for itself. Mutants come +// from files the scope selected and are judged by one command, and when that +// command does not compile a mutant's package the mutant is not badly tested, it +// is unmeasured: nothing the command runs can kill it, so it survives, and a +// score counting it mixes 'your tests missed this' with 'your command cannot see +// this'. Measured on the report that asked for this: 43 mutants, 20 survivors, +// 17 of them in one package the command never builds, +// docs/reports/ditto-mutation-scope.md. +// +// It answers true when the question cannot be answered — a command that is not +// `go test -json` has no readable package scope, and so does one whose toolchain +// cannot be asked — because a wrong accusation fails a run that was correct, +// while a missing one only leaves the caller with what it had before. +// +// The first call is what pays for the baseline on a release that never asks +// otherwise. There is no second cost: the same sandbox and the same once. +func (l *Laboratory) Executes(repository ditto.Repository, relativePath string) bool { + sandbox := l.acquire(repository) + defer l.returnToPool(sandbox) + + l.verifyBaseline(sandbox) + + if l.scope == nil { + return true + } + + return l.scope.Executes(relativePath) +} + // verifyBaseline runs the suite once, on unmutated code, before any mutant is // scored. // @@ -100,6 +157,13 @@ func (l *Laboratory) verifyBaseline(sandbox ditto.TemporaryRepository) { "killed; refusing to score against a red baseline\n\n" + verdict.Text(result.Output(res)))) } + // What a GREEN suite said about itself, which is the part nothing else + // reads. `go test -json` names every package the command executes, and + // this is the one run that produces that stream for free: the baseline + // was already paid for, once per release. A command that emits no stream + // leaves the scope unknown, and unknown refuses nothing. + l.scope = commandscope.New(result.Output(res), sandbox.Root(), scopeRunner) + // This run was already paid for and the clock above already measured it; // throwing the number away was the waste. It is the per-mutant price of // the test command, and printed beside the mutant count the release @@ -137,6 +201,12 @@ func (l *Laboratory) acquire(repository ditto.Repository) ditto.TemporaryReposit func (l *Laboratory) hand(sandbox ditto.TemporaryRepository, file *gomutatedfile.GoMutatedFile) { file.RestoreIn(sandbox) + l.returnToPool(sandbox) +} + +// returnToPool hands a clean sandbox back for the next mutant. It is hand +// without the restore, for a caller that never wrote anything into it. +func (l *Laboratory) returnToPool(sandbox ditto.TemporaryRepository) { l.mutex.Lock() defer l.mutex.Unlock() diff --git a/internal/laboratory/laboratory_test.go b/internal/laboratory/laboratory_test.go index 3b0bfb7..8234ecf 100644 --- a/internal/laboratory/laboratory_test.go +++ b/internal/laboratory/laboratory_test.go @@ -1,6 +1,8 @@ package laboratory_test import ( + "errors" + "path/filepath" "testing" "github.com/Disble/ditto/internal/ditto" @@ -133,6 +135,104 @@ func TestLaboratoryChecksTheBaselineOnce(t *testing.T) { assert.Equal(t, 4, runner.calls, "want one baseline and three mutants") } +// streamingRunner is a command that passes and prints what `go test -json` +// prints. Its stream is the only place the package scope of a release exists. +type streamingRunner struct { + stream string + calls int +} + +func (r *streamingRunner) Test(ditto.TemporaryRepository) result.Result[string] { + r.calls++ + + return result.Err[string](r.stream) +} + +// scriptedToolchain is the toolchain's answer to "what does that compile", and +// it counts how often it was asked. +type scriptedToolchain struct { + output []byte + err error + asked int +} + +func (s *scriptedToolchain) Output(_, _ string, _ ...string) ([]byte, error) { + s.asked++ + + return s.output, s.err +} + +// A release judges mutants from files the scope selected with one configured +// command, and when that command does not compile a mutant's package the mutant +// is unmeasured rather than badly tested: nothing the command runs can kill it. +// +// The scope comes free. The baseline run the laboratory already makes once per +// release prints the stream that names the packages the command executes, and +// the sandbox it ran in is the tree to resolve the rest against — so the answer +// costs one `go list` and no extra suite run, which the counts below hold. +func TestLaboratoryKnowsWhichPackagesTheCommandExecutes(t *testing.T) { + root := "tmpdir-1" + + runner := &streamingRunner{stream: `{"Action":"start","Package":"example.com/mod/internal/observability/syncdiag"}` + "\n"} + + toolchain := &scriptedToolchain{output: []byte(filepath.Join(root, "internal", "observability", "readcap") + "\n" + + filepath.Join(root, "internal", "observability", "syncdiag") + "\n")} + + defer laboratory.SetScopeRunnerForTest(toolchain)() + + subject := laboratory.New(fakelogger.New(), runner, faketempdirectory.NewFakeTemporaryDirectory("tmpdir")) + repository := fakerepository.New(fakerepository.FS{}, fakerepository.NewTemporary()) + + assert.True(t, subject.Executes(repository, "internal/observability/syncdiag/reader.go"), + "the package the command names is not reported as executed") + + // The closure is the load-bearing half: a package the command does not name + // but does compile into a test binary IS killable by that binary's tests, so + // a set comparison against the names alone would accuse a correct run. + assert.True(t, subject.Executes(repository, "internal/observability/readcap/reader.go"), + "a package compiled into the named package's tests is not reported as executed") + + // The report's own case, and the whole point: staged, mutated, never built. + assert.False(t, subject.Executes(repository, "internal/desktop/app_runtime_services.go"), + "a package the command does not compile is reported as executed") + + assert.Equal(t, 1, toolchain.asked, "want one toolchain query per scope") + assert.Equal(t, 1, runner.calls, "want one baseline, not one per question asked of the scope") +} + +// A command that is not `go test -json` has no readable scope — make, gotestsum, +// a wrapper — and ditto refuses nothing over it. Two costs are held here: the +// scope is unknown rather than guessed, and the toolchain is never asked, because +// there is nothing to ask. +func TestLaboratoryRefusesNothingForACommandWithNoReadableScope(t *testing.T) { + runner := &streamingRunner{stream: "ok \texample.com/mod/calc\t0.2s\n"} + toolchain := &scriptedToolchain{output: []byte("anything")} + + defer laboratory.SetScopeRunnerForTest(toolchain)() + + subject := laboratory.New(fakelogger.New(), runner, faketempdirectory.NewFakeTemporaryDirectory("tmpdir")) + repository := fakerepository.New(fakerepository.FS{}, fakerepository.NewTemporary()) + + assert.True(t, subject.Executes(repository, "internal/desktop/app.go"), + "an unreadable command refused a mutant") + assert.Zero(t, toolchain.asked, "the toolchain was asked about a command ditto could not read") +} + +// And the other half of that rule: a scope that could not be resolved is not +// evidence of a mismatch either. +func TestLaboratoryRefusesNothingWhenTheToolchainCannotAnswer(t *testing.T) { + runner := &streamingRunner{stream: `{"Action":"start","Package":"example.com/mod/calc"}` + "\n"} + toolchain := &scriptedToolchain{err: errors.New("no go binary")} + + defer laboratory.SetScopeRunnerForTest(toolchain)() + + subject := laboratory.New(fakelogger.New(), runner, faketempdirectory.NewFakeTemporaryDirectory("tmpdir")) + repository := fakerepository.New(fakerepository.FS{}, fakerepository.NewTemporary()) + + assert.True(t, subject.Executes(repository, "internal/desktop/app.go"), + "a scope that could not be resolved refused a mutant") +} + func TestLaboratory(t *testing.T) { source := dittotesting.Source(` |package source diff --git a/perf/baseline.json b/perf/baseline.json index a62b051..2719541 100644 --- a/perf/baseline.json +++ b/perf/baseline.json @@ -17,7 +17,7 @@ "laboratoryRunsForOneChangedFunction": 4, "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": 8, "testCommandInvocationsPerReleaseWholeFixture": 49, - "mutantsPerReleaseOnThisRepository": 904 + "mutantsPerReleaseOnThisRepository": 905 }, "targets": { "sourceParsesPerReleaseWithThreeViruses": "Reached: 4, one parse per source file, down from 12. GoSourceFile.Incubate now takes the whole mutator set and parses once for all of them. With the default 14 mutators this is 14 parses per file reduced to 1.", @@ -28,6 +28,6 @@ "laboratoryRunsForOneChangedFunction": "4, the mutators that fire on one changed line and nothing else in the repository. This is what WithChangedRanges buys: without it the same fixture charges 48.", "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": "8, exactly twice the single-file number. The two ranges name offsets that exist in both files, because every fixture file has the same byte layout, so a scope holding one flat set of ranges would charge 16 and grow as the square of the file count. Keeping the ranges beside their file makes that impossible rather than merely unlikely.", "testCommandInvocationsPerReleaseWholeFixture": "49 for the 48 mutants of the whole fixture, plus one. That one is the baseline: the laboratory runs the suite once on unmutated code before scoring anything, because a test command that fails before it compiles fails for every mutant too, and ditto recognises a killed mutant by exactly that. Measured on ditto's own gate before the guard existed: 431 of 431 killed in 5.46 seconds, a perfect score for a run that compiled nothing. Every other laboratory counter here goes through a stand-in and cannot see a run the laboratory makes on its own, which is why this one exists — a cost nobody records is one that grows unnoticed, the mirror of the unrecorded gain this file already refuses. It must not grow: one baseline per release, never one per mutant.", - "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md." + "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once." } } From 11813765203b4f4d2b55af78776886a214488bc5 Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 20:09:10 -0500 Subject: [PATCH 4/9] feat(report): leave out, name and fail on the mutants the command cannot compile A mutant the test command never compiles is not badly tested, it is unmeasured: nothing the command runs can kill it, so it survives, and a score counting it mixes 'your tests missed this' with 'your command cannot see this'. Measured on the report that asked for this: 43 mutants, 23 killed, 20 survived, 17 of those survivors in one package the command never builds, printed as 0.53 against a bar of 0.80 over a run whose ceiling was 0.605. docs/reports/ditto-mutation-scope.md. So the report separates the class the field already separates, exactly as it does for a mutant that never compiled (Zhu/Hall/May, S = D/(M - E)): out of the numerator and the denominator, named with the packages that hold them, and the run does not pass whatever the ratio over the rest says. The test command is not started for them either -- a guaranteed survivor bought with a full suite run is not evidence about anybody's tests. | Total: 3 Killed: 2 Survived: 1 Unmeasured: 1 | 1 of the 4 mutants in this scope are never compiled by this test command, so | nothing it runs can kill them and they are out of the score entirely: | internal/untested (1) exit 3 The exit code is the half a wrapper needs: a scope ditto cannot measure and a score below the bar were both 1, and the responses differ -- the second is answered by testing more, the first by naming every package the scope mutates in --test-command or by narrowing the scope, and by no test at all. Two decorators forward the new count, the test double applies the same exclusion, reporter's summary is one pass instead of three, and --test-command's help now names the lever that exists: it said 'name the package that owns the change', which is true only while the scope holds one. The golden moved by one line and it is the only line it moved: asking the laboratory which packages the command can execute is what pays for the baseline, so the baseline announcement now prints before the first file's announcement rather than between it and the first mutant. No verdict moved and every address is identical. --- cmd/ditto/main.go | 58 +++++- cmd/ditto/main_test.go | 45 ++++- internal/consolereporter/consolereporter.go | 181 ++++++++++++++---- internal/consolereporter/unmeasured_test.go | 87 +++++++++ internal/ditto/ditto.go | 94 ++++++++- .../dittotesting/fakereporter/counting.go | 16 +- .../dittotesting/fakereporter/fakereporter.go | 29 ++- internal/dittotesting/fakereporter/summary.go | 5 + internal/gatedreporter/gatedreporter.go | 12 ++ internal/gatedreporter/unmeasured_test.go | 25 +++ internal/verbosereporter/unmeasured_test.go | 25 +++ internal/verbosereporter/verbosereporter.go | 12 ++ perf/baseline.json | 4 +- release.go | 82 +++++++- release_golden_test.go | 8 + testdata/golden/release.txt | 2 +- 16 files changed, 616 insertions(+), 69 deletions(-) create mode 100644 internal/consolereporter/unmeasured_test.go create mode 100644 internal/gatedreporter/unmeasured_test.go create mode 100644 internal/verbosereporter/unmeasured_test.go diff --git a/cmd/ditto/main.go b/cmd/ditto/main.go index 919cba0..a319780 100644 --- a/cmd/ditto/main.go +++ b/cmd/ditto/main.go @@ -27,11 +27,40 @@ import ( // A refusal and a usage error are told apart by their message, not by this. const exitFailure = 1 +// exitUnmeasured is what a run whose scope holds mutants the configured test +// command cannot execute returns. +// +// It is its own code because the response is different, and a wrapper has to be +// able to tell it apart without reading prose: a score below the bar is answered +// by testing more or by agreeing to less, while this one is answered by naming +// every package the scope mutates in --test-command, or by narrowing the scope — +// and by no test at all. Measured on the report that asked for it, where 17 of 20 +// survivors were in a package the command never builds. +// docs/reports/ditto-mutation-scope.md. +const exitUnmeasured = 3 + func main() { - if err := command(os.Args[1:], os.Stderr); err != nil { - fmt.Fprintln(os.Stderr, err) - os.Exit(exitFailure) + err := command(os.Args[1:], os.Stderr) + if err == nil { + return + } + + fmt.Fprintln(os.Stderr, err) + os.Exit(exitCodeOf(err)) +} + +// exitCodeOf is what a shell, a hook or a CI step reads when ditto does not pass. +// +// A refusal, a usage error and a score below the bar stay one code, as they were: +// they are told apart by their message, which a reader has. The unmeasured scope +// is the one condition a wrapper is asked to act on differently, so it is the one +// condition with a number of its own. +func exitCodeOf(err error) int { + if _, ok := errors.AsType[ditto.UnmeasuredScopeError](err); ok { + return exitUnmeasured } + + return exitFailure } // errNoSubcommand is what an invocation with nothing to do reports. @@ -106,6 +135,10 @@ func usage(out io.Writer) { (also -v, --version) Run `+"`ditto run -h`"+` for its flags. + +Exit codes: 0 a score at or above the bar; 1 every other failure, including a +refusal and a usage error; 3 the scope holds mutants the test command does not +compile, which are reported as unmeasured rather than scored. `) } @@ -427,11 +460,22 @@ func describeRanges(ranges []ditto.Range) string { // are deciding what to type, and the readme is not. // The first backquoted word is not decoration: flag.PrintDefaults renders it as // the flag's VALUE NAME. Naming anything else there is how this line used to -// render as `-test-command -json`, which reads like a second flag. +// render as `-test-command -json`, which reads like a second flag. Any further +// backquoted word would render literally, so -json is spelled plainly below. +// +// The sentence this gained next is the second half of the same idea, and it +// replaced a wrong one. It used to say "name the package that owns the change +// instead", which is right only while the scope holds one package: a command +// that names one of five leaves the other four unmeasurable — nothing it runs +// can compile them — and the release now says so and fails. Measured on the +// report at the top of this file's list: 17 of 20 survivors in a package the +// command never builds. const testCommandHelp = "the `command` that decides whether a mutant died. It runs ONCE PER MUTANT, " + - "sequentially, so ./... costs your whole suite times your mutant count -- name the package " + - "that owns the change instead. -json is what lets ditto say WHY a mutant died; without it a " + - "mutant that never compiled is counted as killed" + "sequentially, so ./... costs your whole suite times your mutant count -- name every package the " + + "scope mutates instead, because a command that leaves one out does not score it: those mutants are " + + "reported as unmeasured and the run fails with exit 3. -json is what lets ditto say WHY a mutant " + + "died and which packages the command executes; without it a mutant that never compiled is counted " + + "as killed" // gatedHelp is the description of --gated on all three subcommands, one // constant so the three cannot drift apart. diff --git a/cmd/ditto/main_test.go b/cmd/ditto/main_test.go index 18bd203..ac2d849 100644 --- a/cmd/ditto/main_test.go +++ b/cmd/ditto/main_test.go @@ -3,6 +3,7 @@ package main import ( "bytes" "flag" + "fmt" "strings" "testing" @@ -67,6 +68,39 @@ func TestStagedGated(t *testing.T) { }) } +// TestUnmeasuredScopeHasItsOwnExitCode covers the half of the report that is +// about a wrapper rather than about a reader: a scope ditto cannot measure and a +// score below the bar were both 1, so an automated gate could not tell them apart +// — and the responses differ, because no test answers an unmeasurable scope. +// docs/reports/ditto-mutation-scope.md. +func TestUnmeasuredScopeHasItsOwnExitCode(t *testing.T) { + assert.Equal(t, exitUnmeasured, exitCodeOf(ditto.UnmeasuredScopeError{Unmeasured: 17, Scored: 26})) + + // Wrapped, because the error travels back through RunStaged and RunChanged + // before main sees it and any of them may add context on the way. + assert.Equal(t, exitUnmeasured, exitCodeOf(fmt.Errorf("running the change: %w", + ditto.UnmeasuredScopeError{Unmeasured: 1, Scored: 2}))) +} + +// And every other failure stays one code: a refusal, a usage error and a score +// below the bar are told apart by their message, which a reader has. +func TestEveryOtherFailureKeepsThePlainExitCode(t *testing.T) { + assert.Equal(t, exitFailure, exitCodeOf(ditto.ScoreBelowThresholdError{Minimum: 0.8})) + assert.Equal(t, exitFailure, exitCodeOf(ditto.NoMutantsError{})) + assert.Equal(t, exitFailure, exitCodeOf(errNoSubcommand)) +} + +// The legend is what makes the number usable, and a number nobody can look up is +// a number people guess at. +func TestUsageNamesTheExitCodes(t *testing.T) { + out := &bytes.Buffer{} + + usage(out) + + assert.Contains(t, out.String(), "Exit codes") + assert.Contains(t, out.String(), "unmeasured") +} + // TestTestCommandHelp covers backlog entry 23. The old description was accurate // and told the reader nothing about what the default costs, which is the one // thing they need at the moment of typing. @@ -79,7 +113,16 @@ func TestTestCommandHelp(t *testing.T) { }) t.Run("names the lever rather than only the toll", func(t *testing.T) { - assert.Contains(t, testCommandHelp, "name the package") + // It named "the package that owns the change", which is right only while + // the scope holds one package. A command that names one of five leaves + // the other four unmeasurable, and the release now fails on that instead + // of scoring it, so the lever is every package the scope mutates. + assert.Contains(t, testCommandHelp, "name every package the scope mutates") + }) + + t.Run("says what a command that leaves part of the scope out costs", func(t *testing.T) { + assert.Contains(t, testCommandHelp, "reported as unmeasured") + assert.Contains(t, testCommandHelp, "exit 3") }) t.Run("says what dropping -json costs, not only what it does", func(t *testing.T) { diff --git a/internal/consolereporter/consolereporter.go b/internal/consolereporter/consolereporter.go index 5967c45..afad8da 100644 --- a/internal/consolereporter/consolereporter.go +++ b/internal/consolereporter/consolereporter.go @@ -1,6 +1,7 @@ package consolereporter import ( + "path" "sort" "strings" @@ -24,6 +25,11 @@ type ConsoleReporter struct { // that failed -- and the score alone cannot tell them apart, because the // calculator reports -1 for both an empty run and nothing else. total int + + // unmeasured is how many mutants left the score for the other reason a + // mutant cannot be judged: the test command does not compile the package + // that owns it. See logUnmeasured. + unmeasured int } func New( @@ -46,64 +52,46 @@ func New( // laboratory's counters are, so no reporter is forced to answer. func (r *ConsoleReporter) Total() int { return r.total } +// Unmeasured is how many mutants the last Summarize left out because the test +// command does not compile the package that owns them. +// +// It is a separate count from Total for the reason the exclusion exists: those +// mutants are not a smaller denominator to read a score against, they are the +// part of the scope this run could not measure. The run fails on a non-zero count +// whatever the score says, and the exit code is how a gate hears it. +func (r *ConsoleReporter) Unmeasured() int { return r.unmeasured } + func (r *ConsoleReporter) AddDiagnostic(diagnostic *ditto.Diagnostic) { r.diagnostics = append(r.diagnostics, diagnostic) } func (r *ConsoleReporter) Summarize() result.Result[any] { - var killed, survived int - - survivors := []*ditto.Diagnostic{} - - for _, diagnostic := range r.diagnostics { - // A mutant that never compiled leaves the numerator AND the denominator. - // The kill predicate is undefined for a program that does not exist: - // Zhu, Hall & May, ACM Computing Surveys 29(4) 1997, Def 3.1 -- - // S = D / (M − E) -- and gremlins, cargo-mutants, Stryker and - // go-mutesting all exclude it. It is counted and named below instead, - // because it is a defect of the GENERATOR with a benchmark to answer to. - if diagnostic.IsOk() && diagnostic.Reason() == verdict.BuildFailed { - continue - } - - if diagnostic.IsOk() { - killed++ - } else { - survived++ - - survivors = append(survivors, diagnostic) - } - } + counted := r.count() - total := killed + survived + total := counted.killed + counted.survived r.total = total + r.unmeasured = counted.unmeasured // Addresses first, diffs after. Survivors are the only part of this report // anybody acts on, and printing them after the diffs turned the report into // an index into the log: the author scrolled back through every rendered // diff to recover where each survivor landed. - r.logAddresses(survivors) + r.logAddresses(counted.survivors) - for _, diagnostic := range survivors { + for _, diagnostic := range counted.survivors { r.logDiff(diagnostic) } - unearned := 0 - byVirus := map[string]int{} - - for _, diagnostic := range r.diagnostics { - if diagnostic.IsOk() && diagnostic.Reason() == verdict.BuildFailed { - unearned++ - byVirus[diagnostic.Virus()]++ - } - } - res := result.Ok[any](nil) scoreColor := color.BoldGreen scoreIcon := "✓" - score := r.calculator(total, killed) + score := r.calculator(total, counted.killed) - if score < r.minimumThreshold { + // A run that could not measure part of its scope did not pass, whatever the + // score over the rest says. The score is a measurement of a smaller + // population than the scope, so it cannot be the green one: the marker is + // about the run, and the line below says what left it out. + if score < r.minimumThreshold || counted.unmeasured > 0 { res = result.Err[any]("") scoreColor = color.BoldRed scoreIcon = "⨯" @@ -111,17 +99,128 @@ func (r *ConsoleReporter) Summarize() result.Result[any] { r.logger.Logf("┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┓") r.logger.Logf("┃ • "+color.Bold("Total")+": %8d ┃", total) - r.logger.Logf("┃ • "+color.Bold("Killed")+": %7d ┃", killed) - r.logger.Logf("┃ • "+color.Bold("Survived")+": %5d ┃", survived) + r.logger.Logf("┃ • "+color.Bold("Killed")+": %7d ┃", counted.killed) + r.logger.Logf("┃ • "+color.Bold("Survived")+": %5d ┃", counted.survived) + + // The composition of the score, beside the score: a reader who sees 26 has + // to see that the scope held 43. Printed only when there is something to say, + // because a line on every run is a line people stop reading. + if counted.unmeasured > 0 { + r.logger.Logf("┃ • "+color.Bold("Unmeasured")+": %8d ┃", counted.unmeasured) + } + r.logger.Logf("┠┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┨") r.logger.Logf("┃ " + scoreColor("%s Score: %8.2f (minimum: %.2f)", scoreIcon, score, r.minimumThreshold) + " ┃") r.logger.Logf("┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┛") - r.logUnearned(unearned, len(r.diagnostics), byVirus) + r.logUnmeasured(counted.unmeasured, len(r.diagnostics), counted.byDirectory) + r.logUnearned(counted.nonViable, len(r.diagnostics), counted.byVirus) return res } +// tally is what one pass over the diagnostics found: how many mutants of each +// kind, and the breakdowns the report names them by. +type tally struct { + killed int + survived int + unmeasured int + nonViable int + + survivors []*ditto.Diagnostic + byVirus map[string]int + byDirectory map[string]int +} + +// count is one pass over the diagnostics: what each mutant was, and why. +// +// One pass, because the report asks three questions about the same list and the +// exclusions have to be applied identically to all three. The order of the cases +// is the rule: +// +// - Unmeasured comes first. It is shaped like a survivor and is not one: it +// never ran, so counting it either way would answer a question this run did +// not ask. +// - A mutant that never compiled leaves the numerator AND the denominator. The +// kill predicate is undefined for a program that does not exist: Zhu, Hall & +// May, ACM Computing Surveys 29(4) 1997, Def 3.1 -- S = D / (M − E) -- and +// gremlins, cargo-mutants, Stryker and go-mutesting all exclude it. It is +// counted and named instead, because it is a defect of the GENERATOR with a +// benchmark to answer to. +// - Everything else is a kill or a survivor, which is what the score is. +func (r *ConsoleReporter) count() tally { + counted := tally{ + byVirus: map[string]int{}, + byDirectory: map[string]int{}, + } + + for _, diagnostic := range r.diagnostics { + switch { + case diagnostic.Unmeasured(): + counted.unmeasured++ + counted.byDirectory[path.Dir(diagnostic.Path())]++ + case diagnostic.IsOk() && diagnostic.Reason() == verdict.BuildFailed: + counted.nonViable++ + counted.byVirus[diagnostic.Virus()]++ + case diagnostic.IsOk(): + counted.killed++ + default: + counted.survived++ + counted.survivors = append(counted.survivors, diagnostic) + } + } + + return counted +} + +// logUnmeasured says how much of the scope the score above does not cover, and +// names the packages that left it. +// +// This is the one number the report cannot compute for itself. Ditto scopes +// mutants to a change and judges them with a single configured command, and when +// that command does not compile a mutant's package the mutant is not badly +// tested, it is unmeasured: nothing the command runs can kill it, so it survives, +// and a score counting it mixes "your tests missed this" with "your command +// cannot see this". Measured on the report that asked for it: 43 mutants, 23 +// killed, 20 survived, and 17 of those survivors in one package the command never +// builds -- a run whose ceiling was 0.605, printed as 0.53 against a bar of 0.80. +// docs/reports/ditto-mutation-scope.md. +// +// It says nothing when there is nothing to say, for the reason logUnearned does. +// The packages are named because they are the unit of the fix: one line per +// package, most mutants first, and no line at all for a package whose every +// mutant was measured. +func (r *ConsoleReporter) logUnmeasured(unmeasured, generated int, byDirectory map[string]int) { + if unmeasured == 0 { + return + } + + r.logger.Logf("┃ %s", color.BoldRed( + "%d of the %d mutants in this scope are never compiled by this test command, so", + unmeasured, generated)) + r.logger.Logf("┃ %s", color.BoldRed("nothing it runs can kill them and they are out of the score entirely:")) + + directories := make([]string, 0, len(byDirectory)) + for directory := range byDirectory { + directories = append(directories, directory) + } + + sort.Slice(directories, func(i, j int) bool { + if byDirectory[directories[i]] == byDirectory[directories[j]] { + return directories[i] < directories[j] + } + + return byDirectory[directories[i]] > byDirectory[directories[j]] + }) + + for _, directory := range directories { + r.logger.Logf("┃ %s (%d)", directory, byDirectory[directory]) + } + + r.logger.Logf("┃ %s", color.Bold( + "Name every package the scope mutates in --test-command, or narrow the scope.")) +} + // logAddresses prints one line per survivor: where it is, what hit it, and what // it wrote. It is the first thing on screen, and it says nothing at all when // nothing survived. diff --git a/internal/consolereporter/unmeasured_test.go b/internal/consolereporter/unmeasured_test.go new file mode 100644 index 0000000..3fe3f72 --- /dev/null +++ b/internal/consolereporter/unmeasured_test.go @@ -0,0 +1,87 @@ +package consolereporter_test + +import ( + "strings" + "testing" + + "github.com/Disble/ditto/internal/consolereporter" + "github.com/Disble/ditto/internal/ditto" + "github.com/Disble/ditto/internal/dittotesting/fakelogger" + "github.com/Disble/ditto/internal/dittotesting/fakescorecalculator" + "github.com/Disble/ditto/internal/dittotesting/stubdiffer" + "github.com/Disble/ditto/internal/future" + "github.com/Disble/ditto/internal/gomutatedfile" + "github.com/Disble/ditto/internal/result" + "github.com/stretchr/testify/assert" +) + +// A mutant the command cannot execute is not a smaller denominator to read a +// score against. It is the part of the scope this run could not measure, so it +// leaves the score on both sides, it is named with its package, and the run does +// not pass whatever the ratio over the rest says. +// +// Measured on the report that asked for this: 43 mutants, 23 killed, 20 +// survivors, and 17 of those survivors in a package the command never builds — +// printed as a score of 0.53 against a bar of 0.80, over a run whose ceiling was +// 0.605. docs/reports/ditto-mutation-scope.md. +func TestUnmeasuredMutantsLeaveTheScoreAndAreNamed(t *testing.T) { + logger := fakelogger.New() + + // The score is pinned to one thing only: 1.00, above the bar. A run that + // measured everything it scored passes; this one does not, and that is the + // assertion. + reporter := consolereporter.New(logger, stubdiffer.New(""), fakescorecalculator.Always(1.0), 0.8) + + for range 4 { + reporter.AddDiagnostic(ditto.NewDiagnostic( + future.Resolved(result.Ok("mutant killed")), + gomutatedfile.New("Comparison", "internal/observability/syncdiag/reader.go", nil, nil), + )) + } + + for range 17 { + reporter.AddDiagnostic(ditto.NewUnmeasuredDiagnostic( + gomutatedfile.New("Comparison Invert", "internal/desktop/app_runtime_services.go", nil, nil), + )) + } + + assert.False(t, reporter.Summarize().IsOk(), + "a run that could not measure part of its scope passed on the strength of the rest") + + assert.Equal(t, 4, reporter.Total(), "the unmeasured mutants are still counted in the denominator") + assert.Equal(t, 17, reporter.Unmeasured()) + + report := strings.Join(logger.LoggedLines(), "\n") + + assert.Contains(t, report, "17 of the 21 mutants in this scope are never compiled by this test command") + assert.Contains(t, report, "internal/desktop (17)") + assert.Contains(t, report, "Name every package the scope mutates in --test-command, or narrow the scope.") + + // A package that WAS measured is not named as unmeasured, and an unmeasured + // mutant is not rendered as a survivor: its diff would be a diff of a program + // that never ran. + assert.NotContains(t, report, "internal/observability/syncdiag") + assert.NotContains(t, report, "Mutant survived") + assert.NotContains(t, report, "internal/desktop/app_runtime_services.go") +} + +// And it says nothing when there is nothing to say, for the reason the +// non-viable line says nothing: a line printed on every run is a line people +// stop reading. +func TestAMeasuredScopeSaysNothingAboutBeingUnmeasured(t *testing.T) { + logger := fakelogger.New() + reporter := consolereporter.New(logger, stubdiffer.New(""), fakescorecalculator.Always(1.0), 0.8) + + reporter.AddDiagnostic(ditto.NewDiagnostic( + future.Resolved(result.Ok("mutant killed")), + gomutatedfile.New("Comparison", "calc/calc.go", nil, nil), + )) + + assert.True(t, reporter.Summarize().IsOk()) + assert.Zero(t, reporter.Unmeasured()) + + report := strings.Join(logger.LoggedLines(), "\n") + + assert.NotContains(t, report, "Unmeasured") + assert.NotContains(t, report, "never compiled by this test command") +} diff --git a/internal/ditto/ditto.go b/internal/ditto/ditto.go index 6c96e33..180b531 100644 --- a/internal/ditto/ditto.go +++ b/internal/ditto/ditto.go @@ -47,15 +47,43 @@ type BatchLaboratory interface { type ScoreCalculator func(total, killed int) float32 +// CommandScope answers whether the configured test command can execute the +// package that owns a source path. +// +// It is a separate interface rather than part of Laboratory, for the reason +// BatchLaboratory is: a laboratory that cannot answer is still a laboratory, and +// the release then runs exactly what it ran before. An implementation is allowed +// to answer true because it does not know -- see laboratory.Executes -- and the +// release treats that as "everything is executed", which is what ditto assumed +// until it could ask. +type CommandScope interface { + Executes(repository Repository, relativePath string) bool +} + type Diagnostic struct { res future.Future[result.Result[string]] file *gomutatedfile.GoMutatedFile + + // unmeasured is a mutant ditto chose not to run at all, because the test + // command cannot compile the package that owns it. It is not a verdict about + // the mutant, which is why it does not travel as one: the score leaves it out + // on both sides and the report names it. See NewUnmeasuredDiagnostic. + unmeasured bool } func (d *Diagnostic) IsOk() bool { return d.res.Await().IsOk() } +// Unmeasured reports a mutant that was never run because the test command does +// not compile the package that owns it. +// +// Read it BEFORE IsOk. An unmeasured diagnostic is shaped like a survivor -- that +// is what it would have been, before ditto could tell the difference -- so a +// consumer that ignores this reads exactly what it read before this existed, +// which is the graceful half. The shipped reporter does not ignore it. +func (d *Diagnostic) Unmeasured() bool { return d.unmeasured } + // Reason is why this mutant died, and Unknown when ditto was not told. // // Ditto recognises a killed mutant by a non-zero exit, which a mutant that never @@ -87,6 +115,10 @@ func (d *Diagnostic) Label() string { // rendered: where it is, and what it wrote there. func (d *Diagnostic) Address() string { return d.file.Address() } +// Path is the repository-relative source file this mutant came from. The report +// groups the mutants it could not measure by the directory this names. +func (d *Diagnostic) Path() string { return d.file.Path() } + // Virus names the mutation operator behind this diagnostic, which is the unit a // non-viable mutant is fixed in. docs/metrics.md metric 1. func (d *Diagnostic) Virus() string { return d.file.Virus() } @@ -99,6 +131,27 @@ func NewDiagnostic(res future.Future[result.Result[string]], file *gomutatedfile } } +// NewUnmeasuredDiagnostic is a mutant ditto did not run, because the configured +// test command does not compile the package that owns it. +// +// Nothing the command runs can kill it, so its survival is not evidence about +// anybody's tests: the score leaves it out of the numerator and the denominator, +// exactly as it leaves out a mutant that never compiled, and the report names it +// instead. Measured on the report that asked for it: 43 mutants, 20 survivors, +// 17 of them in one package the command never builds, printed as a score of 0.53 +// against a bar of 0.80. docs/reports/ditto-mutation-scope.md. +// +// It carries the result ditto would have reported before it could tell the +// difference -- a survivor -- so a consumer that does not know about Unmeasured +// reads what it read before, and none of them meets a nil. +func NewUnmeasuredDiagnostic(file *gomutatedfile.GoMutatedFile) *Diagnostic { + return &Diagnostic{ + res: future.Resolved(result.Err[string]("")), + file: file, + unmeasured: true, + } +} + type Reporter interface { AddDiagnostic(diagnostic *Diagnostic) Summarize() result.Result[any] @@ -109,14 +162,26 @@ type Ditto struct { repository Repository laboratory Laboratory reporter Reporter + + // scope is how the release asks whether the configured test command can + // execute a package at all. Nil when nothing can answer, and the release then + // runs every mutant of the scope, which is what it did before this existed. + scope CommandScope } -func New(logger Logger, repository Repository, laboratory Laboratory, reporter Reporter) *Ditto { +func New(logger Logger, repository Repository, laboratory Laboratory, reporter Reporter, scopes ...CommandScope) *Ditto { + var scope CommandScope + + if len(scopes) > 0 { + scope = scopes[0] + } + return &Ditto{ logger: logger, repository: repository, laboratory: laboratory, reporter: reporter, + scope: scope, } } @@ -126,6 +191,11 @@ func New(logger Logger, repository Repository, laboratory Laboratory, reporter R // that compiles once for a whole file has to receive the file's mutants at once. // The order they are reported in is unchanged: sources are walked in order, and // each file's mutants in the order its viruses produced them. +// +// A file whose package the test command cannot execute is not run at all: its +// mutants are recorded as unmeasured, and the report says so. Running them would +// buy a survivor nobody can kill with a full run of the suite, and the number +// that came back would be a mixture of two different questions. func (o *Ditto) Release(viri ...viruses.Virus) { for _, source := range o.repository.ListGoSourceFiles() { mutants := mutate(source.Incubate(viri...)) @@ -133,6 +203,14 @@ func (o *Ditto) Release(viri ...viruses.Virus) { continue } + if !o.executes(mutants[0].Path()) { + for _, mutant := range mutants { + o.reporter.AddDiagnostic(NewUnmeasuredDiagnostic(mutant)) + } + + continue + } + // Said before the file's mutants run, not after, because this is the // number a reader needs in order to know what the silence that follows // is going to cost. It is also the multiplier for the baseline duration @@ -146,6 +224,20 @@ func (o *Ditto) Release(viri ...viruses.Virus) { } } +// executes reports whether the configured test command can execute the package +// that owns a source path. +// +// Everything is executed when nothing can answer the question, and that is not a +// fallback: a command that is not `go test -json` has no readable package scope, +// and ditto has no business failing a run over a question it cannot read. +func (o *Ditto) executes(relativePath string) bool { + if o.scope == nil { + return true + } + + return o.scope.Executes(o.repository, relativePath) +} + func mutate(infected []*goinfectedfile.GoInfectedFile) []*gomutatedfile.GoMutatedFile { mutants := make([]*gomutatedfile.GoMutatedFile, 0, len(infected)) diff --git a/internal/dittotesting/fakereporter/counting.go b/internal/dittotesting/fakereporter/counting.go index 9018e52..42e661b 100644 --- a/internal/dittotesting/fakereporter/counting.go +++ b/internal/dittotesting/fakereporter/counting.go @@ -13,9 +13,16 @@ import ( // drops a capability is refused by nothing — backlog entry 12 — so the check is // written once and applied where each decorator lives. -// Counting reports a fixed total. The field is named Scored because Total is -// the method the optional interface asks for, and a struct cannot have both. -type Counting struct{ Scored int } +// Counting reports fixed counts. The fields are named Scored and Outside because +// Total and Unmeasured are the methods the optional interfaces ask for, and a +// struct cannot have both a field and a method of the same name. +type Counting struct { + Scored int + + // Outside is what Unmeasured reports: mutants outside what the test command + // compiles. + Outside int +} func (Counting) AddDiagnostic(*ditto.Diagnostic) {} func (Counting) Summarize() result.Result[any] { return result.Err[any]("") } @@ -23,6 +30,9 @@ func (Counting) Summarize() result.Result[any] { return result.Err[any]("") } // Total is the count this reporter was built with. func (r Counting) Total() int { return r.Scored } +// Unmeasured is how much of the scope the command could not execute. +func (r Counting) Unmeasured() int { return r.Outside } + // Countless is every reporter that existed before the count did, and any a // caller supplies. type Countless struct{} diff --git a/internal/dittotesting/fakereporter/fakereporter.go b/internal/dittotesting/fakereporter/fakereporter.go index b8887dd..8be9f3d 100644 --- a/internal/dittotesting/fakereporter/fakereporter.go +++ b/internal/dittotesting/fakereporter/fakereporter.go @@ -26,13 +26,21 @@ func (r *FakeReporter) Summarize() result.Result[any] { survived := 0 killed := 0 nonViable := 0 + unmeasured := 0 for _, diagnostic := range r.diagnostics { - // The same exclusion the shipped reporter applies. A double that scores - // differently from the thing it stands in for measures the old rules, - // and every test through it would keep agreeing with a version of ditto - // that no longer exists. A mutant that never compiled is out of both - // sides -- see internal/consolereporter and docs/metrics.md. + // The same exclusions the shipped reporter applies, in the same order. A + // double that scores differently from the thing it stands in for measures + // the old rules, and every test through it would keep agreeing with a + // version of ditto that no longer exists. An unmeasured mutant never ran, + // so it is neither a kill nor a survivor; a mutant that never compiled is + // out of both sides -- see internal/consolereporter and docs/metrics.md. + if diagnostic.Unmeasured() { + unmeasured++ + + continue + } + if diagnostic.IsOk() && diagnostic.Reason() == verdict.BuildFailed { nonViable++ @@ -47,12 +55,15 @@ func (r *FakeReporter) Summarize() result.Result[any] { } r.summary = &Summary{ - Survived: survived, - Killed: killed, - NonViable: nonViable, + Survived: survived, + Killed: killed, + NonViable: nonViable, + Unmeasured: unmeasured, } - if survived > 0 { + // A run that could not measure part of its scope did not pass, whatever the + // score over the rest says. The shipped reporter fails it the same way. + if survived > 0 || unmeasured > 0 { return result.Err[any]("") } diff --git a/internal/dittotesting/fakereporter/summary.go b/internal/dittotesting/fakereporter/summary.go index d9eda8d..a47e935 100644 --- a/internal/dittotesting/fakereporter/summary.go +++ b/internal/dittotesting/fakereporter/summary.go @@ -7,4 +7,9 @@ type Summary struct { // NonViable is the mutants that never compiled, which are out of both the // numerator and the denominator. docs/metrics.md metric 1. NonViable int + + // Unmeasured is the mutants the test command cannot execute, which leave the + // score for the other reason a mutant cannot be judged and fail the run. + // docs/reports/ditto-mutation-scope.md. + Unmeasured int } diff --git a/internal/gatedreporter/gatedreporter.go b/internal/gatedreporter/gatedreporter.go index 0d137de..2358257 100644 --- a/internal/gatedreporter/gatedreporter.go +++ b/internal/gatedreporter/gatedreporter.go @@ -81,3 +81,15 @@ func (r *GatedReporter) Total() int { return counted.Total() } + +// Unmeasured forwards how much of the scope the command could not execute, with +// the same reason and the same sentinel: an unreadable count is negative rather +// than zero, because zero is the answer that lets a run pass. +func (r *GatedReporter) Unmeasured() int { + counted, ok := r.delegate.(interface{ Unmeasured() int }) + if !ok { + return -1 + } + + return counted.Unmeasured() +} diff --git a/internal/gatedreporter/unmeasured_test.go b/internal/gatedreporter/unmeasured_test.go new file mode 100644 index 0000000..fe1509f --- /dev/null +++ b/internal/gatedreporter/unmeasured_test.go @@ -0,0 +1,25 @@ +package gatedreporter_test + +import ( + "testing" + + "github.com/Disble/ditto/internal/dittotesting/fakelogger" + "github.com/Disble/ditto/internal/dittotesting/fakereporter" + "github.com/Disble/ditto/internal/gatedreporter" + "github.com/stretchr/testify/assert" +) + +// This lives here rather than beside the reporter it decorates, for the reason +// total_test.go does: the stack the release actually reads through is this one. + +func TestUnmeasuredIsForwarded(t *testing.T) { + assert.Equal(t, 6, gatedreporter.New(fakelogger.New(), fakeGates{}, fakereporter.Counting{Outside: 6}).Unmeasured(), + "a decorator that drops a capability is refused by nothing") +} + +// An unreadable count is UNKNOWN, not zero. Zero is the answer that says the +// whole scope was measured, so a decorator that cannot forward this must not be +// able to say that: the release would pass a run it could not judge. +func TestAnUnreadableUnmeasuredCountIsNotZero(t *testing.T) { + assert.Equal(t, -1, gatedreporter.New(fakelogger.New(), fakeGates{}, fakereporter.Countless{}).Unmeasured()) +} diff --git a/internal/verbosereporter/unmeasured_test.go b/internal/verbosereporter/unmeasured_test.go new file mode 100644 index 0000000..d368216 --- /dev/null +++ b/internal/verbosereporter/unmeasured_test.go @@ -0,0 +1,25 @@ +package verbosereporter_test + +import ( + "testing" + + "github.com/Disble/ditto/internal/dittotesting/fakelogger" + "github.com/Disble/ditto/internal/dittotesting/fakereporter" + "github.com/Disble/ditto/internal/verbosereporter" + "github.com/stretchr/testify/assert" +) + +// This lives here rather than beside the reporter it decorates, for the reason +// total_test.go does: the stack the release actually reads through is this one. + +func TestUnmeasuredIsForwarded(t *testing.T) { + assert.Equal(t, 6, verbosereporter.New(fakelogger.New(), fakereporter.Counting{Outside: 6}).Unmeasured(), + "a decorator that drops a capability is refused by nothing") +} + +// An unreadable count is UNKNOWN, not zero. Zero is the answer that says the +// whole scope was measured, so a decorator that cannot forward this must not be +// able to say that: the release would pass a run it could not judge. +func TestAnUnreadableUnmeasuredCountIsNotZero(t *testing.T) { + assert.Equal(t, -1, verbosereporter.New(fakelogger.New(), fakereporter.Countless{}).Unmeasured()) +} diff --git a/internal/verbosereporter/verbosereporter.go b/internal/verbosereporter/verbosereporter.go index d3ed122..f14190a 100644 --- a/internal/verbosereporter/verbosereporter.go +++ b/internal/verbosereporter/verbosereporter.go @@ -42,3 +42,15 @@ func (r *VerboseReporter) Total() int { return counted.Total() } + +// Unmeasured forwards how much of the scope the command could not execute, with +// the same reason and the same sentinel: an unreadable count is negative rather +// than zero, because zero is the answer that lets a run pass. +func (r *VerboseReporter) Unmeasured() int { + counted, ok := r.delegate.(interface{ Unmeasured() int }) + if !ok { + return -1 + } + + return counted.Unmeasured() +} diff --git a/perf/baseline.json b/perf/baseline.json index 2719541..9479fe6 100644 --- a/perf/baseline.json +++ b/perf/baseline.json @@ -17,7 +17,7 @@ "laboratoryRunsForOneChangedFunction": 4, "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": 8, "testCommandInvocationsPerReleaseWholeFixture": 49, - "mutantsPerReleaseOnThisRepository": 905 + "mutantsPerReleaseOnThisRepository": 946 }, "targets": { "sourceParsesPerReleaseWithThreeViruses": "Reached: 4, one parse per source file, down from 12. GoSourceFile.Incubate now takes the whole mutator set and parses once for all of them. With the default 14 mutators this is 14 parses per file reduced to 1.", @@ -28,6 +28,6 @@ "laboratoryRunsForOneChangedFunction": "4, the mutators that fire on one changed line and nothing else in the repository. This is what WithChangedRanges buys: without it the same fixture charges 48.", "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": "8, exactly twice the single-file number. The two ranges name offsets that exist in both files, because every fixture file has the same byte layout, so a scope holding one flat set of ranges would charge 16 and grow as the square of the file count. Keeping the ranges beside their file makes that impossible rather than merely unlikely.", "testCommandInvocationsPerReleaseWholeFixture": "49 for the 48 mutants of the whole fixture, plus one. That one is the baseline: the laboratory runs the suite once on unmutated code before scoring anything, because a test command that fails before it compiles fails for every mutant too, and ditto recognises a killed mutant by exactly that. Measured on ditto's own gate before the guard existed: 431 of 431 killed in 5.46 seconds, a perfect score for a run that compiled nothing. Every other laboratory counter here goes through a stand-in and cannot see a run the laboratory makes on its own, which is why this one exists — a cost nobody records is one that grows unnoticed, the mirror of the unrecorded gain this file already refuses. It must not grow: one baseline per release, never one per mutant.", - "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once." + "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once. Then 946 (+41) for the same separation from the report outwards -- the score leaves those mutants out, names the packages that hold them, and the command exits 3 -- attributed per file in the committed state: internal/consolereporter/consolereporter.go +15, internal/ditto/ditto.go +11, internal/dittotesting/fakereporter/fakereporter.go +9, internal/verbosereporter/verbosereporter.go +2, internal/gatedreporter/gatedreporter.go +2 and cmd/ditto/main.go +2. The nine product files that commit touches sum to +51 and not to the +41 the ratchet reported; the difference is release.go's own +10, which this counter never sees because it carries the gate's exclusions and the gate skips release.go. Both numbers are right and they answer different questions: this counter speaks for the scope the mutation gate runs against, and the per-file sum speaks for the change. A first measurement of the same change said +59, and the eight it lost are the two loops of the reporter's summary folded into one pass -- a gain this file is required to record rather than hand back. docs/reports/ditto-mutation-scope.md." } } diff --git a/release.go b/release.go index 7b24d1e..9f5e4b7 100644 --- a/release.go +++ b/release.go @@ -144,6 +144,13 @@ func Run(options ...Option) error { } if !rel.summarize().IsOk() { + // A scope ditto could not measure all of comes first, because it is what + // the run is about: the score over the rest is real and the report says + // what left it, while naming the threshold would name the wrong problem. + if unmeasured := rel.unmeasured(); unmeasured > 0 { + return UnmeasuredScopeError{Unmeasured: unmeasured, Scored: rel.scored()} + } + // A scope with nothing mutable in it is not a suite that failed, and the // score cannot tell the two apart: the calculator reports -1 for an // empty run, which is below every threshold. Measured on ditto's own @@ -186,6 +193,30 @@ func (e ScoreBelowThresholdError) Error() string { return fmt.Sprintf("ditto: the mutation score is below the configured minimum of %.2f", e.Minimum) } +// UnmeasuredScopeError reports a run whose scope holds mutants the configured +// test command cannot execute. +// +// It is neither a refusal nor a threshold failure, which is why it is its own +// error and its own exit code: the mutants that could be judged were, the score +// printed is a real measurement — of a smaller population than the scope — and +// the report names what left it. A gate has to be able to tell the two apart, +// because the response is different: this one is answered by naming every package +// the scope mutates in --test-command, or by narrowing the scope, and no amount +// of testing answers it. docs/reports/ditto-mutation-scope.md. +type UnmeasuredScopeError struct { + // Unmeasured is how many mutants the command could not reach. + Unmeasured int + // Scored is how many it could, and what the ratio above was computed over. + Scored int +} + +func (e UnmeasuredScopeError) Error() string { + return fmt.Sprintf( + "ditto: %d of the %d mutants in this scope are never compiled by the test command, "+ + "so the score above measures the other %d", + e.Unmeasured, e.Unmeasured+e.Scored, e.Scored) +} + // release is one configured run, assembled once and driven by either entry // point. Splitting it out is what keeps the two from drifting: there is one // order in which the decorators wrap, and both callers get it. @@ -194,6 +225,7 @@ type release struct { logger ditto.Logger reporter ditto.Reporter lab ditto.Laboratory + base *laboratory.Laboratory sandboxes interface{ RemoveAll() error } } @@ -237,7 +269,12 @@ func newRelease(options []Option, hostVerbose bool) *release { reporter = verbosereporter.New(logger, reporter) } - lab, gates := assemble(opts, logger, loud) + // Built here rather than inside assemble, because the release keeps it: it is + // the only thing that can say which packages the test command executes, and + // the scope of the run has to be readable above the decorators. + base := laboratory.New(logger, opts.TestRunner, opts.TemporaryDir) + + lab, gates := assemble(base, opts, logger, loud) // Wrapped here, after the verbose decorator and before anything summarises, // so the counts are the last thing a run says. @@ -250,6 +287,7 @@ func newRelease(options []Option, hostVerbose bool) *release { logger: logger, reporter: reporter, lab: lab, + base: base, sandboxes: sandboxes, } } @@ -257,11 +295,31 @@ func newRelease(options []Option, hostVerbose bool) *release { // start is the run itself, identical whichever entry point asked for it. func (r *release) start() { r.logger.Logf("%s %s", color.Yellow("┃"), color.Green("Releasing Ditto…")) - ditto.New(r.logger, r.opts.Repository, r.lab, r.reporter).Release( + ditto.New(r.logger, r.opts.Repository, r.lab, r.reporter, r.scope()).Release( r.opts.Viruses..., ) } +// scope is what the release may ask which packages the test command executes. +// +// It is nil for a gated run, and that is a cost decision with a stated limit. +// Gated() replaces the execution plan of the exact module-scope command and +// nothing else, so the command that runs is `go test -count=1 ./...`: every +// package the module has, which the gated path already resolves for itself. +// Asking the ordinary laboratory would buy a second suite run to learn what the +// gated path knows, and one suite run per release is exactly the cost this +// change refuses to add. What it leaves uncovered on a gated run: a mutant no +// package compiles at all — a file under testdata, or one behind a build tag for +// another operating system — is not named as unmeasured there, and stays the +// survivor it has always been. +func (r *release) scope() ditto.CommandScope { + if r.opts.Gated && r.opts.commandScope == moduleScope { + return nil + } + + return r.base +} + // startWithoutPanicking turns a refusal into a value and leaves every other // panic alone. Recovering more than the one type this package raises on purpose // would turn a defect into an exit code. @@ -305,6 +363,22 @@ func (r *release) scored() int { return counted.Total() } +// unmeasured is how many mutants the run could not judge because the test +// command does not compile their packages, and -1 when the reporter cannot say. +// +// The sentinel is negative for the same reason scored's is: zero is the answer +// that lets a run pass, so "I cannot tell" must never be reported as it. A +// reporter that cannot forward this leaves the run failing on the score it did +// compute, which is the answer it would have given before this existed. +func (r *release) unmeasured() int { + counted, ok := r.reporter.(interface{ Unmeasured() int }) + if !ok { + return -1 + } + + return counted.Unmeasured() +} + // reclaim removes what a run left behind. // // Sandboxes outlive each mutant now, so removing them belongs to the run rather @@ -330,8 +404,8 @@ func (r *release) reclaim() { // are the only thing that can say whether the gated path engaged, and a decorator // above it cannot be asked. The pointer is nil, concretely rather than as an // interface holding a nil, when the run is not gated. -func assemble(opts Options, logger ditto.Logger, loud bool) (ditto.Laboratory, *gatedlaboratory.GatedLaboratory) { - var lab ditto.Laboratory = laboratory.New(logger, opts.TestRunner, opts.TemporaryDir) +func assemble(base *laboratory.Laboratory, opts Options, logger ditto.Logger, loud bool) (ditto.Laboratory, *gatedlaboratory.GatedLaboratory) { + var lab ditto.Laboratory = base var gates *gatedlaboratory.GatedLaboratory diff --git a/release_golden_test.go b/release_golden_test.go index a26d6b8..a0026d4 100644 --- a/release_golden_test.go +++ b/release_golden_test.go @@ -26,6 +26,14 @@ import ( // The fixture is copied to a temporary directory and mutated there. It is never // mutated where it sits, and the run is pointed at the copy — the rule from // AGENTS.md, applied to ditto's own suite rather than quoted at other people. +// +// The sample moved once, in 0.12.0, and the move is the whole reason this note +// exists: a release now asks the laboratory which packages the test command can +// execute before it announces a file, and that first question is what pays for +// the baseline. So the baseline line prints before the first file's +// announcement instead of between it and the first mutant. No verdict moved and +// every address is identical, which is the claim a golden has to be able to +// support when it changes. func TestReleaseGolden(t *testing.T) { if testing.Short() { t.Skip("runs a full release: one test process per mutant") diff --git a/testdata/golden/release.txt b/testdata/golden/release.txt index 335792b..6c99921 100644 --- a/testdata/golden/release.txt +++ b/testdata/golden/release.txt @@ -1,7 +1,7 @@ ┃ Releasing Ditto… +┃ baseline: the suite took on unmutated code, and every mutant runs it again. ┃ calc/calc.go — 7 mutants ┃ calc/calc.go:9:12 → Arithmetic -┃ baseline: the suite took on unmutated code, and every mutant runs it again. ┃ calc/calc.go:12:11 → Arithmetic ┃ calc/calc.go:16:41 → Arithmetic ┃ calc/calc.go:4:41 → Comparison From b57e8ea188214a538b3c5baa6f8fc099e1a01810 Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 20:11:54 -0500 Subject: [PATCH 5/9] test(golden): pin what a release says and exits with when the command cannot compile part of its scope The fixture holds all three classes at once, which is what makes it worth its runtime: two mutants the named package's tests kill, one that survives them, and one in a package the command never compiles. The middle package is the load-bearing one -- `shared` is not named either, and it is NOT unmeasured, because the named package's tests import it and its mutant dies to them. A check that compared names instead of reading the toolchain's closure fails this test. The exit code is pinned beside the output, because it is the half a wrapper reads, and the control is the same fixture under a command that names every package: nothing unmeasured, exit 0, and the shared mutant still judged. So the failure the first half asserts is about the command's scope and not the fixture. Generated with DITTO_GOLDEN_UPDATE=1 and read before it was trusted. --- testdata/golden/unmeasured.txt | 39 ++++++ testdata/unmeasuredproject/covered/covered.go | 6 + .../unmeasuredproject/covered/covered_test.go | 23 ++++ testdata/unmeasuredproject/shared/shared.go | 8 ++ .../unmeasuredproject/untested/untested.go | 8 ++ unmeasured_golden_test.go | 123 ++++++++++++++++++ 6 files changed, 207 insertions(+) create mode 100644 testdata/golden/unmeasured.txt create mode 100644 testdata/unmeasuredproject/covered/covered.go create mode 100644 testdata/unmeasuredproject/covered/covered_test.go create mode 100644 testdata/unmeasuredproject/shared/shared.go create mode 100644 testdata/unmeasuredproject/untested/untested.go create mode 100644 unmeasured_golden_test.go diff --git a/testdata/golden/unmeasured.txt b/testdata/golden/unmeasured.txt new file mode 100644 index 0000000..72c9404 --- /dev/null +++ b/testdata/golden/unmeasured.txt @@ -0,0 +1,39 @@ +┃ Releasing Ditto… +┃ baseline: the suite took on unmutated code, and every mutant runs it again. +┃ covered/covered.go — 2 mutants +┃ covered/covered.go:5:12 → Comparison +┃ covered/covered.go:5:11 → Comparison Invert +┃ shared/shared.go — 1 mutants +┃ shared/shared.go:7:11 → Arithmetic +┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╍┅ +┃ 🧬 Survivors +┠┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄ +┃ covered/covered.go:5:12 → Comparison (deletes =) +┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╍┅ +┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╍┅ +┃ 🧬 Mutant survived: covered/covered.go:5:12 → Comparison +┠┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄ +┃ --- covered/covered.go (original) +┃ +++ covered/covered.go (mutated with 'Comparison') +┃ @@ -2,5 +2,5 @@ +┃ +┃ // Over is exercised from both sides, so its Comparison Invert mutant dies. +┃ func Over(a, b int) bool { +┃ - return a >= b +┃ + return a > b +┃ } +┃ +┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╍┅ +┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┓ +┃ • Total: 3 ┃ +┃ • Killed: 2 ┃ +┃ • Survived: 1 ┃ +┃ • Unmeasured: 1 ┃ +┠┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┨ +┃ ⨯ Score: 0.67 (minimum: 0.00) ┃ +┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┛ +┃ 1 of the 4 mutants in this scope are never compiled by this test command, so +┃ nothing it runs can kill them and they are out of the score entirely: +┃ untested (1) +┃ Name every package the scope mutates in --test-command, or narrow the scope. +ditto: 1 of the 4 mutants in this scope are never compiled by the test command, so the score above measures the other 3 diff --git a/testdata/unmeasuredproject/covered/covered.go b/testdata/unmeasuredproject/covered/covered.go new file mode 100644 index 0000000..4ac51e6 --- /dev/null +++ b/testdata/unmeasuredproject/covered/covered.go @@ -0,0 +1,6 @@ +package covered + +// Over is exercised from both sides, so its Comparison Invert mutant dies. +func Over(a, b int) bool { + return a >= b +} diff --git a/testdata/unmeasuredproject/covered/covered_test.go b/testdata/unmeasuredproject/covered/covered_test.go new file mode 100644 index 0000000..ac9125c --- /dev/null +++ b/testdata/unmeasuredproject/covered/covered_test.go @@ -0,0 +1,23 @@ +package covered + +import ( + "testing" + + "unmeasuredproject/shared" +) + +func TestOver(t *testing.T) { + if !Over(2, 1) || Over(1, 2) { + t.Fatal("Over is wrong") + } +} + +// The fixture's third package reaches the score through this test and not +// through the command's package list: a mutant in shared is compiled into this +// package's test binary, so the command CAN kill it, and the run that follows +// has to judge it rather than report it as unmeasured. +func TestSum(t *testing.T) { + if shared.Sum(2, 1) != 3 { + t.Fatal("Sum is wrong") + } +} diff --git a/testdata/unmeasuredproject/shared/shared.go b/testdata/unmeasuredproject/shared/shared.go new file mode 100644 index 0000000..9a30878 --- /dev/null +++ b/testdata/unmeasuredproject/shared/shared.go @@ -0,0 +1,8 @@ +package shared + +// Sum has no tests of its own. It is compiled into the covered package's test +// binary, which is what makes its mutants killable by a command that never names +// this package. +func Sum(a, b int) int { + return a + b +} diff --git a/testdata/unmeasuredproject/untested/untested.go b/testdata/unmeasuredproject/untested/untested.go new file mode 100644 index 0000000..7973f5a --- /dev/null +++ b/testdata/unmeasuredproject/untested/untested.go @@ -0,0 +1,8 @@ +package untested + +// Reduce is imported by nothing and tested by nothing, so no command that names +// the covered package can compile it. It exists so the golden has a package +// whose mutants are unmeasurable rather than merely untested. +func Reduce(a, b int) int { + return a - b +} diff --git a/unmeasured_golden_test.go b/unmeasured_golden_test.go new file mode 100644 index 0000000..9d36d92 --- /dev/null +++ b/unmeasured_golden_test.go @@ -0,0 +1,123 @@ +package ditto_test + +import ( + "errors" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" +) + +// TestUnmeasuredScopeGolden pins what a release says — and what it exits with — +// when the configured test command cannot compile part of the scope it was +// pointed at. +// +// This is the report's own case, in a fixture that holds all three classes at +// once. Measured on the repository that reported it: 43 mutants, 23 killed, 20 +// survived, and 17 of those survivors in one package the command never builds, +// printed as a score of 0.53 against a bar of 0.80 over a run whose ceiling was +// 0.605. docs/reports/ditto-mutation-scope.md. +// +// The fixture's middle package is the load-bearing one. `shared` is not named by +// the command either, and it is NOT unmeasured: the named package's tests import +// it, so its mutants are compiled into the binary that runs and the command can +// kill them. A check that compared names instead of reading the toolchain's +// closure would fail this test, which is why it is here rather than only in the +// unit tests of internal/commandscope. +func TestUnmeasuredScopeGolden(t *testing.T) { + if testing.Short() { + t.Skip("runs a full release: one test process per mutant") + } + + moduleRoot := moduleRoot(t) + project := t.TempDir() + + copyTree(t, filepath.Join(moduleRoot, "testdata", "unmeasuredproject"), project) + + // The fixture depends on nothing, not even on ditto: it is the tree a run is + // pointed at, and the run under test is the shipped command. + writeFile(t, filepath.Join(project, "go.mod"), "module unmeasuredproject\n\ngo 1.27\n") + + binary := filepath.Join(t.TempDir(), "ditto"+binarySuffix()) + build := command(t, moduleRoot, "go", "build", "-o", binary, "./cmd/ditto") + + if output, err := build.CombinedOutput(); err != nil { + t.Fatalf("building the command: %v\n%s", err, output) + } + + output, exitCode := runOutputAndExit(t, project, binary, + "run", "--root", project, + "--test-command", "go test -count=1 -json ./covered/", + "--threshold", "0", + ) + + got := withoutClocks(output) + golden := filepath.Join(moduleRoot, "testdata", "golden", "unmeasured.txt") + + if os.Getenv("DITTO_GOLDEN_UPDATE") == "1" { + writeFile(t, golden, got) + t.Log("golden updated; rerun without DITTO_GOLDEN_UPDATE to check it") + + return + } + + want := readFile(t, golden) + if got != want { + t.Fatalf("the release said something different.\n--- want ---\n%s\n--- got ---\n%s", want, got) + } + + // The exit code is the half of this that a wrapper reads. A scope ditto could + // not measure and a score below the bar were both 1 before this, and the + // responses differ: the second is answered by testing more, and the first by + // naming every package the scope mutates or by narrowing the scope. + if exitCode != 3 { + t.Fatalf("the run exited %d, want 3 for a scope the command cannot measure", exitCode) + } + + // The control, and it is not decoration: the same fixture, the same mutants, + // one command that names every package. Nothing is unmeasured, the run + // passes, and the `shared` mutant is judged rather than set aside — so the + // failure above is about the command's scope and not about the fixture. + control, controlExit := runOutputAndExit(t, project, binary, + "run", "--root", project, + "--test-command", "go test -count=1 -json ./...", + "--threshold", "0", + ) + + if controlExit != 0 { + t.Fatalf("the control exited %d, want 0:\n%s", controlExit, control) + } + + for _, quiet := range []string{"Unmeasured", "never compiled by this test command"} { + if strings.Contains(control, quiet) { + t.Fatalf("the control reported an unmeasurable scope (%q):\n%s", quiet, control) + } + } + + if !strings.Contains(control, "shared/shared.go") { + t.Fatalf("the control never mentioned the package the narrow command also reaches:\n%s", control) + } +} + +// runOutputAndExit runs the shipped command and returns its output and its exit +// code, because one of them is the subject and the other is the evidence. +func runOutputAndExit(t *testing.T, dir, binary string, args ...string) (string, int) { + t.Helper() + + run := command(t, dir, binary, args...) + + output, err := run.CombinedOutput() + normalised := strings.ReplaceAll(string(output), "\r\n", "\n") + + if err == nil { + return normalised, 0 + } + + var failed *exec.ExitError + if !errors.As(err, &failed) { + t.Fatalf("running the command: %v\n%s", err, output) + } + + return normalised, failed.ExitCode() +} From a5dfe5b6cb11bcf8d242d1885a3c2e117ed92d23 Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 20:16:44 -0500 Subject: [PATCH 6/9] feat(staged): --include-prefix, the mirror of --exclude-prefix The report asked for it in the same breath as the unmeasurable-score fix, and the fix is what makes it necessary: a run now fails when the test command cannot compile part of the scope, and the honest answer is usually to narrow the run to the package the command names rather than to widen the command. One flag does that; --exclude-prefix needs one flag per other package the change happens to touch. The public shape changes with it, and that is the point rather than a side effect: PlanStaged, RunStaged, PlanChanged and RunChanged take a ditto.Prefixes instead of one []string, so choosing what a run is about and narrowing it are one argument instead of two that can drift apart. An empty entry in either list is ignored rather than honoured, on both sides and for the same reason: every path starts with the empty string, so a list carrying one empty entry means 'no filter' when it was built and 'nothing at all' when it was not -- and the second reading silently mutates no files. Include narrows and exclude still removes, which the tests hold as a pair: a file has to survive both to be planned. -h names the pairing, because the moment a reader needs it is the moment a run just refused their scope. --- changed.go | 8 +-- changed_mutation_test.go | 4 +- changed_scope_test.go | 65 ++++++++++++++++++++---- cmd/ditto/main.go | 46 ++++++++++------- cmd/ditto/main_test.go | 41 +++++++++++++++ internal/staged/changed.go | 7 +-- internal/staged/changed_internal_test.go | 4 +- internal/staged/staged.go | 45 +++++++++++++--- internal/staged/staged_internal_test.go | 37 +++++++++++++- perf/baseline.json | 4 +- staged.go | 28 ++++++++-- staged_mutation_test.go | 4 +- 12 files changed, 237 insertions(+), 56 deletions(-) diff --git a/changed.go b/changed.go index ebb91cf..3f509b3 100644 --- a/changed.go +++ b/changed.go @@ -19,13 +19,13 @@ import ( // The scope is `base...HEAD`, the diff against their merge base, so a base that // has moved on since the change was written does not drag somebody else's // commits into the bill. -func PlanChanged(directory, baseRef string, excludePrefixes []string) (StagedPlan, error) { +func PlanChanged(directory, baseRef string, prefixes Prefixes) (StagedPlan, error) { repository, err := staged.New(staged.OSRunner{}, directory) if err != nil { return StagedPlan{}, fmt.Errorf("reading the repository: %w", err) } - files, err := repository.ChangedFiles(baseRef, excludePrefixes) + files, err := repository.ChangedFiles(baseRef, prefixes.Exclude, prefixes.Include) if err != nil { return StagedPlan{}, fmt.Errorf("reading the changed files: %w", err) } @@ -63,8 +63,8 @@ func PlanChanged(directory, baseRef string, excludePrefixes []string) (StagedPla // Everything below the scope is the staged path unchanged: the same sandbox, the // same `.ditto.json` for what git does not carry, the same notice when the diff // could not be turned into ranges. -func RunChanged(directory, baseRef string, excludePrefixes []string, options ...Option) error { - plan, err := PlanChanged(directory, baseRef, excludePrefixes) +func RunChanged(directory, baseRef string, prefixes Prefixes, options ...Option) error { + plan, err := PlanChanged(directory, baseRef, prefixes) if err != nil { return err } diff --git a/changed_mutation_test.go b/changed_mutation_test.go index 8ec6e9e..7cec1fd 100644 --- a/changed_mutation_test.go +++ b/changed_mutation_test.go @@ -38,7 +38,7 @@ func TestChangedMutation(t *testing.T) { t.Skipf("set %s to the ref this change is measured against, for example a release tag", baseRefVariable) } - plan, err := ditto.PlanChanged(".", base, []string{"testdata/"}) + plan, err := ditto.PlanChanged(".", base, ditto.Prefixes{Exclude: []string{"testdata/"}}) if err != nil { t.Fatalf("reading the change since %s: %v", base, err) } @@ -60,7 +60,7 @@ func TestChangedMutation(t *testing.T) { t.Log(notice) } - if err := ditto.RunChanged(".", base, []string{"testdata/"}, + if err := ditto.RunChanged(".", base, ditto.Prefixes{Exclude: []string{"testdata/"}}, ditto.ForceColors(), ditto.WithTestCommand(makeCommand(t)+" test.failfast MAKEFLAGS="), ditto.WithMinimumThreshold(0.5), diff --git a/changed_scope_test.go b/changed_scope_test.go index 5f11dd2..8f6affd 100644 --- a/changed_scope_test.go +++ b/changed_scope_test.go @@ -23,7 +23,7 @@ func TestPlanChangedReadsACommittedChange(t *testing.T) { dittotesting.Git(t, dir, "add", "-A") dittotesting.Git(t, dir, "commit", "-m", "add") - plan, err := ditto.PlanChanged(dir, "base", nil) + plan, err := ditto.PlanChanged(dir, "base", ditto.Prefixes{}) if err != nil { t.Fatalf("planning: %v", err) } @@ -51,7 +51,7 @@ func TestPlanChangedIsEmptyWhenNoGoSourceMoved(t *testing.T) { dittotesting.Git(t, dir, "add", "-A") dittotesting.Git(t, dir, "commit", "-m", "docs") - plan, err := ditto.PlanChanged(dir, "base", nil) + plan, err := ditto.PlanChanged(dir, "base", ditto.Prefixes{}) if err != nil { t.Fatalf("planning: %v", err) } @@ -68,7 +68,7 @@ func TestPlanChangedExcludesByPrefix(t *testing.T) { dittotesting.Git(t, dir, "add", "-A") dittotesting.Git(t, dir, "commit", "-m", "tool") - plan, err := ditto.PlanChanged(dir, "base", []string{"tools/"}) + plan, err := ditto.PlanChanged(dir, "base", ditto.Prefixes{Exclude: []string{"tools/"}}) if err != nil { t.Fatalf("planning: %v", err) } @@ -78,11 +78,56 @@ func TestPlanChangedExcludesByPrefix(t *testing.T) { } } +// The mirror of the exclusion above, and the one the report asked for: a scope +// that holds more packages than the test command can compile is answered by +// narrowing the scope to the package the command names, and one flag does it +// where one exclusion per other package does not. +func TestPlanChangedIncludesByPrefix(t *testing.T) { + dir := dittotesting.GitRepository(t) + + dittotesting.WriteFile(t, dir, "tools/tool.go", "package tools\n\nfunc Tool(a, b int) bool { return a > b }\n") + dittotesting.WriteFile(t, dir, "web/web.go", "package web\n\nfunc Web(a, b int) bool { return a > b }\n") + dittotesting.Git(t, dir, "add", "-A") + dittotesting.Git(t, dir, "commit", "-m", "two packages") + + plan, err := ditto.PlanChanged(dir, "base", ditto.Prefixes{Include: []string{"web/"}}) + if err != nil { + t.Fatalf("planning: %v", err) + } + + if len(plan.Files) != 1 || plan.Files[0] != "web/web.go" { + t.Fatalf("files = %v, want only web/web.go", plan.Files) + } +} + +// Include narrows, exclude still removes: the two are one decision about what +// the run is about, so a file has to survive both to be planned. +func TestPlanChangedIncludesAndExcludesTogether(t *testing.T) { + dir := dittotesting.GitRepository(t) + + dittotesting.WriteFile(t, dir, "web/kept.go", "package web\n\nfunc Kept(a, b int) bool { return a > b }\n") + dittotesting.WriteFile(t, dir, "web/dropped.go", "package web\n\nfunc Dropped(a, b int) bool { return a > b }\n") + dittotesting.Git(t, dir, "add", "-A") + dittotesting.Git(t, dir, "commit", "-m", "one package, two files") + + plan, err := ditto.PlanChanged(dir, "base", ditto.Prefixes{ + Include: []string{"web/"}, + Exclude: []string{"web/dropped"}, + }) + if err != nil { + t.Fatalf("planning: %v", err) + } + + if len(plan.Files) != 1 || plan.Files[0] != "web/kept.go" { + t.Fatalf("files = %v, want only web/kept.go", plan.Files) + } +} + // A base that does not exist is an error rather than an empty scope. The two are // the same exit code and opposite meanings: one is a change with nothing in it, // the other is a question git could not answer. func TestPlanChangedRefusesAnUnknownBase(t *testing.T) { - _, err := ditto.PlanChanged(dittotesting.GitRepository(t), "no-such-ref", nil) + _, err := ditto.PlanChanged(dittotesting.GitRepository(t), "no-such-ref", ditto.Prefixes{}) if err == nil { t.Fatal("an unknown base was accepted") } @@ -93,7 +138,7 @@ func TestPlanChangedRefusesAnUnknownBase(t *testing.T) { } func TestPlanChangedRefusesSomewhereThatIsNotARepository(t *testing.T) { - if _, err := ditto.PlanChanged(t.TempDir(), "base", nil); err == nil { + if _, err := ditto.PlanChanged(t.TempDir(), "base", ditto.Prefixes{}); err == nil { t.Fatal("a directory outside any repository was accepted") } } @@ -112,7 +157,7 @@ func TestRunChangedRefusesAStagedChange(t *testing.T) { dittotesting.WriteFile(t, dir, "kept.go", "package fixture\n\nfunc Kept() int { return 2 }\n") dittotesting.Git(t, dir, "add", "kept.go") - err := ditto.RunChanged(dir, "base", nil) + err := ditto.RunChanged(dir, "base", ditto.Prefixes{}) if err == nil { t.Fatal("a staged change was accepted") } @@ -131,19 +176,19 @@ func TestRunChangedDoesNothingWhenNothingChanged(t *testing.T) { dittotesting.Git(t, dir, "add", "-A") dittotesting.Git(t, dir, "commit", "-m", "docs") - if err := ditto.RunChanged(dir, "base", nil); err != nil { + if err := ditto.RunChanged(dir, "base", ditto.Prefixes{}); err != nil { t.Fatalf("a docs-only commit was reported as a failure: %v", err) } } func TestRunChangedRefusesAnUnknownBase(t *testing.T) { - if err := ditto.RunChanged(dittotesting.GitRepository(t), "no-such-ref", nil); err == nil { + if err := ditto.RunChanged(dittotesting.GitRepository(t), "no-such-ref", ditto.Prefixes{}); err == nil { t.Fatal("an unknown base was accepted") } } func TestRunChangedRefusesSomewhereThatIsNotARepository(t *testing.T) { - if err := ditto.RunChanged(t.TempDir(), "base", nil); err == nil { + if err := ditto.RunChanged(t.TempDir(), "base", ditto.Prefixes{}); err == nil { t.Fatal("a directory outside any repository was accepted") } } @@ -168,7 +213,7 @@ func TestRunChangedAcceptsAChangeWithNothingMutableInIt(t *testing.T) { "gated": {ditto.Gated()}, } { t.Run(name, func(t *testing.T) { - if err := ditto.RunChanged(dir, "base", nil, options...); err != nil { + if err := ditto.RunChanged(dir, "base", ditto.Prefixes{}, options...); err != nil { t.Fatalf("a change with nothing mutable in it was reported as a failure: %v", err) } }) diff --git a/cmd/ditto/main.go b/cmd/ditto/main.go index a319780..fabf4a4 100644 --- a/cmd/ditto/main.go +++ b/cmd/ditto/main.go @@ -142,14 +142,14 @@ compile, which are reported as unmeasured rather than scored. `) } -// excludes collects a flag that may appear more than once, because a repository -// rarely has exactly one thing worth leaving out. -type excludes []string +// prefixes collects a flag that may appear more than once, because a repository +// rarely has exactly one thing worth leaving out — or one thing worth keeping. +type prefixes []string -func (e *excludes) String() string { return strings.Join(*e, ",") } +func (p *prefixes) String() string { return strings.Join(*p, ",") } -func (e *excludes) Set(value string) error { - *e = append(*e, value) +func (p *prefixes) Set(value string) error { + *p = append(*p, value) return nil } @@ -164,7 +164,7 @@ func runCommand(args []string) error { loud := flags.Bool("verbose", false, "print what the run is doing as it does it") sandbox := flags.String("sandbox", "", `how each file reaches the sandbox: "copy" (default), "hardlink" or "link"`) - var exclude excludes + var exclude prefixes flags.Var(&exclude, "exclude", "regexp of source paths not to mutate; repeatable") @@ -198,7 +198,7 @@ func optionsFor( root, testCommand string, threshold float32, gated, confirm, loud bool, - exclude excludes, + exclude prefixes, ) ([]ditto.Option, error) { options := []ditto.Option{ ditto.WithRepositoryRoot(root), @@ -242,9 +242,10 @@ func stagedCommand(args []string, out io.Writer) error { loud := flags.Bool("verbose", false, "print what the run is doing as it does it") sandbox := flags.String("sandbox", "", `how each file reaches the sandbox: "copy" (default), "hardlink" or "link"`) - var exclude excludes + var exclude, include prefixes flags.Var(&exclude, "exclude-prefix", "repository-relative prefix never worth mutating; repeatable") + flags.Var(&include, "include-prefix", "only mutate paths under this repository-relative prefix; repeatable") // There is no flag for .ditto.json, so `-h` is the one place a reader would // look and not find it. Named here rather than left to the readme. @@ -269,12 +270,13 @@ func stagedCommand(args []string, out io.Writer) error { } if *dry { - return reportPlan(*directory, exclude, out) + return reportPlan(*directory, ditto.Prefixes{Exclude: exclude, Include: include}, out) } options := stagedOptions(*testCommand, float32(*threshold), *gated, *confirm, *loud, *sandbox) - return ditto.RunStaged(*directory, exclude, options...) //nolint:wrapcheck // this is the top of the program: the message is already the one a reader needs + //nolint:wrapcheck // this is the top of the program: the message is already the one a reader needs + return ditto.RunStaged(*directory, ditto.Prefixes{Exclude: exclude, Include: include}, options...) } // stagedOptions is what the staged flags mean, kept apart from reading them. @@ -326,9 +328,10 @@ func changedCommand(args []string, out io.Writer) error { loud := flags.Bool("verbose", false, "print what the run is doing as it does it") sandbox := flags.String("sandbox", "", `how each file reaches the sandbox: "copy" (default), "hardlink" or "link"`) - var exclude excludes + var exclude, include prefixes flags.Var(&exclude, "exclude-prefix", "repository-relative prefix never worth mutating; repeatable") + flags.Var(&include, "include-prefix", "only mutate paths under this repository-relative prefix; repeatable") flags.Usage = func() { fmt.Fprintln(os.Stderr, "Usage of ditto changed:") @@ -353,12 +356,13 @@ func changedCommand(args []string, out io.Writer) error { } if *dry { - return reportChangedPlan(*directory, *since, exclude, out) + return reportChangedPlan(*directory, *since, ditto.Prefixes{Exclude: exclude, Include: include}, out) } options := stagedOptions(*testCommand, float32(*threshold), *gated, *confirm, *loud, *sandbox) - return ditto.RunChanged(*directory, *since, exclude, options...) //nolint:wrapcheck // this is the top of the program: the message is already the one a reader needs + //nolint:wrapcheck // this is the top of the program: the message is already the one a reader needs + return ditto.RunChanged(*directory, *since, ditto.Prefixes{Exclude: exclude, Include: include}, options...) } // errNoBaseRef refuses to guess. There is no default that is right in a CI @@ -380,8 +384,8 @@ while nothing is modified or staged. // reportChangedPlan answers what a committed change would cost without paying // for it. -func reportChangedPlan(directory, baseRef string, exclude excludes, out io.Writer) error { - plan, err := ditto.PlanChanged(directory, baseRef, exclude) +func reportChangedPlan(directory, baseRef string, prefixes ditto.Prefixes, out io.Writer) error { + plan, err := ditto.PlanChanged(directory, baseRef, prefixes) if err != nil { return fmt.Errorf("reading the change: %w", err) } @@ -407,8 +411,8 @@ func reportChangedPlan(directory, baseRef string, exclude excludes, out io.Write // reportPlan answers what a staged change would cost without paying for it. A // dry run that materialised a sandbox or started a suite would not be one. -func reportPlan(directory string, exclude excludes, out io.Writer) error { - plan, err := ditto.PlanStaged(directory, exclude) +func reportPlan(directory string, prefixes ditto.Prefixes, out io.Writer) error { + plan, err := ditto.PlanStaged(directory, prefixes) if err != nil { return fmt.Errorf("reading the staged change: %w", err) } @@ -517,4 +521,10 @@ paths git does not carry, in a .ditto.json at its root: They are copied from the working tree after the index is materialised, and each copy is announced. Naming a path git tracks is refused: the index version is the one a staged run measures. + +--exclude-prefix and --include-prefix narrow what is mutated, and the second is +the one to reach for when a release reports mutants its test command cannot +compile: narrowing the run to the package the command names is cheaper and more +precise than widening the command, and it is one flag instead of one exclusion +per package the change happens to touch. Exit 3 is that report. ` diff --git a/cmd/ditto/main_test.go b/cmd/ditto/main_test.go index ac2d849..ec321bb 100644 --- a/cmd/ditto/main_test.go +++ b/cmd/ditto/main_test.go @@ -192,6 +192,47 @@ func TestGatedHelpRendersWithoutAValueName(t *testing.T) { assert.NotContains(t, rendered.String(), "-gated ", "a value name rendered after the bool flag") } +// TestIncludePrefixReachesThePlan covers the flag the report asked for: the +// mirror of --exclude-prefix, so the honest pass it wants — one run per owning +// package — is one flag instead of one exclusion for each package the change +// happens to touch. +// +// Asserted through the two subcommands rather than through ditto.Prefixes, +// because the wiring between them is the part a unit test of the plan cannot +// see, and it is the part that would silently do nothing. +func TestIncludePrefixReachesThePlan(t *testing.T) { + t.Run("changed", func(t *testing.T) { + dir := dittotesting.GitRepository(t) + + dittotesting.WriteFile(t, dir, "web/web.go", "package web\n\nfunc Web(a, b int) bool { return a > b }\n") + dittotesting.WriteFile(t, dir, "tools/tool.go", "package tools\n\nfunc Tool(a, b int) bool { return a > b }\n") + dittotesting.Git(t, dir, "add", "-A") + dittotesting.Git(t, dir, "commit", "-m", "two packages") + + out := &bytes.Buffer{} + err := changedCommand([]string{"--since", "base", "--dry", "--cwd", dir, "--include-prefix", "web/"}, out) + + require.NoError(t, err) + assert.Contains(t, out.String(), "web/web.go") + assert.NotContains(t, out.String(), "tools/tool.go") + }) + + t.Run("staged", func(t *testing.T) { + dir := dittotesting.GitRepository(t) + + dittotesting.WriteFile(t, dir, "web/web.go", "package web\n\nfunc Web(a, b int) bool { return a > b }\n") + dittotesting.WriteFile(t, dir, "tools/tool.go", "package tools\n\nfunc Tool(a, b int) bool { return a > b }\n") + dittotesting.Git(t, dir, "add", "-A") + + out := &bytes.Buffer{} + err := stagedCommand([]string{"--cwd", dir, "--dry", "--include-prefix", "web/"}, out) + + require.NoError(t, err) + assert.Contains(t, out.String(), "web/web.go") + assert.NotContains(t, out.String(), "tools/tool.go") + }) +} + // TestChangedRefusesToGuessABase covers the one decision `changed` deliberately // does not make for you. There is no default that is right on a CI checkout, in // a working tree and on a branch at once, and a base guessed wrong is either a diff --git a/internal/staged/changed.go b/internal/staged/changed.go index 2f32320..5b17212 100644 --- a/internal/staged/changed.go +++ b/internal/staged/changed.go @@ -16,14 +16,15 @@ import ( // of a different pair of trees. // ChangedFiles lists the Go sources a range touched, on the same terms Files -// uses: not tests, and not anything under an excluded prefix. -func (r *Repository) ChangedFiles(baseRef string, excludedPrefixes []string) ([]string, error) { +// uses: not tests, not anything under an excluded prefix, and, when the caller +// named prefixes to include, only what they cover. +func (r *Repository) ChangedFiles(baseRef string, excludedPrefixes, includedPrefixes []string) ([]string, error) { output, err := r.git("diff", "--name-only", "--diff-filter=ACMR", "-z", rangeOf(baseRef)) if err != nil { return nil, fmt.Errorf("listing the files changed since %s: %w", baseRef, err) } - return selectMutable(splitNUL(output), excludedPrefixes), nil + return selectMutable(splitNUL(output), excludedPrefixes, includedPrefixes), nil } // ChangedScopeOf converts each file's range diff into byte ranges of HEAD. diff --git a/internal/staged/changed_internal_test.go b/internal/staged/changed_internal_test.go index 51122ec..86770e1 100644 --- a/internal/staged/changed_internal_test.go +++ b/internal/staged/changed_internal_test.go @@ -45,7 +45,7 @@ func TestChangedFilesReadsARangeRatherThanTheIndex(t *testing.T) { "diff --name-only": "internal/thing/thing.go\x00internal/thing/thing_test.go\x00readme.md\x00", }) - files, err := repository.ChangedFiles("v0.7.0", []string{"testdata/"}) + files, err := repository.ChangedFiles("v0.7.0", []string{"testdata/"}, nil) if err != nil { t.Fatalf("listing the changed files: %v", err) } @@ -198,7 +198,7 @@ func TestChangedFilesSurvivesAnEmptyRange(t *testing.T) { repository, _ := scripted(map[string]string{"diff --name-only": ""}) - files, err := repository.ChangedFiles("v0.7.0", nil) + files, err := repository.ChangedFiles("v0.7.0", nil, nil) if err != nil { t.Fatalf("listing the changed files: %v", err) } diff --git a/internal/staged/staged.go b/internal/staged/staged.go index 035ccec..fde4a9e 100644 --- a/internal/staged/staged.go +++ b/internal/staged/staged.go @@ -130,23 +130,24 @@ func (r *Repository) git(args ...string) ([]byte, error) { } // Files lists the staged Go sources worth mutating: not tests, because they are -// the oracle rather than the subject, and not anything under a prefix the caller -// excluded. -func (r *Repository) Files(excludedPrefixes []string) ([]string, error) { +// the oracle rather than the subject, not anything under a prefix the caller +// excluded, and, when the caller named prefixes to include, only what they +// cover. +func (r *Repository) Files(excludedPrefixes, includedPrefixes []string) ([]string, error) { output, err := r.git("diff", "--cached", "--name-only", "--diff-filter=ACMR", "-z") if err != nil { return nil, fmt.Errorf("listing the staged files: %w", err) } - return selectMutable(splitNUL(output), excludedPrefixes), nil + return selectMutable(splitNUL(output), excludedPrefixes, includedPrefixes), nil } -func selectMutable(files, excludedPrefixes []string) []string { +func selectMutable(files, excludedPrefixes, includedPrefixes []string) []string { selected := []string{} for _, file := range files { file = strings.ReplaceAll(file, "\\", "/") - if file != "" && isMutable(file, excludedPrefixes) { + if file != "" && isMutable(file, excludedPrefixes, includedPrefixes) { selected = append(selected, file) } } @@ -154,11 +155,15 @@ func selectMutable(files, excludedPrefixes []string) []string { return selected } -func isMutable(file string, excludedPrefixes []string) bool { +func isMutable(file string, excludedPrefixes, includedPrefixes []string) bool { if !strings.HasSuffix(file, ".go") || strings.HasSuffix(file, "_test.go") { return false } + if !underAnyPrefix(file, includedPrefixes) { + return false + } + for _, prefix := range excludedPrefixes { if prefix != "" && strings.HasPrefix(file, prefix) { return false @@ -168,6 +173,32 @@ func isMutable(file string, excludedPrefixes []string) bool { return true } +// underAnyPrefix answers the include half, and answers true when the caller +// named nothing to include: an empty list is no narrowing. +// +// An empty prefix is ignored rather than honoured, exactly as it is on the +// exclusion side. Every path starts with the empty string, so a list that +// carries one empty entry means "no filter" when it was built -- and "nothing" +// when it was not, which is the reading that would silently mutate no files at +// all. +func underAnyPrefix(file string, prefixes []string) bool { + named := false + + for _, prefix := range prefixes { + if prefix == "" { + continue + } + + named = true + + if strings.HasPrefix(file, prefix) { + return true + } + } + + return !named +} + // RejectPartial refuses a staged file that also has unstaged edits. // // The scope is derived from the index and the mutants are written into a copy of diff --git a/internal/staged/staged_internal_test.go b/internal/staged/staged_internal_test.go index f6b68fa..f9ca011 100644 --- a/internal/staged/staged_internal_test.go +++ b/internal/staged/staged_internal_test.go @@ -96,7 +96,7 @@ func TestSelectMutableKeepsProductionSourcesOnly(t *testing.T) { "", } - selected := selectMutable(files, []string{"tools/"}) + selected := selectMutable(files, []string{"tools/"}, nil) want := []string{"internal/calc/calc.go"} if !reflect.DeepEqual(selected, want) { @@ -109,12 +109,45 @@ func TestSelectMutableKeepsProductionSourcesOnly(t *testing.T) { func TestSelectMutableIgnoresAnEmptyPrefix(t *testing.T) { t.Parallel() - selected := selectMutable([]string{"a.go"}, []string{""}) + selected := selectMutable([]string{"a.go"}, []string{""}, nil) if len(selected) != 1 { t.Fatalf("selected = %v, want a.go kept", selected) } } +// The include half, which is the mirror of the exclusion above and the reason +// the pair exists: a scope that holds more packages than the test command can +// compile is narrowed to the package the command names. +func TestSelectMutableKeepsOnlyIncludedPrefixes(t *testing.T) { + t.Parallel() + + files := []string{"internal/desktop/app.go", "internal/observability/syncdiag/reader.go", "README.md"} + + selected := selectMutable(files, nil, []string{"internal/observability/"}) + + want := []string{"internal/observability/syncdiag/reader.go"} + if !reflect.DeepEqual(selected, want) { + t.Fatalf("selected = %v, want %v", selected, want) + } +} + +// An empty include list is no narrowing at all -- the same reading as the +// exclusion side, and the one that keeps every caller that never used the flag +// mutating what it always did. A list that carries only an empty entry is the +// same list: every path starts with the empty string, so honouring it would +// mutate nothing at all. +func TestSelectMutableWithoutIncludesKeepsEverything(t *testing.T) { + t.Parallel() + + files := []string{"a.go", "internal/b.go"} + + for _, included := range [][]string{nil, {""}} { + if selected := selectMutable(files, nil, included); !reflect.DeepEqual(selected, files) { + t.Fatalf("selected = %v for includes %v, want every file", selected, included) + } + } +} + func TestWithoutGitEnvironmentRemovesTheInheritedAddressing(t *testing.T) { t.Parallel() diff --git a/perf/baseline.json b/perf/baseline.json index 9479fe6..7d294d9 100644 --- a/perf/baseline.json +++ b/perf/baseline.json @@ -17,7 +17,7 @@ "laboratoryRunsForOneChangedFunction": 4, "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": 8, "testCommandInvocationsPerReleaseWholeFixture": 49, - "mutantsPerReleaseOnThisRepository": 946 + "mutantsPerReleaseOnThisRepository": 949 }, "targets": { "sourceParsesPerReleaseWithThreeViruses": "Reached: 4, one parse per source file, down from 12. GoSourceFile.Incubate now takes the whole mutator set and parses once for all of them. With the default 14 mutators this is 14 parses per file reduced to 1.", @@ -28,6 +28,6 @@ "laboratoryRunsForOneChangedFunction": "4, the mutators that fire on one changed line and nothing else in the repository. This is what WithChangedRanges buys: without it the same fixture charges 48.", "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": "8, exactly twice the single-file number. The two ranges name offsets that exist in both files, because every fixture file has the same byte layout, so a scope holding one flat set of ranges would charge 16 and grow as the square of the file count. Keeping the ranges beside their file makes that impossible rather than merely unlikely.", "testCommandInvocationsPerReleaseWholeFixture": "49 for the 48 mutants of the whole fixture, plus one. That one is the baseline: the laboratory runs the suite once on unmutated code before scoring anything, because a test command that fails before it compiles fails for every mutant too, and ditto recognises a killed mutant by exactly that. Measured on ditto's own gate before the guard existed: 431 of 431 killed in 5.46 seconds, a perfect score for a run that compiled nothing. Every other laboratory counter here goes through a stand-in and cannot see a run the laboratory makes on its own, which is why this one exists — a cost nobody records is one that grows unnoticed, the mirror of the unrecorded gain this file already refuses. It must not grow: one baseline per release, never one per mutant.", - "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once. Then 946 (+41) for the same separation from the report outwards -- the score leaves those mutants out, names the packages that hold them, and the command exits 3 -- attributed per file in the committed state: internal/consolereporter/consolereporter.go +15, internal/ditto/ditto.go +11, internal/dittotesting/fakereporter/fakereporter.go +9, internal/verbosereporter/verbosereporter.go +2, internal/gatedreporter/gatedreporter.go +2 and cmd/ditto/main.go +2. The nine product files that commit touches sum to +51 and not to the +41 the ratchet reported; the difference is release.go's own +10, which this counter never sees because it carries the gate's exclusions and the gate skips release.go. Both numbers are right and they answer different questions: this counter speaks for the scope the mutation gate runs against, and the per-file sum speaks for the change. A first measurement of the same change said +59, and the eight it lost are the two loops of the reporter's summary folded into one pass -- a gain this file is required to record rather than hand back. docs/reports/ditto-mutation-scope.md." + "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once. Then 946 (+41) for the same separation from the report outwards -- the score leaves those mutants out, names the packages that hold them, and the command exits 3 -- attributed per file in the committed state: internal/consolereporter/consolereporter.go +15, internal/ditto/ditto.go +11, internal/dittotesting/fakereporter/fakereporter.go +9, internal/verbosereporter/verbosereporter.go +2, internal/gatedreporter/gatedreporter.go +2 and cmd/ditto/main.go +2. The nine product files that commit touches sum to +51 and not to the +41 the ratchet reported; the difference is release.go's own +10, which this counter never sees because it carries the gate's exclusions and the gate skips release.go. Both numbers are right and they answer different questions: this counter speaks for the scope the mutation gate runs against, and the per-file sum speaks for the change. A first measurement of the same change said +59, and the eight it lost are the two loops of the reporter's summary folded into one pass -- a gain this file is required to record rather than hand back. Then 949 (+3) for --include-prefix, the mirror of --exclude-prefix that the same report asked for: the four product files it touches are internal/staged/staged.go, internal/staged/changed.go, staged.go and changed.go, and the whole delta is three mutants because the filter is one more question asked of a predicate that already existed. It is not attributed per file here: the change is one predicate threaded through four files, and splitting +3 across them by arithmetic would be a number invented rather than measured. What it buys is the honest pass the report wanted -- one run per owning package, in one flag, without rewriting the test command per run -- and the shape it changes is exported: PlanStaged, RunStaged, PlanChanged and RunChanged take a ditto.Prefixes instead of one []string, so a change of scope and a narrowing of it are one argument rather than two that can drift. docs/reports/ditto-mutation-scope.md." } } diff --git a/staged.go b/staged.go index c2d94b7..670f1f4 100644 --- a/staged.go +++ b/staged.go @@ -7,6 +7,26 @@ import ( "github.com/Disble/ditto/internal/staged" ) +// Prefixes narrows which files of a change a scoped run is worth asking about. +// +// It is the pair the two scoped subcommands expose as --exclude-prefix and +// --include-prefix, and it is a type rather than two slice arguments because the +// two are one decision: what this run is about. Empty is no narrowing at all, +// which is what both entry points did before this existed. +// +// Include is the one to reach for when a release refuses a scope its test +// command cannot compile: narrowing the run to the package the command names is +// cheaper and more precise than widening the command, and it is one flag instead +// of one per package the change happens to touch. +type Prefixes struct { + // Exclude names repository-relative prefixes never worth mutating. + Exclude []string + // Include names the only repository-relative prefixes worth mutating. Empty + // means every prefix, so a run with no Include mutates everything Exclude + // left. + Include []string +} + // StagedPlan is what a staged change justifies mutating, before anything runs. type StagedPlan struct { // Root is the repository the plan was read from. @@ -53,13 +73,13 @@ func (p StagedPlan) Mutable() bool { return len(p.Files) > 0 } // asking on its own, and answering it must not write a sandbox or start a suite. // It therefore does not read `.ditto.json` either -- that names what a sandbox // needs, and this builds none. See RunStaged. -func PlanStaged(directory string, excludePrefixes []string) (StagedPlan, error) { +func PlanStaged(directory string, prefixes Prefixes) (StagedPlan, error) { repository, err := staged.New(staged.OSRunner{}, directory) if err != nil { return StagedPlan{}, fmt.Errorf("reading the repository: %w", err) } - files, err := repository.Files(excludePrefixes) + files, err := repository.Files(prefixes.Exclude, prefixes.Include) if err != nil { return StagedPlan{}, fmt.Errorf("reading the staged files: %w", err) } @@ -104,8 +124,8 @@ func PlanStaged(directory string, excludePrefixes []string) (StagedPlan, error) // // Options are applied after the scope and the root, so a caller can set a // threshold or a test command but cannot quietly point the run somewhere else. -func RunStaged(directory string, excludePrefixes []string, options ...Option) error { - plan, err := PlanStaged(directory, excludePrefixes) +func RunStaged(directory string, prefixes Prefixes, options ...Option) error { + plan, err := PlanStaged(directory, prefixes) if err != nil { return err } diff --git a/staged_mutation_test.go b/staged_mutation_test.go index a5a5ac2..ff9fb49 100644 --- a/staged_mutation_test.go +++ b/staged_mutation_test.go @@ -24,7 +24,7 @@ import ( // It skips when nothing is staged, because a scope of nothing is not a failure // -- it is a commit that changed no Go source, and there is nothing to judge. func TestStagedMutation(t *testing.T) { - plan, err := ditto.PlanStaged(".", []string{"testdata/"}) + plan, err := ditto.PlanStaged(".", ditto.Prefixes{Exclude: []string{"testdata/"}}) if err != nil { t.Fatalf("reading the staged change: %v", err) } @@ -44,7 +44,7 @@ func TestStagedMutation(t *testing.T) { // the worktree, with one tracked file left dirty and unstaged, seven of // eight verdicts moved. Scoping correctly and then measuring the wrong bytes // would be the same defect wearing the fix's clothes. - if err := ditto.RunStaged(".", []string{"testdata/"}, + if err := ditto.RunStaged(".", ditto.Prefixes{Exclude: []string{"testdata/"}}, ditto.ForceColors(), ditto.WithTestCommand(makeCommand(t)+" test.failfast MAKEFLAGS="), ditto.WithMinimumThreshold(0.5), From 82d6cf0a6faf3ca9c473f6e013253f203a56d21c Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 20:19:43 -0500 Subject: [PATCH 7/9] docs: record what the scope check does, and what it deliberately does not The readme gains the section the report asked for, with the real box: what a run prints when its command cannot compile part of the scope, why each of the three decisions is there -- out of both sides, not run, exit 3 -- and the two limits where ditto says nothing rather than guessing. docs/metrics.md gains the classification row, because that table is where this repository decides what a verdict means, and this is the first row whose fix is not in the tests. docs/backlog.md gains entries 28 and 29: the two ways the check stays silent, and per-package runs as a cost-model change with the measurement that would have to argue for it. The learning log gains four lines, three of which are things that cost a measurement to find: a name comparison would have failed correct runs, result.Output answered only for a kill, and asking a question that pays for the baseline moves a line in the report. docs/reports/ditto-mutation-scope.md is committed with the release it produced, in the reporting repository's words rather than paraphrased. --- CHANGELOG.md | 122 ++++++++++++++++++++ docs/backlog.md | 55 +++++++++ docs/learning-log.md | 4 + docs/metrics.md | 10 ++ docs/reports/ditto-mutation-scope.md | 164 +++++++++++++++++++++++++++ readme.md | 74 ++++++++++-- 6 files changed, 419 insertions(+), 10 deletions(-) create mode 100644 docs/reports/ditto-mutation-scope.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 58207aa..4f70974 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,128 @@ All notable changes to this project are documented here. The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [0.12.0] - 2026-09-19 + +A release about one number that measured two things at once. Mutants come from a +change, they are judged by one configured test command, and when that command +cannot compile part of the scope the mutants it cannot reach were counted as +evidence about the tests. Reported from a consumer of ditto that had staged five +packages and named one of them: + + ditto staged --exclude-prefix frontend/ --threshold 0.80 \ + --test-command "go test -count=1 -timeout 120s -json ./internal/observability/syncdiag/" + + Total: 43 Killed: 23 Survived: 20 Score: 0.53 (minimum: 0.80) + +17 of those 20 survivors lived in `internal/desktop`, a package the command never +builds, so nothing it ran could have killed them: the run's ceiling was 0.605 and +the number presented as a verdict was measuring the tests and the scope together. +They are now named, left out of the score, not run at all, and failed on. +`docs/reports/ditto-mutation-scope.md` is the report that produced this release. + +### Breaking + +- **A scope holding mutants the test command cannot compile now fails with its +own exit code, 3.** Those mutants leave the numerator **and** the denominator, +exactly as a mutant that never compiled does, so the printed score is a real +measurement of what the command could judge rather than a mixture of two +questions. They are not run, because a guaranteed survivor bought with a full run +of the suite is not evidence about anybody's tests, and they are named with the +package that holds them: + + ┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┓ + ┃ • Total: 26 ┃ + ┃ • Killed: 23 ┃ + ┃ • Survived: 3 ┃ + ┃ • Unmeasured: 17 ┃ + ┠┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┨ + ┃ ⨯ Score: 0.88 (minimum: 0.80) ┃ + ┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┛ + ┃ 17 of the 43 mutants in this scope are never compiled by this test command, so + ┃ nothing it runs can kill them and they are out of the score entirely: + ┃ internal/desktop (17) + ┃ Name every package the scope mutates in --test-command, or narrow the scope. + + Exit 3 is the other half: a scope ditto cannot measure and a score below the bar + were both 1, and a gate has to be able to tell them apart, because no test + answers the first. The two levers are to name every package the scope mutates in + `--test-command`, or to narrow the scope — see `--include-prefix` below. + + **The check only speaks when the command reports itself.** It reads the packages + the command executes out of its own `go test -json` stream, so a `--test-command` + that is `make`, `gotestsum`, a wrapper script or `go test` without `-json` gets no + check at all, and keeps the behaviour it always had. It also runs per package and + not per file: a file behind a build tag for another operating system is inside a + package the command does compile, and its mutants stay ordinary survivors. Both + limits are recorded in `docs/backlog.md`, entry 28. + +- **`PlanStaged`, `RunStaged`, `PlanChanged` and `RunChanged` take a + `ditto.Prefixes`** where they took one `[]string`: what a run is about and what + it leaves out are one decision, so they are one argument instead of two that can + drift. Every existing call site is a one-line change: + `ditto.RunStaged(dir, ditto.Prefixes{Exclude: []string{"tools/"}})`. + +- **One line of output moved.** The baseline announcement prints before the first + file's announcement now, because asking the laboratory which packages the test + command can compile is what pays for the baseline. No verdict moved and every + mutant address is identical; the golden was updated deliberately, and says so + where someone will look. + +### Added + +- **`--include-prefix`, the mirror of `--exclude-prefix`,** on `staged` and + `changed`. It exists because the fix above makes it necessary: a run now refuses + a scope its command cannot compile, and the honest answer is usually to narrow + the run to the package the command names — one flag, where one exclusion per + other package the change happens to touch would do the same. `-h` names the + pairing, because the moment a reader needs it is the moment a run just refused + their scope. + +- **`ditto.Prefixes`,** the exported pair those flags set. + +- **A distinct exit code for an unmeasurable scope,** documented in `ditto -h` + beside the other two: 0 a score at or above the bar, 1 every other failure, 3 a + scope the test command cannot compile. + +### Changed + +- **`--test-command`'s help names the lever that exists.** It said "name the + package that owns the change instead", which is right only while the scope holds + one package: a command that names one of five leaves the other four unmeasurable, + and the run now says so and fails. It also names what keeping `-json` buys beyond + the reason to keep it. + +- **The report's box gained `• Unmeasured: N`,** printed only when there is + something to say, for the reason the non-viable line is: a line on every run is a + line people stop reading. An unmeasured mutant is never rendered as a survivor — + a diff of a program that never ran is not evidence about anything. + +- **The reporter's summary is one pass over its diagnostics instead of three.** + The exclusions now live in one place, in the order that is the rule: unmeasured, + then never-compiled, then killed or survived. + +### Documentation + +- `docs/reports/ditto-mutation-scope.md`, the report in the reporting + repository's words, committed with the release it produced. +- `docs/metrics.md` gains the classification row for it: out of both sides, named, + and the one row whose fix is not in the tests. +- `docs/backlog.md` entries 28 and 29: the two ways the check stays silent, and + per-package runs as a cost-model change rather than a feature to add quietly. + +### Performance + +- The scope costs **one `go list -deps -test` process per release**, measured at + 0.28–0.31 s over this repository's 64 executed packages, and it runs only for a + command that emitted a stream. It is not paid per mutant, and it is not paid at + all for a command ditto cannot read. + +- `mutantsPerReleaseOnThisRepository` moved 873 → 949 over the four commits that + added product code here: +31 for the scope, +1 for the laboratory answering, + +41 for the report and the exit code, and +3 for `--include-prefix`. Each is + attributed per file in `perf/baseline.json` rather than as one number at the + end, including the +8 the reporter's folded summary gave back. + ## [0.11.0] - 2026-09-19 ### Changed diff --git a/docs/backlog.md b/docs/backlog.md index 1ee79b3..fb96bd3 100644 --- a/docs/backlog.md +++ b/docs/backlog.md @@ -809,3 +809,58 @@ which is the difference between an obvious cost and an arguable one. It is a design question rather than a defect, and it is on this list because it now has a repository behind it instead of a hypothetical. + +## 28. The scope check only reads a command that reports itself + +**Measured 2026-09-19**, by building the binary and running it against a fixture +with three packages: one the command names, one its tests import, and one nothing +compiles. `docs/reports/ditto-mutation-scope.md` is the report it answers. + +A release now reads which packages the configured test command executes, out of +the command's own `go test -json` stream, and reports the mutants it cannot +compile instead of scoring them. Three limits come with that, and they are named +here so the silence is not mistaken for a pass: + +- **A command that emits no such stream gets no check at all.** `make`, + `gotestsum`, a wrapper script, or `go test` without `-json` — the stream is the + only truthful answer available without parsing the command string, which + `options.go`'s command-scope table already refuses to do on purpose. The check + therefore says nothing rather than guessing, and a run configured that way keeps + the behaviour it always had. +- **A gated run is not checked.** `Gated()` replaces the execution plan of the + exact module-scope command and nothing else, so the command that runs is + `go test -count=1 ./...` — every package the module has, which the gated path + already resolves for itself. Asking the ordinary laboratory would buy a second + suite run to learn what the gated path knows. +- **The check is per package, not per file.** A file behind a build tag for + another operating system is inside a package the command does execute, so its + mutants stay ordinary survivors — which they are: the command builds the + package, on a platform where that file is not part of it. + +What would close the first: a `--test-command` contract that carries the scope +explicitly, or a second subcommand that names the package set. Both are API +decisions rather than measurements, and neither has a repository asking for it +yet. + +## 29. One test command per owning package is a cost-model change + +**Proposed by the same report, 2026-09-19**, and deliberately not shipped with it. + +The report's third suggestion: group a scope's mutants by owning package and +derive one test command per group, so a five-package change does not need a +`--test-command` that names all five. It is the version of this that would make +the scope check unnecessary rather than enforceable. + +It is not shipped, and the reason is written in the report itself: ditto's `-h` +says the command runs once per mutant, changed to once per package would be a +different cost model presented as the same flag, and the measured gain is not +obvious — a package's suite is a fraction of the module's, but the fixed 750–950 +ms start of the test command is paid per invocation and would be paid per package +per mutant unless the compilation is shared too, which is what the gated path +already does for the exact module-scope command. + +What would close it: a measurement of a real staged run against a per-package +command set, with the counter that matters (`testCommandInvocationsPerRelease`) +beside the wall clock, and a verdict comparison showing the narrower command set +does not change a single mutant's answer. The scope check this entry's report did +buy makes that measurement cheaper to argue for and no less necessary. diff --git a/docs/learning-log.md b/docs/learning-log.md index 4ab320f..8e20bdf 100644 --- a/docs/learning-log.md +++ b/docs/learning-log.md @@ -58,3 +58,7 @@ file only explains the _why_; it never replaces the _how_. - [2026-09-18]: Timing the configured test command at its own process boundary showed it is 99.3% of a real staged run (shares 0.993109, 0.993183, 0.993131; non-test residual an upper-bounded 2.3–4.4 s), so with every viable mutant required to run the complete suite there is no measured precision-preserving headroom left inside ditto — the only drastic gain this branch had was running fewer tests, which was refused as a trade against accuracy (`docs/experiments/failfast-cost-ceiling.md`). - [2026-09-19]: An explicit two-worker Go scheduler proved exact overlap and byte-identical mutation answers with ample Windows memory headroom, yet reduced a real fail-fast staged run only 5.62% (ratio 0.943784 against the pre-registered 0.75 limit) because every overlapping mutant command became slower — dynamic admission would regulate a lever that did not earn its production complexity (`docs/experiments/explicit-adaptive-scheduler-poc.md`). - [2026-09-19]: A lint verdict can depend on the platform the linter runs on — `unparam` flagged `validateBatches`'s target-OS parameter on Linux ("goos always receives \"linux\"") and stayed silent on Windows, because the only varying call sites were tests passing the literal "linux" and on a Linux host that literal equals `runtime.GOOS`; `GOOS=linux golangci-lint run` reproduced the cloud failure locally, and a test that passes "windows" cleared it on the host that runs CI while adding the case-fold coverage the parameter exists for (`internal/gobuildrunner/module_scope.go`). +- [2026-09-19]: A scope check that compares the packages a test command names against the packages a scope mutates would have failed correct runs — a mutant in a package the command does not name IS killable by that named package's tests, because the dependency is compiled into its test binary — so the answer had to come from the toolchain's own closure (`go list -deps -test`) and `-test` is what makes a test-only import count. +- [2026-09-19]: `result.Output` answered only for an `Ok`, and the one run whose stream carries a command's package scope is the GREEN one, which is an `Err` — so a contract written for refusals quietly made the new question unanswerable, and widening it to "what the command printed, whichever way it ended" cost three lines. +- [2026-09-19]: Asking the laboratory which packages the command compiles is what pays for the baseline, so the baseline announcement moved above the first file's announcement — the golden caught it on the first run, the two lines were the whole diff, and the move was written down as deliberate in the test rather than absorbed by updating the file. +- [2026-09-19]: The same fixture with three packages — one named by the command, one imported by the named package's tests, one compiled by nothing — is the smallest case that separates all three classes, and it is what turned the report's 43-mutant example into a golden, an exit code and a control in one test. diff --git a/docs/metrics.md b/docs/metrics.md index 0dc0455..016531a 100644 --- a/docs/metrics.md +++ b/docs/metrics.md @@ -218,9 +218,19 @@ Settled by the literature, not chosen here. See | the mutant | treatment | why | | --- | --- | --- | | does not compile | out of the numerator **and** the denominator | the kill predicate is undefined for a program that does not exist | +| in a package the test command does not compile | out of the numerator **and** the denominator, named with its package, and the run fails with its own exit code (3) | the kill predicate is undefined for code no run can compile, and it is a defect of the *scope*, not of the tests: answered by naming every package the scope mutates or by narrowing the scope, and by no test at all | | timed out | **a kill**, reported as its own reason | unanimous across PIT, Stryker and Infection | | suspected equivalent | counted as **survived**, and the metric renamed to a stated lower bound | equivalence is undecidable | +The second row is the one this repository learned last, and it is the only row +whose fix is not in the tests. Measured on the report that produced it: a staged +change spanning five packages, judged by a command naming one of them, printed +**0.53 against a bar of 0.80** for a run whose ceiling was 0.605 — 17 of its 20 +survivors lived in a package the command never builds. A score that mixes "your +tests missed this" with "your command cannot see this" is not a weaker +measurement of the same thing; it is a measurement of two things, and nobody can +tell which. `docs/reports/ditto-mutation-scope.md`. + The canonical definition, Zhu, Hall & May, *ACM Computing Surveys* 29(4), 1997, Definition 3.1: **`S = D / (M − E)`** — dead mutants over all mutants minus the equivalent ones. The denominator has never been "every mutant generated". diff --git a/docs/reports/ditto-mutation-scope.md b/docs/reports/ditto-mutation-scope.md new file mode 100644 index 0000000..b16830b --- /dev/null +++ b/docs/reports/ditto-mutation-scope.md @@ -0,0 +1,164 @@ +# ditto: a mutation score that mixes "your tests missed this" with "your command cannot see this" + +- **Reported by**: autoreas-bridge (private repository, consumer of ditto) +- **ditto version**: `v0.11.0` (`ditto version`) +- **Toolchain**: `go1.27.0 windows/amd64`, Windows 11, MINGW64/Git Bash +- **Date**: 2026-09-19 + +## Summary + +`ditto staged` scopes mutants to the **staged production diff**, but scores them against a +single `--test-command`. When the staged diff spans more than one Go package and the test +command names only some of them, the mutants in the unnamed packages are structurally +unkillable, and the resulting number is reported with the same confidence as a measured score. + +We are not asking ditto to slice our work units. We are asking it not to score a situation it +can already recognize as unmeasurable. + +## What we ran + +The staged change touched five production packages: + +``` +internal/observability/syncdiag +internal/observability/readcap +internal/mcp/requestcapture +internal/desktop +internal/api/contracts +``` + +Command: + +```bash +ditto staged --exclude-prefix frontend/ --threshold 0.80 \ + --test-command "go test -count=1 -timeout 120s -json ./internal/observability/syncdiag/" +``` + +Result: + +``` +Total: 43 +Killed: 23 +Survived: 20 +Score: 0.53 (minimum: 0.80) +2 of the 45 mutants generated never compiled, and are out of the score entirely. +``` + +The survivor list ditto printed: + +``` +internal/desktop/app_defaults.go:164:25 -> Comparison Invert +internal/desktop/app_device_sync_diagnostics.go:14:22 -> Comparison Invert +internal/desktop/app_device_sync_diagnostics.go:21:9 -> Comparison Invert +internal/desktop/app_device_sync_diagnostics.go:24:56 -> Integer Decrement +internal/desktop/app_device_sync_diagnostics.go:24:56 -> Integer Increment +internal/desktop/app_device_sync_diagnostics.go:26:3 -> Range Break +internal/desktop/app_runtime_services.go:110:22 -> Comparison Invert +internal/desktop/app_runtime_services.go:110:43 -> Comparison Invert +internal/desktop/app_runtime_services.go:110:73 -> Comparison Invert +internal/desktop/app_runtime_services.go:114:40 -> Comparison Invert +internal/desktop/app_runtime_services.go:114:65 -> Comparison Invert +internal/desktop/app_runtime_services.go:110:5 -> Comparison Replace +internal/desktop/app_runtime_services.go:110:53 -> Comparison Replace +internal/desktop/app_runtime_services.go:110:5 -> Comparison Replace +internal/desktop/app_runtime_services.go:110:32 -> Comparison Replace +internal/desktop/app_runtime_services.go:114:30 -> Comparison Replace +internal/desktop/app_runtime_services.go:114:50 -> Comparison Replace +internal/observability/syncdiag/reader.go:132:5 -> Comparison Replace +internal/observability/syncdiag/reader.go:169:14 -> Integer Increment +internal/observability/syncdiag/reader.go:172:12 -> Comparison +``` + +**17 of the 20 survivors are in `internal/desktop/`, a package the supplied test command never +executes.** No test we could write inside `internal/observability/syncdiag` can kill them. Even +if we had killed every killable survivor that run could reach, the score could not have exceeded +roughly `26/43 = 0.605`. + +The same staged set, with every owning package named in the test command, then reported: + +``` +Total: 16 +Killed: 16 +Survived: 0 +Score: 1.00 (minimum: 0.80) +``` + +The two totals differ because the first run still had `internal/observability/syncdiag` staged +while the second had already committed it. That is the point: the *scope* is the index, not the +question the score appears to answer. + +## Why this misleads, rather than merely annoys + +The number looks like a coverage verdict. It is partly an arithmetic artifact of the scope, and +nothing in the output distinguishes the two: + +- A mutant that survived because a test asserted the wrong thing. +- A mutant that survived because the command cannot reach its package. + +Three natural responses, all wrong: + +1. Write tests for the unreachable mutants. Impossible inside the named package, and impossible + in general without widening the command. +2. Lower `--threshold`. Makes the unmeasurable number "pass". +3. Conclude the change is under-tested. It may be perfectly tested. + +The correct response — re-slice the work unit, or name every owning package in the command — is +exactly the one the output hides. + +This is the same error class as reporting `0` from a failed read: **a low score from an +unexecutable mutation set is not a measured score.** + +## What ditto already gets right + +Worth stating, because it is why this case stands out as a remaining hole rather than a pattern: + +- **Build failure is not scored.** Staging one package while a needed new file stayed out of the + index produced `[build failed]` and no score. That is the correct treatment for the dependency + case, and it is the honest failure mode. +- **Non-compiling mutants are excluded from the denominator**, and said so: "N mutants generated + never compiled, and are out of the score entirely". +- **The mutation scope is already printed per file** (`internal/desktop/app_runtime_services.go — + 11 mutants`), which is precisely the information needed to detect the mismatch. +- **The command's cost is measured and stated**: `baseline: the suite took 6.88s on unmutated + code, and every mutant runs it again.` +- `staged` materialising the index and announcing generated-path copies is an explicit, honest + model. The generated-path behaviour in `.ditto.json` is documented where a user will find it. + +So ditto already knows the mutants' files, the command, and the command's cost. It has everything +needed to notice that the command cannot execute the packages that own 17 of them. + +## Suggested behaviour, from cheap to expansive + +1. **Detect and declare (minimal).** Derive the owning package set from the mutated files, and + compare it with the `--test-command`. If some mutants live in packages the command does not + execute, say so explicitly instead of, or beside, the score: + + > 17 mutants are in packages this test command does not execute + > (`internal/desktop`). Their survival cannot be attributed to it. + + A distinct non-zero exit code for "unmeasurable" would let a wrapper tell that condition apart + from "below threshold". Today both are `1`. + +2. **Offer scope narrowing.** An `--include-prefix` mirror of `--exclude-prefix`, so one honest + pass per owning package is expressible without rewriting the command per run. + +3. **Per-package runs (opt-in).** Group mutants by owning package and derive one test command per + group. We are **not** asking for this as a default: the `-h` text is explicit that the command + runs once per mutant, and changing that cost model silently would be worse than the current + behaviour. As an opt-in mode it would be genuinely useful. + +## What is not ditto's jurisdiction + +Work-unit granularity is a repository policy question. Our own policy says "stage the production +change and run mutation testing against its owning package" — the same single-owner assumption +`ditto staged -h` makes when it advises "name the package that owns the change instead". + +Our defect was bundling five packages into one staged change. ditto cannot decide our slicing for +us, and we are not proposing that it try. The request is narrower: **when the tool can see that +the score is not measurable, it should not present it as though it were.** + +## Cost to us + +Three tool round-trips and roughly forty minutes spent on a number that was not measuring what it +appeared to measure, including one attempt to "fix" the score by staging packages one at a time, +which failed to build for an unrelated and correctly-reported reason. diff --git a/readme.md b/readme.md index 17f1cb0..827b135 100644 --- a/readme.md +++ b/readme.md @@ -104,11 +104,16 @@ score of 0.13 against 1.00 for the identical eight mutants of an identical file. Checking that the staged files themselves are clean does not cover it, because the file that moved them was never staged. -Policy stays with you: `--threshold`, `--test-command`, and `--exclude-prefix` -(repeatable) are yours to set, and Ditto has an opinion about none of them -beyond its defaults. +Policy stays with you: `--threshold`, `--test-command`, `--exclude-prefix` and +`--include-prefix` (the last two repeatable) are yours to set, and Ditto has an +opinion about none of them beyond its defaults. -**Name the package that owns the change in `--test-command`.** This is the one +The exit code is the one thing a wrapper has to read: **0** a score at or above +the bar, **1** every other failure including a refusal and a usage error, and +**3** a scope that holds mutants the test command does not compile — the case the +next section is about. + +**Name every package the scope mutates in `--test-command`.** This is the one default that will surprise you, so it is here rather than only in `-h`: the test command runs **once per mutant, sequentially**, so the default `./...` costs your whole suite times your mutant count. Reported from a repository whose suite takes @@ -119,14 +124,63 @@ twice, and was read as a hang. It was not stuck. It was paying that bill. ditto staged --test-command "go test -count=1 -json ./internal/thepackage/" ``` -Seconds instead. `--exclude-prefix` and a `--threshold` below 1.00 are the other -two levers, and the sandbox strategy is not one — `--sandbox hardlink` buys back -a fixed fifteen seconds on a two-thousand-file repository, and `--sandbox link` -cannot work at all in a repository with a `go:embed` directive, because embed -refuses an irregular file. +Seconds instead. `--exclude-prefix`, `--include-prefix` and a `--threshold` below +1.00 are the other levers, and the sandbox strategy is not one — `--sandbox +hardlink` buys back a fixed fifteen seconds on a two-thousand-file repository, +and `--sandbox link` cannot work at all in a repository with a `go:embed` +directive, because embed refuses an irregular file. Keep `-json`. It is what lets ditto tell a mutant that never compiled from one a -test caught; without it, the first is counted as the second. +test caught; without it, the first is counted as the second. It is also what lets +ditto read which packages your command executes, which is the next section. + +### When the command cannot see part of the scope + +A command that names one package is the right command for a change that touches +one package. When the same command meets a change that touches five, the four it +does not name are not badly tested — they are **unmeasured**: nothing the command +runs can compile them, so every mutant in them survives, and a score counting +them mixes "your tests missed this" with "your command cannot see this". + +Reported from a consumer of ditto, which had done exactly that and read 0.53 as a +verdict: + +``` +┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┓ +┃ • Total: 26 ┃ +┃ • Killed: 23 ┃ +┃ • Survived: 3 ┃ +┃ • Unmeasured: 17 ┃ +┠┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┄┨ +┃ ⨯ Score: 0.88 (minimum: 0.80) ┃ +┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┛ +┃ 17 of the 43 mutants in this scope are never compiled by this test command, so +┃ nothing it runs can kill them and they are out of the score entirely: +┃ internal/desktop (17) +┃ Name every package the scope mutates in --test-command, or narrow the scope. +``` + +Three things are happening there, and each is a decision. The mutants the command +cannot compile leave the numerator **and** the denominator, exactly as a mutant +that never compiled does — the score above is a real measurement of the 26 that +could be judged, not of 43. They are not run at all, because a guaranteed +survivor bought with a full suite run is not evidence about anybody's tests. And +the run exits **3**, so a gate can tell "unmeasurable scope" from "below the bar": +the second is answered by testing more, and the first by naming every package the +scope mutates — or, when the change really is five packages wide, by running one +honest pass per package with a narrower scope: + +```shell +ditto staged --include-prefix internal/thepackage/ \ + --test-command "go test -count=1 -json ./internal/thepackage/" +``` + +Two limits are worth knowing, because they are where this says nothing rather +than guessing. The check reads the packages your command reports executing, which +means a `--test-command` that is not `go test -json` — `make`, `gotestsum`, a +wrapper script — gets no check at all. And it works per package, not per file: a +file behind a build tag for another operating system is inside a package the +command does execute, and its mutants stay ordinary survivors. ### In CI, where nothing is staged From ae977a75d6757a7fb142a9cb08f23badd9c20f3b Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 20:20:39 -0500 Subject: [PATCH 8/9] style(scope): carry the blank line the linter's own fix adds MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit .golangci.yml sets fix: true, so the pre-commit gate writes wsl's whitespace fix into the working tree on every run — the file is never committed dirty, but it stays dirty after every commit until the fix is carried. One line, and this is the commit that carries it rather than the next one rediscovering it. --- internal/commandscope/commandscope_test.go | 1 + 1 file changed, 1 insertion(+) diff --git a/internal/commandscope/commandscope_test.go b/internal/commandscope/commandscope_test.go index 2a5611c..bcb6d53 100644 --- a/internal/commandscope/commandscope_test.go +++ b/internal/commandscope/commandscope_test.go @@ -22,6 +22,7 @@ type scriptedToolchain struct { func (s *scriptedToolchain) Output(_, name string, args ...string) ([]byte, error) { s.asked++ + s.lastRun = append([]string{name}, args...) return s.output, s.err From 04501624dfc37112615c372947da20459f15a7b4 Mon Sep 17 00:00:00 2001 From: disble Date: Sat, 19 Sep 2026 20:44:22 -0500 Subject: [PATCH 9/9] refactor(laboratory): delete a guard no test could distinguish MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The MUTATE pass over this branch removed the nil check Executes made on its scope and every test still passed, which is what that check was worth: verifyBaseline always produces a scope — commandscope.New answers for an empty stream by declining to refuse anything — so the branch was unreachable, and a guard whose removal is invisible is dead weight wearing defense as a costume. The fail-open it looked like it was doing lives one layer down, where a test holds it: the scope answers true when the command named no package. Deleting this one is what makes that the only place the rule lives. mutantsPerReleaseOnThisRepository 949 -> 948, and the ratchet asked for the deletion to be recorded rather than kept as an unclaimed gain. --- internal/laboratory/laboratory.go | 10 ++++++---- perf/baseline.json | 4 ++-- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/internal/laboratory/laboratory.go b/internal/laboratory/laboratory.go index 9c7e5fe..4804054 100644 --- a/internal/laboratory/laboratory.go +++ b/internal/laboratory/laboratory.go @@ -109,16 +109,18 @@ func (l *Laboratory) Test( // // The first call is what pays for the baseline on a release that never asks // otherwise. There is no second cost: the same sandbox and the same once. +// +// There is no nil check on the scope, and that is deliberate rather than an +// omission. The baseline always produces one — commandscope.New answers for an +// empty stream by declining to refuse anything — so a guard here would be +// unreachable, and a guard no test can distinguish from its absence is dead +// weight pretending to be defense. Measured: deleting it changed no test. func (l *Laboratory) Executes(repository ditto.Repository, relativePath string) bool { sandbox := l.acquire(repository) defer l.returnToPool(sandbox) l.verifyBaseline(sandbox) - if l.scope == nil { - return true - } - return l.scope.Executes(relativePath) } diff --git a/perf/baseline.json b/perf/baseline.json index 7d294d9..6137d6a 100644 --- a/perf/baseline.json +++ b/perf/baseline.json @@ -17,7 +17,7 @@ "laboratoryRunsForOneChangedFunction": 4, "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": 8, "testCommandInvocationsPerReleaseWholeFixture": 49, - "mutantsPerReleaseOnThisRepository": 949 + "mutantsPerReleaseOnThisRepository": 948 }, "targets": { "sourceParsesPerReleaseWithThreeViruses": "Reached: 4, one parse per source file, down from 12. GoSourceFile.Incubate now takes the whole mutator set and parses once for all of them. With the default 14 mutators this is 14 parses per file reduced to 1.", @@ -28,6 +28,6 @@ "laboratoryRunsForOneChangedFunction": "4, the mutators that fire on one changed line and nothing else in the repository. This is what WithChangedRanges buys: without it the same fixture charges 48.", "laboratoryRunsForOneChangedFunctionInEachOfTwoFiles": "8, exactly twice the single-file number. The two ranges name offsets that exist in both files, because every fixture file has the same byte layout, so a scope holding one flat set of ranges would charge 16 and grow as the square of the file count. Keeping the ranges beside their file makes that impossible rather than merely unlikely.", "testCommandInvocationsPerReleaseWholeFixture": "49 for the 48 mutants of the whole fixture, plus one. That one is the baseline: the laboratory runs the suite once on unmutated code before scoring anything, because a test command that fails before it compiles fails for every mutant too, and ditto recognises a killed mutant by exactly that. Measured on ditto's own gate before the guard existed: 431 of 431 killed in 5.46 seconds, a perfect score for a run that compiled nothing. Every other laboratory counter here goes through a stand-in and cannot see a run the laboratory makes on its own, which is why this one exists — a cost nobody records is one that grows unnoticed, the mirror of the unrecorded gain this file already refuses. It must not grow: one baseline per release, never one per mutant.", - "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once. Then 946 (+41) for the same separation from the report outwards -- the score leaves those mutants out, names the packages that hold them, and the command exits 3 -- attributed per file in the committed state: internal/consolereporter/consolereporter.go +15, internal/ditto/ditto.go +11, internal/dittotesting/fakereporter/fakereporter.go +9, internal/verbosereporter/verbosereporter.go +2, internal/gatedreporter/gatedreporter.go +2 and cmd/ditto/main.go +2. The nine product files that commit touches sum to +51 and not to the +41 the ratchet reported; the difference is release.go's own +10, which this counter never sees because it carries the gate's exclusions and the gate skips release.go. Both numbers are right and they answer different questions: this counter speaks for the scope the mutation gate runs against, and the per-file sum speaks for the change. A first measurement of the same change said +59, and the eight it lost are the two loops of the reporter's summary folded into one pass -- a gain this file is required to record rather than hand back. Then 949 (+3) for --include-prefix, the mirror of --exclude-prefix that the same report asked for: the four product files it touches are internal/staged/staged.go, internal/staged/changed.go, staged.go and changed.go, and the whole delta is three mutants because the filter is one more question asked of a predicate that already existed. It is not attributed per file here: the change is one predicate threaded through four files, and splitting +3 across them by arithmetic would be a number invented rather than measured. What it buys is the honest pass the report wanted -- one run per owning package, in one flag, without rewriting the test command per run -- and the shape it changes is exported: PlanStaged, RunStaged, PlanChanged and RunChanged take a ditto.Prefixes instead of one []string, so a change of scope and a narrowing of it are one argument rather than two that can drift. docs/reports/ditto-mutation-scope.md." + "mutantsPerReleaseOnThisRepository": "727, what one full run of the gate has to pay for. It was 660 before internal/verdict, and 684 before verdict.Text and the wiring, all on 2026-08-28; the ratchet fired on that commit and the number was written down rather than discovered later. Every other counter here is measured against the six-file fixture, which does not change when the repository does -- so all eight stayed green while the real cost grew from 431 mutants to 660 between 2026-08-15 and 2026-08-28 and pushed the gate past its 30-minute timeout. Measured on the two CI logs: cost per mutant moved 2.269s to 2.435s, seven percent, while the count moved forty-three; and every new mutant belonged to a file that did not exist in August -- internal/staged (133), cmd/ditto/main.go (59), staged.go (21), internal/filecopy (9). No old file grew. This counter is what perfbench's own doc comment already promised in prose: 'how many mutants a scope produces -- these are integers, identical on every machine, and a change in one is always meaningful.' It was the one number in that sentence nothing measured. It grows when code is added, and that is the point: the growth is real cost and has to be written down rather than discovered by a timeout. Moved to 734 on 2026-08-30 by backlog entries 22-26: +4 for the command line (the version subcommand, staged gating and the option builder they share) +3 for saying what a run is doing (internal/progresslaboratory and the baseline announcement), and +2 for confirming assertion kills (internal/confirminglaboratory). Each part was measured as it landed rather than one number attributed to the whole change afterwards. Then 736 to 783 (+47) for the range scope that closes backlog 21 -- changed.go, internal/staged/changed.go and the changed subcommand. That the change which SHRINKS the gate grows this counter by 47 is not a contradiction, it is the entry: this number is what the repository-sized question costs, and the gate no longer asks it on every push. Then 784, a net +1, when RunStaged and RunChanged were folded onto one runInSandbox and internal/dittotesting/gitrepository.go was added. The split between what the fold removed and what the new file added was not measured separately, so it is not claimed. Then 785 (+1) for resolving git to an absolute path rather than letting PATH decide, which SonarQube refused as a security rating on new code. Then 789 (+4) for telling an empty scope apart from a failing one: the reporter now counts what it scored and two decorators forward that count. Scoped exactly like ditto_mutation_test.go, so it speaks for the run it is named after. Counting costs no test command and no sandbox. docs/experiments/counting-the-real-repository.md. Then 813 (+24) for the module-scope core, measured per file rather than attributed to the change as a whole: internal/gobuildrunner/module_scope.go is new and contributes 23, internal/gatedlaboratory/gatedlaboratory.go contributes 1, and options.go contributes none -- the command-scope table replaced a branchy classifier with the same number of mutable sites. The +1 was measured by counting that file before this change and after it, not inferred from the total; the three numbers sum to the 24 the ratchet reported. This growth is produced code the run now pays for, and the module-scope path is what brings the run's cost down. The two are kept apart on purpose: this counter speaks for how many mutants a scope produces, and says nothing about what judging them costs. Then 818 (+5) for carrying the verdict reason onto the module path: a package test binary cannot emit `go test -json`, so `go tool test2json` converts a failing package's output and a module-path kill reports assertion instead of unknown. The file that grew is internal/gobuildrunner/module_scope.go, 23 before and 28 after, measured on that file alone and matching the reported total. Then 846 (+28) for the observability closure, attributed per file: internal/gobuildrunner/module_scope.go grew from 28 to 54 (+26) and internal/gatedlaboratory/gatedlaboratory.go from 37 to 39 (+2). The two sum to the 28 the ratchet reported. This is the change that divides the module path's selections x packages factor; it costs produced code and buys executions, and the two are counted by different instruments on purpose. Then 850 (+4) for reusing one sandbox and one compilation directory across the batches of a release: internal/gatedlaboratory/gatedlaboratory.go grew from 39 to 42 and internal/gobuildrunner/module_scope.go from 54 to 55, summing to the 4 the ratchet reported. Measured payoff on a ten-package module: the gated ratio fell from 0.4005-0.4040 to 0.2230-0.2312, because the paths stopped changing between batches and the toolchain's up-to-date check could finally fire. The directory alone had already been built and reverted for paying nothing; the sandbox is what makes the pair work, and that order was measured rather than argued. Then 873 (+23) for letting a module whose packages collide compile at all: the same session counted the unchanged tree at 850 and this tree at 873, both on .git-free disposable copies, and the whole delta is internal/gobuildrunner/module_scope.go -- internal/perfbench counts product .go files only (internal/fsrepository/fsrepository.go skips _test.go), so the two changed test files contribute zero mutants. That file was recorded at 55 at the 850 step, which makes it 78 by arithmetic (55 plus the measured 23); no per-file counter exists, so 78 is derived, not separately measured. One correction belongs here rather than being smoothed away: an earlier state of this same change counted 874, and the extraction that brought prepare's cyclomatic complexity inside the gate's cyclop limit moved the count by one. The number written down is the one the gate itself counted on the committed tree, because that is the tree the ratchet speaks for. The defect the change fixes -- test binaries named from path.Base colliding across packages, so module-scope gating refused the whole repository -- and the end-to-end evidence through the shipped binary live in docs/performance-core-log.md, entry 022. Then 904 (+31) for knowing which packages a test command actually executes: the whole delta is internal/commandscope/commandscope.go, measured per file with this same counter at 31 mutants in that file alone, and internal/result/result.go contributes zero on both sides of the widening it took for the scope to be readable. The ratchet reported the same 31, so nothing here is attributed that was not measured. What that file buys is the number the report that produced it could not trust: a release now separates the mutants its command can kill from the ones its command never builds, instead of printing one ratio over both. docs/reports/ditto-mutation-scope.md. Then 905 (+1) for asking that scope the one question the report cannot answer for itself: internal/laboratory/laboratory.go is the only product file that commit touches, and internal/perfbench counts product files only, so the +1 is that file's and needs no per-file instrument to place. It buys Executes, which answers whether the configured command compiles the package that owns a file, and it costs no additional suite run: the baseline that already prints the stream is the same once. Then 946 (+41) for the same separation from the report outwards -- the score leaves those mutants out, names the packages that hold them, and the command exits 3 -- attributed per file in the committed state: internal/consolereporter/consolereporter.go +15, internal/ditto/ditto.go +11, internal/dittotesting/fakereporter/fakereporter.go +9, internal/verbosereporter/verbosereporter.go +2, internal/gatedreporter/gatedreporter.go +2 and cmd/ditto/main.go +2. The nine product files that commit touches sum to +51 and not to the +41 the ratchet reported; the difference is release.go's own +10, which this counter never sees because it carries the gate's exclusions and the gate skips release.go. Both numbers are right and they answer different questions: this counter speaks for the scope the mutation gate runs against, and the per-file sum speaks for the change. A first measurement of the same change said +59, and the eight it lost are the two loops of the reporter's summary folded into one pass -- a gain this file is required to record rather than hand back. Then 949 (+3) for --include-prefix, the mirror of --exclude-prefix that the same report asked for: the four product files it touches are internal/staged/staged.go, internal/staged/changed.go, staged.go and changed.go, and the whole delta is three mutants because the filter is one more question asked of a predicate that already existed. It is not attributed per file here: the change is one predicate threaded through four files, and splitting +3 across them by arithmetic would be a number invented rather than measured. What it buys is the honest pass the report wanted -- one run per owning package, in one flag, without rewriting the test command per run -- and the shape it changes is exported: PlanStaged, RunStaged, PlanChanged and RunChanged take a ditto.Prefixes instead of one []string, so a change of scope and a narrowing of it are one argument rather than two that can drift. Then 948 (-1), a deletion rather than a gain: the MUTATE pass deleted the laboratory's nil check on its scope and no test could tell the difference, because the baseline always produces one -- a guard that survives its own deletion is dead weight, and this file records the removal it earned as it records everything else. docs/reports/ditto-mutation-scope.md." } }