From a06ebd126762ea6a0d631aa244649e3ae87a7628 Mon Sep 17 00:00:00 2001 From: Manuel de Brito Fontes Date: Tue, 15 Sep 2026 22:25:47 -0300 Subject: [PATCH 1/3] ci: the BuildKit cache is written at all, and each build gets its own scope MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two faults, either of which alone makes every CI run a cold build. The first is that nothing was ever stored. BuildKit authenticates to GitHub's cache service with $ACTIONS_RUNTIME_TOKEN and $ACTIONS_RESULTS_URL, and those reach JavaScript actions, not `run:` steps. Every build here goes through `task`, which is a `run:` step, so each --cache-to type=gha was accepted, stored nothing, and said nothing. It is invisible in a log; where it shows is the cache itself, which held one entry — setup-go's 14 MiB — for a repository that compiles QEMU and a kernel. crazy-max/ghaction-github-runtime exposes the variables, and Docker's documentation names it for exactly this. The second would have bitten the moment the first was fixed. `type=gha` defaults to the scope `buildkit`, and per Docker's documentation, "each build will overwrite the cache of the previous, leaving only the final cache." The release job runs four builds in one job, so QEMU, the kernel, e2fsprogs and the image each destroyed the last one's cache and missed on the next run. So the scopes are named, and named in Taskfile.yml rather than in the workflows: they have to match between the per-artefact lanes and the release, or a release starts cold on work main already paid for. CI now sets CACHE_BACKEND=gha and nothing else; a developer still gets the local directory, byte for byte as before. The QEMU runtime image reads scope=qemu to reuse the builder stage and writes only its own. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/image.yml | 13 +++++----- .github/workflows/kernel.yml | 10 ++++---- .github/workflows/qemu.yml | 34 +++++++++++++++++++++----- .github/workflows/release.yml | 22 ++++++++++------- Taskfile.yml | 45 ++++++++++++++++++++++++----------- 5 files changed, 86 insertions(+), 38 deletions(-) diff --git a/.github/workflows/image.yml b/.github/workflows/image.yml index b32bdcd..369a3f1 100644 --- a/.github/workflows/image.yml +++ b/.github/workflows/image.yml @@ -46,6 +46,11 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # Without this, `--cache-to type=gha` below is a silent no-op: the tokens BuildKit + # needs reach JavaScript actions and not `run:` steps. See the long note in qemu.yml. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # image/build.sh asserts what came out on the finished filesystem rather than # trusting the build that made it: that there is an /sbin/init and a /bin/sh, that # the directories the guest mounts over exist, that no identity is baked in, and @@ -59,15 +64,11 @@ jobs: # the build container's, and image/build.sh stops when it is not there. - name: Build e2fsprogs run: | - task e2fsprogs:build \ - E2FSPROGS_CACHE_FROM=type=gha \ - E2FSPROGS_CACHE_TO=type=gha,mode=max + CACHE_BACKEND=gha task e2fsprogs:build - name: Build the base image run: | - task image:build \ - IMAGE_CACHE_FROM=type=gha \ - IMAGE_CACHE_TO=type=gha,mode=max + CACHE_BACKEND=gha task image:build - name: What this image is run: | diff --git a/.github/workflows/kernel.yml b/.github/workflows/kernel.yml index de51e64..1aa9c43 100644 --- a/.github/workflows/kernel.yml +++ b/.github/workflows/kernel.yml @@ -44,15 +44,17 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # Without this, `--cache-to type=gha` below is a silent no-op: the tokens BuildKit + # needs reach JavaScript actions and not `run:` steps. See the long note in qemu.yml. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # Ends in `task kernel:verify`, which opens the ELF and checks the Xen PVH notes # survived the strip. Without them QEMU has no entry point into this kernel and the # failure is a VM that does not start, in whichever lane boots one next. - name: Build the kernel run: | - task kernel:build \ - KERNEL_CACHE_FROM=type=gha \ - KERNEL_CACHE_TO=type=gha,mode=max \ - KERNEL_NPROC=4 + CACHE_BACKEND=gha task kernel:build KERNEL_NPROC=4 # Compared, now, against the last released machine — which is the reference this step # used to say it did not have. It does: every release publishes its machine.env as an diff --git a/.github/workflows/qemu.yml b/.github/workflows/qemu.yml index 0e4e535..3141940 100644 --- a/.github/workflows/qemu.yml +++ b/.github/workflows/qemu.yml @@ -58,6 +58,24 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # What makes `--cache-to type=gha` do anything at all. + # + # BuildKit writes to GitHub's cache service with $ACTIONS_RUNTIME_TOKEN and + # $ACTIONS_RESULTS_URL, and those are handed to JavaScript actions, not to `run:` + # steps. Docker's documentation is explicit — "If you invoke the `docker buildx` + # command manually from an inline step, then the variables must be manually exposed" + # — and names this action for it. Every build here runs through `task`, which is a + # `run:` step. + # + # Without it the flags are accepted, nothing is stored, and nothing says so: the + # builds pass and take the cold time, every run. That failure is invisible in a log. + # Where it shows is `gh api repos/{owner}/{repo}/actions/cache/usage` — a repository + # that builds QEMU and a kernel and reports a handful of megabytes is caching neither. + # + # Remove this step and nothing breaks; everything just gets slow again. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # The pinned version comes from the Taskfile, never from a literal here: two places # to bump is how CI and production end up on different QEMUs. # @@ -73,10 +91,7 @@ jobs: - name: Build QEMU and verify what came out run: | - task qemu:build \ - QEMU_CACHE_FROM=type=gha \ - QEMU_CACHE_TO=type=gha,mode=max \ - QEMU_JOBS=4 + CACHE_BACKEND=gha task qemu:build QEMU_JOBS=4 # vhost-vsock is a kernel module, and the machine's vsock device cannot be created # without /dev/vhost-vsock. The runner has the node and gives the user no access to @@ -145,6 +160,12 @@ jobs: # Two tags: the QEMU version, which is what everything pins, and the commit, which is # what makes a specific build reachable when the version tag has moved because a flag # changed. + # + # Two --cache-from and one --cache-to, which is not symmetry gone wrong. This target + # sits on top of the builder stage the step above just compiled, so it reads the + # `qemu` scope to get it and writes only its own — sharing one scope would have the + # two builds overwriting each other's cache, which is the default behaviour that made + # every scope in this repository distinct in the first place. - name: Publish the runtime image if: github.event_name != 'pull_request' run: | @@ -152,8 +173,9 @@ jobs: --file qemu/Dockerfile \ --target runtime \ --platform linux/amd64 \ - --cache-from type=gha \ - --cache-to type=gha,mode=max \ + --cache-from type=gha,scope=qemu \ + --cache-from type=gha,scope=qemu-image \ + --cache-to type=gha,scope=qemu-image,mode=max \ --build-arg QEMU_VERSION=${{ steps.qemu.outputs.version }} \ --build-arg JOBS=4 \ --label org.opencontainers.image.source=https://github.com/${{ github.repository }} \ diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index dcb2004..babf2b4 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -63,6 +63,11 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # Without this, `--cache-to type=gha` below is a silent no-op: the tokens BuildKit + # needs reach JavaScript actions and not `run:` steps. See the long note in qemu.yml. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # Three sources, in order: what the caller typed, the tag that triggered this, or a # new one for today. The generated form is deliberately not `date +%s` or a commit # hash — it has to be readable, sortable, and the same shape as one typed by hand. @@ -107,14 +112,15 @@ jobs: - name: Build the release env: VERSION: ${{ steps.version.outputs.version }} - QEMU_CACHE_FROM: type=gha - QEMU_CACHE_TO: type=gha,mode=max - KERNEL_CACHE_FROM: type=gha - KERNEL_CACHE_TO: type=gha,mode=max - E2FSPROGS_CACHE_FROM: type=gha - E2FSPROGS_CACHE_TO: type=gha,mode=max - IMAGE_CACHE_FROM: type=gha - IMAGE_CACHE_TO: type=gha,mode=max + # The same scopes the per-artefact workflows write on main, decided in + # Taskfile.yml. That is the whole point of them being named there: a release + # reads what the lane on main already paid to compile, instead of starting cold + # because a workflow file spelled the scope differently. + # + # This run's own writes are mostly wasted — it is triggered by a tag, and a cache + # written under one tag ref cannot be read from another tag or from a branch. The + # reads are what matter, and those come from the default branch. + CACHE_BACKEND: gha run: task release QEMU_JOBS=4 KERNEL_NPROC=4 # The machine's identity, and what it costs whoever holds templates. diff --git a/Taskfile.yml b/Taskfile.yml index 884013b..9bf01cb 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -97,21 +97,38 @@ vars: KERNEL_ARCH: '{{.KERNEL_ARCH | default "x86_64"}}' KERNEL_NPROC: '{{.KERNEL_NPROC | default 8}}' - # BuildKit cache backends. Local by default; CI overrides them with `type=gha` so the - # build is not duplicated in a workflow file — the Dockerfiles and the flags stay in one - # place, which is the rule these files exist for. + # Where BuildKit reads and writes its cache: a directory under the repository for a + # developer, GitHub's cache service in CI. One name to switch — `CACHE_BACKEND=gha` — and + # every scope below is decided here, next to the builds they belong to, rather than as + # eight cache strings pasted into workflow files. + # + # The scopes are why this is a var and not a string in a workflow. `type=gha` defaults to + # the scope `buildkit`, and Docker's own documentation says what that means for a + # repository with more than one build: "each build will overwrite the cache of the + # previous, leaving only the final cache." Four builds sharing that default is four builds + # each destroying the last one's cache and then missing on the next run. The names have to + # be distinct, and — this is the part that is easy to get wrong — they have to be the + # *same* names in the per-artefact workflows and in the release, or a release warms + # nothing that the lane on main already paid for. + # + # `gha` needs one thing this file cannot give it — see the note in + # .github/workflows/qemu.yml. The variables that authorise a write to the cache service + # are not in a plain `run:` step's environment, and without them `--cache-to type=gha` is + # accepted, stores nothing, and says nothing. + CACHE_BACKEND: '{{.CACHE_BACKEND | default "local"}}' BUILDKIT_CACHE_DIR: '{{.BUILDKIT_CACHE_DIR | default (printf "%s/.cache/buildkit" .ROOT_DIR)}}' - QEMU_CACHE_FROM: '{{.QEMU_CACHE_FROM | default (printf "type=local,src=%s/qemu" .BUILDKIT_CACHE_DIR)}}' - QEMU_CACHE_TO: '{{.QEMU_CACHE_TO | default (printf "type=local,dest=%s/qemu,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' - KERNEL_CACHE_FROM: '{{.KERNEL_CACHE_FROM | default (printf "type=local,src=%s/kernel" .BUILDKIT_CACHE_DIR)}}' - KERNEL_CACHE_TO: '{{.KERNEL_CACHE_TO | default (printf "type=local,dest=%s/kernel,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' - E2FSPROGS_CACHE_FROM: '{{.E2FSPROGS_CACHE_FROM | default (printf "type=local,src=%s/e2fsprogs" .BUILDKIT_CACHE_DIR)}}' - E2FSPROGS_CACHE_TO: '{{.E2FSPROGS_CACHE_TO | default (printf "type=local,dest=%s/e2fsprogs,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' - # The container that builds the base image. Small — mkosi, qemu-utils — but it - # was the one buildx invocation here with no cache at all, so every CI run reinstalled it - # from the archive before any image work started. - IMAGE_CACHE_FROM: '{{.IMAGE_CACHE_FROM | default (printf "type=local,src=%s/image" .BUILDKIT_CACHE_DIR)}}' - IMAGE_CACHE_TO: '{{.IMAGE_CACHE_TO | default (printf "type=local,dest=%s/image,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' + + QEMU_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=qemu{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/qemu{{end}}' + QEMU_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=qemu,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/qemu,mode=max,compression=zstd{{end}}' + KERNEL_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=kernel{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/kernel{{end}}' + KERNEL_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=kernel,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/kernel,mode=max,compression=zstd{{end}}' + E2FSPROGS_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=e2fsprogs{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/e2fsprogs{{end}}' + E2FSPROGS_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=e2fsprogs,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/e2fsprogs,mode=max,compression=zstd{{end}}' + # The container that builds the base image. Small — mkosi, qemu-utils — and cached like + # the rest: uncached, every CI run reinstalls it from the archive before any image work + # starts. + IMAGE_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=image{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/image{{end}}' + IMAGE_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=image,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/image,mode=max,compression=zstd{{end}}' # mkosi's package cache, which is not a BuildKit cache because mkosi does not run under # BuildKit (see image/Dockerfile). A rebuild of the base image is ~1.5 GB of apt. From 3a560436b7271a661c5e29d60ab1017c496d11bc Mon Sep 17 00:00:00 2001 From: Manuel de Brito Fontes Date: Tue, 15 Sep 2026 22:25:58 -0300 Subject: [PATCH 2/3] Comments say what is true, not what the tree used to hold MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A comment that narrates this repository's own past — "this used to be X", "it did not work before " — is answerable by git, and it is the half of a comment that rots. What earns its place is the dead end someone would otherwise walk into again, and the measurement that settles it. So the rule applied here is: keep the trap and the number, drop the history. "It used to be turned off here, on the grounds that..." becomes "Turning it off here is the tempting change, on the grounds that..." — and every measurement stays: nohz=off still carries its 1.7% of a core and its 821-853 ms, pahole 1.24 still explains why the kernel toolchain cannot go back to bookworm, and the missing journal still produces "structure needs cleaning" on a copy taken seconds after a grow. Two paragraphs are deleted rather than rewritten, both in optimize-systemd.sh. They were commentary on an earlier revision of the comment above them, and nothing outlives the deletion. The open problem in spin-machine-console.service is untouched, only put in the present tense: the reason that unit exists no longer holds, and the hole it leaves is still the thing to close. Co-Authored-By: Claude Opus 5 (1M context) --- .github/workflows/kernel.yml | 13 ++++---- .github/workflows/qemu.yml | 6 ++-- .github/workflows/release.yml | 25 ++++++++------- README.md | 31 +++++++++---------- Taskfile.yml | 27 ++++++++-------- boot/bench_test.go | 4 +-- boot/phases.go | 16 +++++----- image/Taskfile.yml | 8 ++--- image/build.sh | 4 +-- .../local/lib/spin-base/configure-system.sh | 7 ++--- .../local/lib/spin-base/optimize-systemd.sh | 8 ----- .../system/spin-machine-console.service | 12 +++---- kernel/Dockerfile | 8 ++--- machine/cmdline.go | 10 +++--- machine/machine_test.go | 4 +-- qemu/Dockerfile | 6 ++-- qemu/devices.mak | 12 +++---- 17 files changed, 95 insertions(+), 106 deletions(-) diff --git a/.github/workflows/kernel.yml b/.github/workflows/kernel.yml index 1aa9c43..45f6271 100644 --- a/.github/workflows/kernel.yml +++ b/.github/workflows/kernel.yml @@ -56,17 +56,16 @@ jobs: run: | CACHE_BACKEND=gha task kernel:build KERNEL_NPROC=4 - # Compared, now, against the last released machine — which is the reference this step - # used to say it did not have. It does: every release publishes its machine.env as an - # asset, and the kernel's checksum is one of the two numbers in it that decide whether - # a template still matches. + # Compared against the last released machine, which every release publishes as a + # machine.env asset. The kernel's checksum is one of the two numbers in it that decide + # whether a template still matches. # # So a pull request that changes the config, or bumps the Debian toolchain the kernel # is compiled with, says on its own summary that it invalidates every template in # existence. That is not a reason to reject it — it is a release this repository makes - # deliberately, and the step exits 0 either way — but it was previously a sentence - # nobody wrote down, and the last time it happened it was a comment in this directory's - # Dockerfile that nobody expected to rebuild anything. + # deliberately, and the step exits 0 either way. It is a reason to know: the change + # that does this need not look like much, and a comment edited in this directory's + # Dockerfile is enough. # # It speaks only for the kernel: this workflow builds no QEMU, and the verdict says so # rather than implying it checked the whole machine. diff --git a/.github/workflows/qemu.yml b/.github/workflows/qemu.yml index 3141940..10c9c88 100644 --- a/.github/workflows/qemu.yml +++ b/.github/workflows/qemu.yml @@ -9,9 +9,9 @@ name: QEMU on: # qemu/** and nothing else. The version pin is the ARG in qemu/Dockerfile, so a bump is a -# change under qemu/ and fires this; the root Taskfile holds only how many jobs to compile -# with and which cache to use, and used to be listed here — which meant every edit to it -# rebuilt QEMU, the kernel and the base image for nothing. +# change under qemu/ and fires this. The root Taskfile is deliberately not listed: it holds +# only how many jobs to compile with and which cache to use, so listing it would rebuild +# QEMU, the kernel and the base image for an edit that cannot change any of them. push: branches: [main] paths: diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index babf2b4..fa263ce 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -26,12 +26,12 @@ permissions: contents: write packages: write -# One release at a time, across every ref. The group used to be per-ref, which serialised -# two runs on the same tag and nothing else: a workflow_dispatch on main and a tag push are -# different refs, so both counted the tags already taken today, both generated .01, and the -# loser found out at `git tag` — after six hours of building. There is no reason to build -# two releases at once, and the version is decided at the start of a run and pushed at the -# end, so the window is the whole build. +# One release at a time, across every ref, and not per-ref. Per-ref serialises two runs on +# the same tag and nothing else: a workflow_dispatch on main and a tag push are different +# refs, so both count the tags already taken today, both generate .01, and the loser finds +# out at `git tag` — after six hours of building. The version is decided at the start of a +# run and pushed at the end, so that window is the whole build, and there is no reason to +# build two releases at once anyway. concurrency: group: release cancel-in-progress: false @@ -125,14 +125,15 @@ jobs: # The machine's identity, and what it costs whoever holds templates. # - # "why did every template in the fleet stop matching?" used to be answered by opening - # two release pages and comparing sixty-four hex digits by eye, which is a thing - # nobody does and therefore an answer nobody had. hack/fingerprint-diff fetches the - # previous release's machine.env and states the consequence in a sentence, at the top - # of the notes, where a consumer deciding whether to take this release reads it. + # The alternative to hack/fingerprint-diff is opening two release pages and comparing + # sixty-four hex digits by eye, which is a thing nobody does — so the answer to "why + # did every template in the fleet stop matching?" is one nobody has. The script + # fetches the previous release's machine.env and states the consequence in a sentence, + # at the top of the notes, where a consumer deciding whether to take this release + # reads it. # # It exits 0 on every verdict: a changed machine is a release this repository makes on - # purpose. The point is that it can no longer be made quietly. + # purpose. The point is not to prevent it but to keep it from being quiet. - name: What this release is, and what it breaks id: manifest env: diff --git a/README.md b/README.md index 1f85ae2..68780a0 100644 --- a/README.md +++ b/README.md @@ -69,9 +69,9 @@ One tarball: `task build` writes that same tree into `_output/`, byte for byte the layout above, and `machine.Open` reads either. There is one layout: nothing rearranges the files on the way -out of a build, into a tarball or into a consumer, because the three used to differ and -what fell out of the translation between them was a path that existed and held the -previous release's kernel. +out of a build, into a tarball or into a consumer. Let the three differ and what falls out +of the translation between them is a path that exists and holds the previous release's +kernel. ```go rel, err := machine.Open("/usr/share/spin-stack") // says which file is missing, if one is @@ -174,10 +174,9 @@ It boots this QEMU and this kernel over a throwaway qcow2 overlay on `rootfs.qco It boots with no initrd at all: `root=/dev/vda rw init=/sbin/init`, which the kernel can serve because virtio-blk and ext4 are built in and `image/build.sh` writes a partitionless -filesystem. There was a debug initramfs here until 2026-09-10 — static Go that mounted the -API filesystems, found the root disk and exec'd — and it was a second init, doing what the -init a consumer brings already does. What removed the need for it was turning `systemd-udevd` -back on. +filesystem. There is no debug initramfs: one that mounted the API filesystems, found the +root disk and exec'd would be a second init, doing what the init a consumer brings already +does. What removes the need for one is `systemd-udevd` being on. Three things the shell has found, all of them true of the image and none of them visible from outside: @@ -190,12 +189,12 @@ from outside: its own with the dependency removed. Masking bought nothing measurable and cost a ten-second `dev-ttyS0.device` timeout on every boot that had no initrd; the numbers are in `optimize-systemd.sh`, where the decision is. -- **`ssh.service` used to fail five times and give up.** The image ships no host keys — they - are identity — and the distribution's `sshd-keygen.service` carries - `ConditionFirstBoot=yes`, so it did not run before `sshd` was asked to validate a - configuration with no keys. Fixed by generating them at boot instead - (`spin-machine-sshd-keygen.service`): every boot here *is* a first boot, since the root - filesystem is a fresh overlay, so the keys live exactly as long as the VM does. +- **`sshd` needs host keys generated at boot.** The image ships none — they are identity — + and the distribution's `sshd-keygen.service` carries `ConditionFirstBoot=yes`, so it does + not run before `sshd` is asked to validate a configuration with no keys, and + `ssh.service` fails five times and gives up. `spin-machine-sshd-keygen.service` generates + them instead: every boot here *is* a first boot, since the root filesystem is a fresh + overlay, so the keys live exactly as long as the VM does. - **Two thirds of the boot was systemd asking the console questions.** The kernel execs `/sbin/init` at 59 ms and systemd's first log line arrived at 852, with nothing running in between. It was a terminfo query and a terminal reset, 334 ms of timeout each, waiting for @@ -283,9 +282,9 @@ which is a different fingerprint. It moves on purpose, not when Debian publishes release. And the build is reproducible: two builds from the same inputs produce the same `vmlinux`, -byte for byte. It was not until 2026-09-14: bookworm's pahole 1.24 encoded `.BTF` with make's -`-j` in whatever order its threads finished, and every release had a new fingerprint whether -or not the kernel had changed. trixie's 1.30 is deterministic with `-j`. +byte for byte. That needs trixie's pahole 1.30, which is deterministic with `-j`. Bookworm's +1.24 encodes `.BTF` in whatever order make's `-j` threads finish, which gives every release a +new fingerprint whether or not the kernel changed. ### Base image diff --git a/Taskfile.yml b/Taskfile.yml index 9bf01cb..17cdc1e 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -20,10 +20,10 @@ # qemu/Dockerfile and nobody reads three hundred lines to change one flag. # # What is deliberately NOT here: the software that runs inside a guest. This repository -# builds a machine and knows nothing about what boots on it, and there is no longer an -# exception to that — the debug initramfs that used to be one was a second init, doing what -# the init a consumer already brings does, and the machine boots straight into /sbin/init -# without it. +# builds a machine and knows nothing about what boots on it, and there is no exception to +# that. A debug initramfs is the one that keeps being proposed: it is a second init, doing +# what the init a consumer already brings does, and the machine boots straight into +# /sbin/init without one. version: '3' @@ -65,17 +65,16 @@ vars: # One version for all three artefacts. See the header. # - # --dirty, and it is the whole point of the line. This used to be --exact-match with a - # bare `git rev-parse --short HEAD` behind it, neither of which says anything about - # uncommitted changes — so `task release` on a tagged commit with edits in the tree - # produced spin-machine-v20260914.03-linux-x86_64.tar.gz with version=v20260914.03 in its - # manifest: the same name and the same declared identity as the release that was actually - # published, around different bytes. machine.env is how a host explains why its templates - # stopped matching, and that made it lie. + # --dirty, and it is the whole point of the line. Without it, `task release` on a tagged + # commit with edits in the tree produces spin-machine-v20260914.03-linux-x86_64.tar.gz + # with version=v20260914.03 in its manifest: the same name and the same declared identity + # as the release that was actually published, around different bytes. machine.env is how + # a host explains why its templates stopped matching, so it is the one file that must not + # be able to claim an identity the bytes do not have. # - # The fallback is --tags --always --dirty rather than a bare hash, which is what - # hack/release already defaulted to on its own: v20260914.03-1-g86d2386 says which release - # this is one commit past, where 86d2386 says nothing until somebody looks it up. + # The fallback is --tags --always --dirty rather than a bare hash, matching what + # hack/release defaults to on its own: v20260914.03-1-g86d2386 says which release this is + # one commit past, where 86d2386 says nothing until somebody looks it up. VERSION: sh: | echo "${VERSION:-$(git describe --tags --exact-match --dirty 2>/dev/null \ diff --git a/boot/bench_test.go b/boot/bench_test.go index 1cdb271..8078f15 100644 --- a/boot/bench_test.go +++ b/boot/bench_test.go @@ -32,8 +32,8 @@ type variant struct { // implemented by systemd-debug-generator, which optimize-systemd.sh symlinks to /dev/null, // so it parses, reaches /proc/cmdline and does nothing: verified 2026-09-10 with // `systemd.mask=chrony.service` on the command line and `systemctl is-active chrony` -// answering `active`. Everything this repository concluded from that parameter was concluded -// from a boot in which nothing had been masked. +// answering `active`. Anything concluded from that parameter is concluded from a boot in +// which nothing was masked. var udevUnits = []string{ "systemd-udevd.service", "systemd-udevd-control.socket", diff --git a/boot/phases.go b/boot/phases.go index 5358146..74aa0a3 100644 --- a/boot/phases.go +++ b/boot/phases.go @@ -2,10 +2,10 @@ // Package boot measures what a boot of this machine costs, from the host. // -// It exists because the number this repository had been quoting was systemd's own -// `Startup finished`, which begins counting when the kernel hands over. Exec'ing QEMU, the -// firmware and loading the kernel are all before it; a login prompt is after it. Neither end -// had ever been measured, and both are part of what an operator waits for. +// It exists because the number a machine reports for itself — systemd's `Startup +// finished` — begins counting when the kernel hands over. Exec'ing QEMU, the firmware and +// loading the kernel are all before it; a login prompt is after it. Neither end is in that +// number, and both are part of what an operator waits for. // // The package is split so that the part that can be wrong is testable without a VM: Watch // turns a stream of console bytes into phase timings, and Percentile turns a set of runs @@ -91,8 +91,8 @@ const ( // the VMM is doing. // // None of that says the firmware is cheap, and it is not: its work comes after the - // banner, and it had never been measured until 2026-09-13. QEMU exec to the kernel's PVH - // entry is 43.9 and 45.1 ms, 40 boots in each of two runs, against 34.6 to the banner — + // banner. Measured 2026-09-13, QEMU exec to the kernel's PVH entry is 43.9 and 45.1 ms, + // 40 boots in each of two runs, against 34.6 to the banner — // so SeaBIOS's POST (PCI enumeration, SMM and MTRR setup, the ACPI table loader, a scan of // storage and input the machine does not have, then pvh.bin) is ~10 ms. Two ways of // spending less on it, and what each came to: @@ -127,8 +127,8 @@ const ( // PID1 is the kernel handing over. The kernel must be printing at loglevel 7 for this // to appear at all, so it is absent from a quiet boot rather than zero. PID1 - // Started is systemd's `Startup finished`, the number this repository used to quote in - // full. Kept so the old number stays comparable to the new ones. + // Started is systemd's `Startup finished`: the number the machine reports for itself, + // and the one most projects quote. Kept so these timings stay comparable to it. Started // Usable is a login prompt. It is the only one of these that is not the machine making // a claim about itself, and it is the one an operator is actually waiting for. diff --git a/image/Taskfile.yml b/image/Taskfile.yml index f7ac9df..3ce56ed 100644 --- a/image/Taskfile.yml +++ b/image/Taskfile.yml @@ -89,12 +89,12 @@ tasks: # there is nothing for one to do: mounting root is the kernel's job here and # systemd mounts the rest. # - # This used to boot a debug initramfs, for two reasons that were both real and are + # Booting a debug initramfs here is a second init, and the two reasons for one are # both gone. It mounted /proc, /sys, devtmpfs and devpts, which systemd does itself # and only a bare `init=/bin/sh` ever needed. And it wrote a symlink enabling a - # serial login, because this image masked systemd-udevd, so no dev-ttyS0.device - # unit existed and serial-getty@ttyS0's BindsTo could never be satisfied. udev - # runs now; the getty starts on its own. Measured 2026-09-10: 207ms to a login + # serial login, which was needed while this image masked systemd-udevd: no + # dev-ttyS0.device unit existed and serial-getty@ttyS0's BindsTo could never be + # satisfied. udev runs; the getty starts on its own. Measured 2026-09-10: 207ms to a login # prompt this way, against 163ms through the initramfs — 44ms for a second init # doing what the first one does. # diff --git a/image/build.sh b/image/build.sh index dabe333..f5e8417 100755 --- a/image/build.sh +++ b/image/build.sh @@ -154,8 +154,8 @@ echo "==> $(wc -l < "$SHARE/packages.txt") packages recorded" # its own — but the filesystem in it is: an overlay holds this same ext4, mounted read-write # and grown onto the disk the overlay was made at. A copy of that overlay taken while the guest # writes, or the overlay of a machine that died, is a filesystem interrupted mid-update, and -# without a journal nothing puts its metadata back together at the next mount. It was built -# without one until 2026-09-14, and a copy taken seconds after a grow would not mount: +# without a journal nothing puts its metadata back together at the next mount. Measured +# 2026-09-14: built without one, a copy taken seconds after a grow does not mount — # "structure needs cleaning", a corrupt group descriptor. # # Each overlay replays and writes its own journal; the blocks it writes land in that overlay, diff --git a/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh b/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh index 7fc5320..ad8ccb5 100755 --- a/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh +++ b/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh @@ -32,8 +32,7 @@ sed -i 's/#PubkeyAuthentication yes/PubkeyAuthentication yes/' /etc/ssh/sshd_con # # Without this line the control plane signs five-minute certificates that nothing on earth # accepts, and every login quietly falls back to a long-lived key in authorized_keys that -# nothing ever rewrites. That was the state of things until 2026-09-11, and it was invisible -# because logins kept working. +# nothing ever rewrites. It is invisible when it happens, because logins keep working. mkdir -p /etc/ssh/sshd_config.d cat <<'EOF' > /etc/ssh/sshd_config.d/10-spin-user-ca.conf TrustedUserCAKeys /etc/ssh/spin_user_ca.pub @@ -116,8 +115,8 @@ esac MOTD chmod 0755 /etc/update-motd.d/00-spin-boot -# pam_motd stays enabled, including for ssh, where it had been commented out. There is -# something worth printing now. +# pam_motd stays enabled, including for ssh, where the distribution ships it commented +# out. There is something worth printing. sed -i 's/^#\(session.*pam_motd.so\)/\1/' /etc/pam.d/sshd diff --git a/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh b/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh index a2c3c0c..3d55c56 100755 --- a/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh +++ b/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh @@ -37,10 +37,6 @@ MASK_UNITS=( # spin-machine-console.service, and the decision left open there. # * A hot-plugged CPU or memory block never produced an add event, so nothing could # act on one — see 40-spin-hotadd.rules, which is the thing that acts on it. - # - # The numbers here used to compare a boot with the debug initramfs against one without. - # That initramfs was deleted the same day, so the comparison it made cannot be re-run; - # what is above is the one the harness measures now. # Time sync - handled by host systemd-timesyncd.service @@ -139,10 +135,6 @@ MASK_UNITS=( # tmpfiles-setup (19ms + 13ms + 11ms). The directories it would create are already # there: /tmp is in the image at mode 1777 (image/build.sh makes it), /run is a tmpfs # systemd mounts itself before any unit runs, and /dev is devtmpfs with udev on top. - # - # The reason recorded here until 2026-09-10 was that an init outside this repository - # created them, which was both wrong — it named a consumer, and this machine now boots - # root=/dev/vda with no initrd at all — and unfalsifiable from inside this tree. systemd-tmpfiles-setup.service systemd-tmpfiles-setup-dev.service systemd-tmpfiles-setup-dev-early.service diff --git a/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service b/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service index cac7295..62d624b 100644 --- a/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service +++ b/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service @@ -13,15 +13,15 @@ # systemd gave up. # # Why this and not the distribution's serial-getty@.service: it carries -# `BindsTo=dev-%i.device`, a .device unit exists only if udev announced it, and this image -# used to mask systemd-udevd. Enabling serial-getty then got +# `BindsTo=dev-%i.device`, and a .device unit exists only if udev announced it. With +# systemd-udevd masked, enabling serial-getty gets # `[DEPEND] Dependency failed for serial-getty@ttyS0.service`, and a drop-in clearing -# BindsTo did not lift it either. This unit has no device dependency to fail. +# BindsTo does not lift it either. This unit has no device dependency to fail. # -# THAT REASON EXPIRED ON 2026-09-10 and left a hole this file is the wrong place to close. -# udev runs now, so serial-getty@ttyS0 works — and systemd-getty-generator instantiates it +# THAT REASON NO LONGER HOLDS, and it leaves a hole this file is the wrong place to close. +# udev runs, so serial-getty@ttyS0 works — and systemd-getty-generator instantiates it # on its own from console=ttyS0, with nothing in optimize-systemd.sh disabling that -# generator. So the split this file is built around no longer exists: production gets a +# generator. So the split this file is built around does not exist: production gets a # login prompt on the console it prints its kernel log to, which is what the paragraph above # says production must not have, and it gets it without anybody enabling anything. # diff --git a/kernel/Dockerfile b/kernel/Dockerfile index 83dcb2c..2e7a9c0 100644 --- a/kernel/Dockerfile +++ b/kernel/Dockerfile @@ -340,10 +340,10 @@ EOT # ============================================ # # The tree this stage writes is the release tree: kernel/vmlinux, at the path a release -# carries it and the path machine.Release reads it from. It used to extract to the root and -# let hack/release do the moving, which meant the same four files had one layout here, a -# second in the tarball and a third wherever a consumer rearranged them into — with a -# translation step between each pair, and each of those a place to be wrong. +# carries it and the path machine.Release reads it from. Extracting to the root and letting +# hack/release do the moving gives the same four files one layout here, a second in the +# tarball and a third wherever a consumer rearranges them into — with a translation step +# between each pair, and each of those a place to be wrong. FROM scratch AS kernel COPY --from=kernel-build /build/vmlinux /kernel/vmlinux COPY --from=kernel-build /build/kernel-config /kernel/kernel-config diff --git a/machine/cmdline.go b/machine/cmdline.go index acd4089..d37cf8b 100644 --- a/machine/cmdline.go +++ b/machine/cmdline.go @@ -130,13 +130,13 @@ func (c Cmdline) String() string { // The tick stops when a CPU has nothing to do, which is what CONFIG_NO_HZ_IDLE // is compiled in for. // - // It used to be turned off here, on the grounds that a short-lived VM never - // amortises the tickless machinery's setup cost and that the timer interrupt - // it saves is cheap on a guest. Both halves were wrong when measured - // (2026-09-09). The interrupt is not cheap: a guest kernel at CONFIG_HZ=1000 + // Turning it off here is the tempting change, on the grounds that a short-lived + // VM never amortises the tickless machinery's setup cost and that the timer + // interrupt it saves is cheap on a guest. Both halves are wrong, measured + // 2026-09-09. The interrupt is not cheap: a guest kernel at CONFIG_HZ=1000 // with the tick forced on burns 1.7% of a core per vCPU doing nothing at all, // because every one of those thousand wakeups a second is a vmexit. And the - // setup cost it was buying back does not show: with the tick left alone, a + // setup cost it would buy back does not show: with the tick left alone, a // machine's boot and a workspace's restore both measured what they did before // — 821-853 ms to boot and build a template, 113-221 ms to a usable workspace, // against 843-846 ms and 124-216 ms with nohz=off. diff --git a/machine/machine_test.go b/machine/machine_test.go index c586666..d06bf67 100644 --- a/machine/machine_test.go +++ b/machine/machine_test.go @@ -315,9 +315,9 @@ func TestFingerprintCoversTheDevicesPresentAtRestore(t *testing.T) { // A NIC is on the command line, so it is present when state is loaded and a machine // with one cannot restore from a template frozen without one. // - // This case said the opposite until 2026-09-09 — that a NIC must not change the + // The opposite is the tempting assertion — that a NIC must not change the // fingerprint, because "it is cold-plugged after a restore, so a VM would never find - // its template". That was true of disks and asserted of both. + // its template". That is true of disks, and does not carry across to NICs. withNIC := base withNIC.NICs = []NIC{{TapFD: 3, MAC: "52:54:00:00:00:01"}} got, err := withNIC.Fingerprint() diff --git a/qemu/Dockerfile b/qemu/Dockerfile index 43d04f8..598519f 100644 --- a/qemu/Dockerfile +++ b/qemu/Dockerfile @@ -392,9 +392,9 @@ RUN for b in /opt/qemu/bin/qemu-system-x86_64 /opt/qemu/bin/qemu-img \ # # Two binaries and five files on a base that exists only so a human can open a shell in it # and look around. There are no libraries to install: that is the whole difference static -# linking makes here. This stage used to be an Ubuntu with a hand-written apt list that had -# to be kept in step with what the builder linked against, and the way it was found to be -# wrong was a CI run. +# linking makes here. An Ubuntu base with a hand-written apt list would have to be kept in +# step with what the builder links against, and the way that is found to be wrong is a CI +# run. FROM alpine:${ALPINE_VERSION} AS runtime COPY --from=builder /opt/qemu/bin/qemu-system-x86_64 /usr/local/bin/qemu-system-x86_64 diff --git a/qemu/devices.mak b/qemu/devices.mak index 5e8d1fe..29445ec 100644 --- a/qemu/devices.mak +++ b/qemu/devices.mak @@ -11,14 +11,14 @@ # upstream shipped depends on whether some earlier build happened to configure. This file # is copied in and read on every build. # -# ## An allowlist, since 2026-09-10 +# ## An allowlist, not a denylist # -# This file used to include upstream's default.mak and switch off about forty symbols. That -# is a denylist, and it loses by construction: it can only remove what somebody thought to +# The obvious alternative is to include upstream's default.mak and switch off some forty +# symbols. That is a denylist, and it loses by construction: it can only remove what somebody thought to # name, and every QEMU release adds devices that arrive switched on. The binary it produced -# could instantiate ich9-ahci, ide-hd, sb16, isa-fdc, isa-parallel, VGA, virtio-vga, -# vfio-pci, intel-iommu and amd-iommu — audited on the shipped binary with `-device help`, -# not read off this file. +# binary it produces can instantiate ich9-ahci, ide-hd, sb16, isa-fdc, isa-parallel, VGA, +# virtio-vga, vfio-pci, intel-iommu and amd-iommu — audited on a shipped binary with +# `-device help`, not read off a file like this one. # # Now the build is configured `--without-default-devices`, which makes meson run # scripts/minikconf.py with `--allnoconfig` instead of `--defconfig`: nothing is on unless From 1a4e3ba675665291d89655f5de9da56fbb98e5ba Mon Sep 17 00:00:00 2001 From: Manuel de Brito Fontes Date: Fri, 18 Sep 2026 18:24:09 -0300 Subject: [PATCH 3/3] machine: a restore opens its template read-only memory-backend-file with share=off still opened the file read-write. So every VM restored from a template could write the file the next restore reads, and a QEMU that does not own the file could not open it at all. A restore now passes readonly=on,rom=off: the file is opened read-only, and the guest's RAM stays writable in its private copy. verify:args gains a restore from a 0444 file, which QEMU 11.1.1 refused with EACCES before this change. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_019GHRRFvbnZi5q3KSTvyW5a --- machine/machine.go | 18 ++++++++++++------ machine/machine_test.go | 31 +++++++++++++++++++++++++++++++ machine/qemu_accepts_test.go | 21 +++++++++++++++++++++ 3 files changed, 64 insertions(+), 6 deletions(-) diff --git a/machine/machine.go b/machine/machine.go index b9674cd..1d6106e 100644 --- a/machine/machine.go +++ b/machine/machine.go @@ -207,8 +207,9 @@ type Memory struct { File string // Shared maps that file MAP_SHARED. A VM being frozen into a template needs // this — the pages it dirties must reach the file the restores will read. A - // VM restoring from one passes false, mapping the same file MAP_PRIVATE: it - // sees the template's memory and anything it writes stays private to it. + // VM restoring from one passes false, opening the same file read-only and + // mapping it MAP_PRIVATE: it sees the template's memory and anything it writes + // stays private to it. // That is the whole copy-on-write story, and it is why one template file can // serve many VMs without being copied. Shared bool @@ -642,13 +643,18 @@ func (s Spec) Args() ([]string, error) { } if s.Memory.File != "" { - share := "off" + // A restore opens the file read-only. QEMU otherwise opens it read-write even to + // map it private, so every VM restored from a template could write the template + // the next one restores from — and a QEMU that is not root cannot open it at all, + // the file being its host's and not the VM's. rom=off keeps the guest's RAM + // writable: its writes land in its private copy, which is what share=off meant. + backing := "share=off,readonly=on,rom=off" if s.Memory.Shared { - share = "on" + backing = "share=on" } args = append(args, "-object", - fmt.Sprintf("memory-backend-file,id=%s,size=%dM,mem-path=%s,share=%s", - MemoryBackendID, s.Memory.SizeMB, s.Memory.File, share)) + fmt.Sprintf("memory-backend-file,id=%s,size=%dM,mem-path=%s,%s", + MemoryBackendID, s.Memory.SizeMB, s.Memory.File, backing)) } // S3 and S4 are suspend states this machine cannot come back from and that a diff --git a/machine/machine_test.go b/machine/machine_test.go index d06bf67..56baa7a 100644 --- a/machine/machine_test.go +++ b/machine/machine_test.go @@ -210,6 +210,37 @@ func TestMemoryFileChangesTheShape(t *testing.T) { } } +// A template's source maps its file shared and writes it; a restore maps the same file +// private and opens it read-only, so no VM restored from a template can change it for the +// next, and a QEMU that does not own the file can still restore from it. +func TestARestoreOpensItsTemplateReadOnly(t *testing.T) { + for _, tc := range []struct { + name string + shared bool + want string + refuses string + }{ + {"the template's source", true, "mem-path=/tmp/pc.ram,share=on", "readonly=on"}, + {"a restore", false, "mem-path=/tmp/pc.ram,share=off,readonly=on,rom=off", "rom=on"}, + } { + t.Run(tc.name, func(t *testing.T) { + s := spec(t) + s.Memory.File, s.Memory.Shared = "/tmp/pc.ram", tc.shared + args, err := s.Args() + if err != nil { + t.Fatal(err) + } + line := strings.Join(args, " ") + if !strings.Contains(line, tc.want) { + t.Errorf("the memory backend is not %q: %s", tc.want, line) + } + if strings.Contains(line, tc.refuses) { + t.Errorf("the memory backend says %q: %s", tc.refuses, line) + } + }) + } +} + // The whole point of the fingerprint: two machines may exchange templates only // if the binary, the kernel, the initrd and the shape all agree. Each of these // changes is one a caller could make without noticing. diff --git a/machine/qemu_accepts_test.go b/machine/qemu_accepts_test.go index 695174d..8849f57 100644 --- a/machine/qemu_accepts_test.go +++ b/machine/qemu_accepts_test.go @@ -138,6 +138,17 @@ func TestQEMUAcceptsEveryArgument(t *testing.T) { if err := os.Truncate(memFile, 512<<20); err != nil { t.Fatal(err) } + // A published template, which a restoring QEMU may read and not write. + sealedMemFile := filepath.Join(t.TempDir(), "sealed-memory") + if err := os.WriteFile(sealedMemFile, nil, 0o600); err != nil { + t.Fatal(err) + } + if err := os.Truncate(sealedMemFile, 512<<20); err != nil { + t.Fatal(err) + } + if err := os.Chmod(sealedMemFile, 0o444); err != nil { + t.Fatal(err) + } base := func() Spec { return Spec{ @@ -177,6 +188,16 @@ func TestQEMUAcceptsEveryArgument(t *testing.T) { s.IncomingDefer = true return s }, + }, { + // Restored from a file it may not write: a published template, which belongs to + // the host and not to the VM. Root would open it anyway, so under root this case + // says nothing; the gate runs as a user. + name: "restore target, read-only template", + spec: func(s Spec) Spec { + s.Memory.File = sealedMemFile + s.IncomingDefer = true + return s + }, }, { // A memory ceiling, which is what adds virtio-mem and its second memory // backend, and a vCPU ceiling, which is what adds maxcpus=.