diff --git a/.github/workflows/image.yml b/.github/workflows/image.yml index b32bdcd..369a3f1 100644 --- a/.github/workflows/image.yml +++ b/.github/workflows/image.yml @@ -46,6 +46,11 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # Without this, `--cache-to type=gha` below is a silent no-op: the tokens BuildKit + # needs reach JavaScript actions and not `run:` steps. See the long note in qemu.yml. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # image/build.sh asserts what came out on the finished filesystem rather than # trusting the build that made it: that there is an /sbin/init and a /bin/sh, that # the directories the guest mounts over exist, that no identity is baked in, and @@ -59,15 +64,11 @@ jobs: # the build container's, and image/build.sh stops when it is not there. - name: Build e2fsprogs run: | - task e2fsprogs:build \ - E2FSPROGS_CACHE_FROM=type=gha \ - E2FSPROGS_CACHE_TO=type=gha,mode=max + CACHE_BACKEND=gha task e2fsprogs:build - name: Build the base image run: | - task image:build \ - IMAGE_CACHE_FROM=type=gha \ - IMAGE_CACHE_TO=type=gha,mode=max + CACHE_BACKEND=gha task image:build - name: What this image is run: | diff --git a/.github/workflows/kernel.yml b/.github/workflows/kernel.yml index de51e64..45f6271 100644 --- a/.github/workflows/kernel.yml +++ b/.github/workflows/kernel.yml @@ -44,27 +44,28 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # Without this, `--cache-to type=gha` below is a silent no-op: the tokens BuildKit + # needs reach JavaScript actions and not `run:` steps. See the long note in qemu.yml. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # Ends in `task kernel:verify`, which opens the ELF and checks the Xen PVH notes # survived the strip. Without them QEMU has no entry point into this kernel and the # failure is a VM that does not start, in whichever lane boots one next. - name: Build the kernel run: | - task kernel:build \ - KERNEL_CACHE_FROM=type=gha \ - KERNEL_CACHE_TO=type=gha,mode=max \ - KERNEL_NPROC=4 + CACHE_BACKEND=gha task kernel:build KERNEL_NPROC=4 - # Compared, now, against the last released machine — which is the reference this step - # used to say it did not have. It does: every release publishes its machine.env as an - # asset, and the kernel's checksum is one of the two numbers in it that decide whether - # a template still matches. + # Compared against the last released machine, which every release publishes as a + # machine.env asset. The kernel's checksum is one of the two numbers in it that decide + # whether a template still matches. # # So a pull request that changes the config, or bumps the Debian toolchain the kernel # is compiled with, says on its own summary that it invalidates every template in # existence. That is not a reason to reject it — it is a release this repository makes - # deliberately, and the step exits 0 either way — but it was previously a sentence - # nobody wrote down, and the last time it happened it was a comment in this directory's - # Dockerfile that nobody expected to rebuild anything. + # deliberately, and the step exits 0 either way. It is a reason to know: the change + # that does this need not look like much, and a comment edited in this directory's + # Dockerfile is enough. # # It speaks only for the kernel: this workflow builds no QEMU, and the verdict says so # rather than implying it checked the whole machine. diff --git a/.github/workflows/qemu.yml b/.github/workflows/qemu.yml index 0e4e535..10c9c88 100644 --- a/.github/workflows/qemu.yml +++ b/.github/workflows/qemu.yml @@ -9,9 +9,9 @@ name: QEMU on: # qemu/** and nothing else. The version pin is the ARG in qemu/Dockerfile, so a bump is a -# change under qemu/ and fires this; the root Taskfile holds only how many jobs to compile -# with and which cache to use, and used to be listed here — which meant every edit to it -# rebuilt QEMU, the kernel and the base image for nothing. +# change under qemu/ and fires this. The root Taskfile is deliberately not listed: it holds +# only how many jobs to compile with and which cache to use, so listing it would rebuild +# QEMU, the kernel and the base image for an edit that cannot change any of them. push: branches: [main] paths: @@ -58,6 +58,24 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # What makes `--cache-to type=gha` do anything at all. + # + # BuildKit writes to GitHub's cache service with $ACTIONS_RUNTIME_TOKEN and + # $ACTIONS_RESULTS_URL, and those are handed to JavaScript actions, not to `run:` + # steps. Docker's documentation is explicit — "If you invoke the `docker buildx` + # command manually from an inline step, then the variables must be manually exposed" + # — and names this action for it. Every build here runs through `task`, which is a + # `run:` step. + # + # Without it the flags are accepted, nothing is stored, and nothing says so: the + # builds pass and take the cold time, every run. That failure is invisible in a log. + # Where it shows is `gh api repos/{owner}/{repo}/actions/cache/usage` — a repository + # that builds QEMU and a kernel and reports a handful of megabytes is caching neither. + # + # Remove this step and nothing breaks; everything just gets slow again. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # The pinned version comes from the Taskfile, never from a literal here: two places # to bump is how CI and production end up on different QEMUs. # @@ -73,10 +91,7 @@ jobs: - name: Build QEMU and verify what came out run: | - task qemu:build \ - QEMU_CACHE_FROM=type=gha \ - QEMU_CACHE_TO=type=gha,mode=max \ - QEMU_JOBS=4 + CACHE_BACKEND=gha task qemu:build QEMU_JOBS=4 # vhost-vsock is a kernel module, and the machine's vsock device cannot be created # without /dev/vhost-vsock. The runner has the node and gives the user no access to @@ -145,6 +160,12 @@ jobs: # Two tags: the QEMU version, which is what everything pins, and the commit, which is # what makes a specific build reachable when the version tag has moved because a flag # changed. + # + # Two --cache-from and one --cache-to, which is not symmetry gone wrong. This target + # sits on top of the builder stage the step above just compiled, so it reads the + # `qemu` scope to get it and writes only its own — sharing one scope would have the + # two builds overwriting each other's cache, which is the default behaviour that made + # every scope in this repository distinct in the first place. - name: Publish the runtime image if: github.event_name != 'pull_request' run: | @@ -152,8 +173,9 @@ jobs: --file qemu/Dockerfile \ --target runtime \ --platform linux/amd64 \ - --cache-from type=gha \ - --cache-to type=gha,mode=max \ + --cache-from type=gha,scope=qemu \ + --cache-from type=gha,scope=qemu-image \ + --cache-to type=gha,scope=qemu-image,mode=max \ --build-arg QEMU_VERSION=${{ steps.qemu.outputs.version }} \ --build-arg JOBS=4 \ --label org.opencontainers.image.source=https://github.com/${{ github.repository }} \ diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index dcb2004..fa263ce 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -26,12 +26,12 @@ permissions: contents: write packages: write -# One release at a time, across every ref. The group used to be per-ref, which serialised -# two runs on the same tag and nothing else: a workflow_dispatch on main and a tag push are -# different refs, so both counted the tags already taken today, both generated .01, and the -# loser found out at `git tag` — after six hours of building. There is no reason to build -# two releases at once, and the version is decided at the start of a run and pushed at the -# end, so the window is the whole build. +# One release at a time, across every ref, and not per-ref. Per-ref serialises two runs on +# the same tag and nothing else: a workflow_dispatch on main and a tag push are different +# refs, so both count the tags already taken today, both generate .01, and the loser finds +# out at `git tag` — after six hours of building. The version is decided at the start of a +# run and pushed at the end, so that window is the whole build, and there is no reason to +# build two releases at once anyway. concurrency: group: release cancel-in-progress: false @@ -63,6 +63,11 @@ jobs: - name: Set up Buildx uses: docker/setup-buildx-action@v4 + # Without this, `--cache-to type=gha` below is a silent no-op: the tokens BuildKit + # needs reach JavaScript actions and not `run:` steps. See the long note in qemu.yml. + - name: Expose the Actions cache to buildx + uses: crazy-max/ghaction-github-runtime@v4 + # Three sources, in order: what the caller typed, the tag that triggered this, or a # new one for today. The generated form is deliberately not `date +%s` or a commit # hash — it has to be readable, sortable, and the same shape as one typed by hand. @@ -107,26 +112,28 @@ jobs: - name: Build the release env: VERSION: ${{ steps.version.outputs.version }} - QEMU_CACHE_FROM: type=gha - QEMU_CACHE_TO: type=gha,mode=max - KERNEL_CACHE_FROM: type=gha - KERNEL_CACHE_TO: type=gha,mode=max - E2FSPROGS_CACHE_FROM: type=gha - E2FSPROGS_CACHE_TO: type=gha,mode=max - IMAGE_CACHE_FROM: type=gha - IMAGE_CACHE_TO: type=gha,mode=max + # The same scopes the per-artefact workflows write on main, decided in + # Taskfile.yml. That is the whole point of them being named there: a release + # reads what the lane on main already paid to compile, instead of starting cold + # because a workflow file spelled the scope differently. + # + # This run's own writes are mostly wasted — it is triggered by a tag, and a cache + # written under one tag ref cannot be read from another tag or from a branch. The + # reads are what matter, and those come from the default branch. + CACHE_BACKEND: gha run: task release QEMU_JOBS=4 KERNEL_NPROC=4 # The machine's identity, and what it costs whoever holds templates. # - # "why did every template in the fleet stop matching?" used to be answered by opening - # two release pages and comparing sixty-four hex digits by eye, which is a thing - # nobody does and therefore an answer nobody had. hack/fingerprint-diff fetches the - # previous release's machine.env and states the consequence in a sentence, at the top - # of the notes, where a consumer deciding whether to take this release reads it. + # The alternative to hack/fingerprint-diff is opening two release pages and comparing + # sixty-four hex digits by eye, which is a thing nobody does — so the answer to "why + # did every template in the fleet stop matching?" is one nobody has. The script + # fetches the previous release's machine.env and states the consequence in a sentence, + # at the top of the notes, where a consumer deciding whether to take this release + # reads it. # # It exits 0 on every verdict: a changed machine is a release this repository makes on - # purpose. The point is that it can no longer be made quietly. + # purpose. The point is not to prevent it but to keep it from being quiet. - name: What this release is, and what it breaks id: manifest env: diff --git a/README.md b/README.md index 1f85ae2..68780a0 100644 --- a/README.md +++ b/README.md @@ -69,9 +69,9 @@ One tarball: `task build` writes that same tree into `_output/`, byte for byte the layout above, and `machine.Open` reads either. There is one layout: nothing rearranges the files on the way -out of a build, into a tarball or into a consumer, because the three used to differ and -what fell out of the translation between them was a path that existed and held the -previous release's kernel. +out of a build, into a tarball or into a consumer. Let the three differ and what falls out +of the translation between them is a path that exists and holds the previous release's +kernel. ```go rel, err := machine.Open("/usr/share/spin-stack") // says which file is missing, if one is @@ -174,10 +174,9 @@ It boots this QEMU and this kernel over a throwaway qcow2 overlay on `rootfs.qco It boots with no initrd at all: `root=/dev/vda rw init=/sbin/init`, which the kernel can serve because virtio-blk and ext4 are built in and `image/build.sh` writes a partitionless -filesystem. There was a debug initramfs here until 2026-09-10 — static Go that mounted the -API filesystems, found the root disk and exec'd — and it was a second init, doing what the -init a consumer brings already does. What removed the need for it was turning `systemd-udevd` -back on. +filesystem. There is no debug initramfs: one that mounted the API filesystems, found the +root disk and exec'd would be a second init, doing what the init a consumer brings already +does. What removes the need for one is `systemd-udevd` being on. Three things the shell has found, all of them true of the image and none of them visible from outside: @@ -190,12 +189,12 @@ from outside: its own with the dependency removed. Masking bought nothing measurable and cost a ten-second `dev-ttyS0.device` timeout on every boot that had no initrd; the numbers are in `optimize-systemd.sh`, where the decision is. -- **`ssh.service` used to fail five times and give up.** The image ships no host keys — they - are identity — and the distribution's `sshd-keygen.service` carries - `ConditionFirstBoot=yes`, so it did not run before `sshd` was asked to validate a - configuration with no keys. Fixed by generating them at boot instead - (`spin-machine-sshd-keygen.service`): every boot here *is* a first boot, since the root - filesystem is a fresh overlay, so the keys live exactly as long as the VM does. +- **`sshd` needs host keys generated at boot.** The image ships none — they are identity — + and the distribution's `sshd-keygen.service` carries `ConditionFirstBoot=yes`, so it does + not run before `sshd` is asked to validate a configuration with no keys, and + `ssh.service` fails five times and gives up. `spin-machine-sshd-keygen.service` generates + them instead: every boot here *is* a first boot, since the root filesystem is a fresh + overlay, so the keys live exactly as long as the VM does. - **Two thirds of the boot was systemd asking the console questions.** The kernel execs `/sbin/init` at 59 ms and systemd's first log line arrived at 852, with nothing running in between. It was a terminfo query and a terminal reset, 334 ms of timeout each, waiting for @@ -283,9 +282,9 @@ which is a different fingerprint. It moves on purpose, not when Debian publishes release. And the build is reproducible: two builds from the same inputs produce the same `vmlinux`, -byte for byte. It was not until 2026-09-14: bookworm's pahole 1.24 encoded `.BTF` with make's -`-j` in whatever order its threads finished, and every release had a new fingerprint whether -or not the kernel had changed. trixie's 1.30 is deterministic with `-j`. +byte for byte. That needs trixie's pahole 1.30, which is deterministic with `-j`. Bookworm's +1.24 encodes `.BTF` in whatever order make's `-j` threads finish, which gives every release a +new fingerprint whether or not the kernel changed. ### Base image diff --git a/Taskfile.yml b/Taskfile.yml index 884013b..17cdc1e 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -20,10 +20,10 @@ # qemu/Dockerfile and nobody reads three hundred lines to change one flag. # # What is deliberately NOT here: the software that runs inside a guest. This repository -# builds a machine and knows nothing about what boots on it, and there is no longer an -# exception to that — the debug initramfs that used to be one was a second init, doing what -# the init a consumer already brings does, and the machine boots straight into /sbin/init -# without it. +# builds a machine and knows nothing about what boots on it, and there is no exception to +# that. A debug initramfs is the one that keeps being proposed: it is a second init, doing +# what the init a consumer already brings does, and the machine boots straight into +# /sbin/init without one. version: '3' @@ -65,17 +65,16 @@ vars: # One version for all three artefacts. See the header. # - # --dirty, and it is the whole point of the line. This used to be --exact-match with a - # bare `git rev-parse --short HEAD` behind it, neither of which says anything about - # uncommitted changes — so `task release` on a tagged commit with edits in the tree - # produced spin-machine-v20260914.03-linux-x86_64.tar.gz with version=v20260914.03 in its - # manifest: the same name and the same declared identity as the release that was actually - # published, around different bytes. machine.env is how a host explains why its templates - # stopped matching, and that made it lie. + # --dirty, and it is the whole point of the line. Without it, `task release` on a tagged + # commit with edits in the tree produces spin-machine-v20260914.03-linux-x86_64.tar.gz + # with version=v20260914.03 in its manifest: the same name and the same declared identity + # as the release that was actually published, around different bytes. machine.env is how + # a host explains why its templates stopped matching, so it is the one file that must not + # be able to claim an identity the bytes do not have. # - # The fallback is --tags --always --dirty rather than a bare hash, which is what - # hack/release already defaulted to on its own: v20260914.03-1-g86d2386 says which release - # this is one commit past, where 86d2386 says nothing until somebody looks it up. + # The fallback is --tags --always --dirty rather than a bare hash, matching what + # hack/release defaults to on its own: v20260914.03-1-g86d2386 says which release this is + # one commit past, where 86d2386 says nothing until somebody looks it up. VERSION: sh: | echo "${VERSION:-$(git describe --tags --exact-match --dirty 2>/dev/null \ @@ -97,21 +96,38 @@ vars: KERNEL_ARCH: '{{.KERNEL_ARCH | default "x86_64"}}' KERNEL_NPROC: '{{.KERNEL_NPROC | default 8}}' - # BuildKit cache backends. Local by default; CI overrides them with `type=gha` so the - # build is not duplicated in a workflow file — the Dockerfiles and the flags stay in one - # place, which is the rule these files exist for. + # Where BuildKit reads and writes its cache: a directory under the repository for a + # developer, GitHub's cache service in CI. One name to switch — `CACHE_BACKEND=gha` — and + # every scope below is decided here, next to the builds they belong to, rather than as + # eight cache strings pasted into workflow files. + # + # The scopes are why this is a var and not a string in a workflow. `type=gha` defaults to + # the scope `buildkit`, and Docker's own documentation says what that means for a + # repository with more than one build: "each build will overwrite the cache of the + # previous, leaving only the final cache." Four builds sharing that default is four builds + # each destroying the last one's cache and then missing on the next run. The names have to + # be distinct, and — this is the part that is easy to get wrong — they have to be the + # *same* names in the per-artefact workflows and in the release, or a release warms + # nothing that the lane on main already paid for. + # + # `gha` needs one thing this file cannot give it — see the note in + # .github/workflows/qemu.yml. The variables that authorise a write to the cache service + # are not in a plain `run:` step's environment, and without them `--cache-to type=gha` is + # accepted, stores nothing, and says nothing. + CACHE_BACKEND: '{{.CACHE_BACKEND | default "local"}}' BUILDKIT_CACHE_DIR: '{{.BUILDKIT_CACHE_DIR | default (printf "%s/.cache/buildkit" .ROOT_DIR)}}' - QEMU_CACHE_FROM: '{{.QEMU_CACHE_FROM | default (printf "type=local,src=%s/qemu" .BUILDKIT_CACHE_DIR)}}' - QEMU_CACHE_TO: '{{.QEMU_CACHE_TO | default (printf "type=local,dest=%s/qemu,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' - KERNEL_CACHE_FROM: '{{.KERNEL_CACHE_FROM | default (printf "type=local,src=%s/kernel" .BUILDKIT_CACHE_DIR)}}' - KERNEL_CACHE_TO: '{{.KERNEL_CACHE_TO | default (printf "type=local,dest=%s/kernel,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' - E2FSPROGS_CACHE_FROM: '{{.E2FSPROGS_CACHE_FROM | default (printf "type=local,src=%s/e2fsprogs" .BUILDKIT_CACHE_DIR)}}' - E2FSPROGS_CACHE_TO: '{{.E2FSPROGS_CACHE_TO | default (printf "type=local,dest=%s/e2fsprogs,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' - # The container that builds the base image. Small — mkosi, qemu-utils — but it - # was the one buildx invocation here with no cache at all, so every CI run reinstalled it - # from the archive before any image work started. - IMAGE_CACHE_FROM: '{{.IMAGE_CACHE_FROM | default (printf "type=local,src=%s/image" .BUILDKIT_CACHE_DIR)}}' - IMAGE_CACHE_TO: '{{.IMAGE_CACHE_TO | default (printf "type=local,dest=%s/image,mode=max,compression=zstd" .BUILDKIT_CACHE_DIR)}}' + + QEMU_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=qemu{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/qemu{{end}}' + QEMU_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=qemu,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/qemu,mode=max,compression=zstd{{end}}' + KERNEL_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=kernel{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/kernel{{end}}' + KERNEL_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=kernel,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/kernel,mode=max,compression=zstd{{end}}' + E2FSPROGS_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=e2fsprogs{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/e2fsprogs{{end}}' + E2FSPROGS_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=e2fsprogs,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/e2fsprogs,mode=max,compression=zstd{{end}}' + # The container that builds the base image. Small — mkosi, qemu-utils — and cached like + # the rest: uncached, every CI run reinstalls it from the archive before any image work + # starts. + IMAGE_CACHE_FROM: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=image{{else}}type=local,src={{.BUILDKIT_CACHE_DIR}}/image{{end}}' + IMAGE_CACHE_TO: '{{if eq .CACHE_BACKEND "gha"}}type=gha,scope=image,mode=max{{else}}type=local,dest={{.BUILDKIT_CACHE_DIR}}/image,mode=max,compression=zstd{{end}}' # mkosi's package cache, which is not a BuildKit cache because mkosi does not run under # BuildKit (see image/Dockerfile). A rebuild of the base image is ~1.5 GB of apt. diff --git a/boot/bench_test.go b/boot/bench_test.go index 1cdb271..8078f15 100644 --- a/boot/bench_test.go +++ b/boot/bench_test.go @@ -32,8 +32,8 @@ type variant struct { // implemented by systemd-debug-generator, which optimize-systemd.sh symlinks to /dev/null, // so it parses, reaches /proc/cmdline and does nothing: verified 2026-09-10 with // `systemd.mask=chrony.service` on the command line and `systemctl is-active chrony` -// answering `active`. Everything this repository concluded from that parameter was concluded -// from a boot in which nothing had been masked. +// answering `active`. Anything concluded from that parameter is concluded from a boot in +// which nothing was masked. var udevUnits = []string{ "systemd-udevd.service", "systemd-udevd-control.socket", diff --git a/boot/phases.go b/boot/phases.go index 5358146..74aa0a3 100644 --- a/boot/phases.go +++ b/boot/phases.go @@ -2,10 +2,10 @@ // Package boot measures what a boot of this machine costs, from the host. // -// It exists because the number this repository had been quoting was systemd's own -// `Startup finished`, which begins counting when the kernel hands over. Exec'ing QEMU, the -// firmware and loading the kernel are all before it; a login prompt is after it. Neither end -// had ever been measured, and both are part of what an operator waits for. +// It exists because the number a machine reports for itself — systemd's `Startup +// finished` — begins counting when the kernel hands over. Exec'ing QEMU, the firmware and +// loading the kernel are all before it; a login prompt is after it. Neither end is in that +// number, and both are part of what an operator waits for. // // The package is split so that the part that can be wrong is testable without a VM: Watch // turns a stream of console bytes into phase timings, and Percentile turns a set of runs @@ -91,8 +91,8 @@ const ( // the VMM is doing. // // None of that says the firmware is cheap, and it is not: its work comes after the - // banner, and it had never been measured until 2026-09-13. QEMU exec to the kernel's PVH - // entry is 43.9 and 45.1 ms, 40 boots in each of two runs, against 34.6 to the banner — + // banner. Measured 2026-09-13, QEMU exec to the kernel's PVH entry is 43.9 and 45.1 ms, + // 40 boots in each of two runs, against 34.6 to the banner — // so SeaBIOS's POST (PCI enumeration, SMM and MTRR setup, the ACPI table loader, a scan of // storage and input the machine does not have, then pvh.bin) is ~10 ms. Two ways of // spending less on it, and what each came to: @@ -127,8 +127,8 @@ const ( // PID1 is the kernel handing over. The kernel must be printing at loglevel 7 for this // to appear at all, so it is absent from a quiet boot rather than zero. PID1 - // Started is systemd's `Startup finished`, the number this repository used to quote in - // full. Kept so the old number stays comparable to the new ones. + // Started is systemd's `Startup finished`: the number the machine reports for itself, + // and the one most projects quote. Kept so these timings stay comparable to it. Started // Usable is a login prompt. It is the only one of these that is not the machine making // a claim about itself, and it is the one an operator is actually waiting for. diff --git a/image/Taskfile.yml b/image/Taskfile.yml index f7ac9df..3ce56ed 100644 --- a/image/Taskfile.yml +++ b/image/Taskfile.yml @@ -89,12 +89,12 @@ tasks: # there is nothing for one to do: mounting root is the kernel's job here and # systemd mounts the rest. # - # This used to boot a debug initramfs, for two reasons that were both real and are + # Booting a debug initramfs here is a second init, and the two reasons for one are # both gone. It mounted /proc, /sys, devtmpfs and devpts, which systemd does itself # and only a bare `init=/bin/sh` ever needed. And it wrote a symlink enabling a - # serial login, because this image masked systemd-udevd, so no dev-ttyS0.device - # unit existed and serial-getty@ttyS0's BindsTo could never be satisfied. udev - # runs now; the getty starts on its own. Measured 2026-09-10: 207ms to a login + # serial login, which was needed while this image masked systemd-udevd: no + # dev-ttyS0.device unit existed and serial-getty@ttyS0's BindsTo could never be + # satisfied. udev runs; the getty starts on its own. Measured 2026-09-10: 207ms to a login # prompt this way, against 163ms through the initramfs — 44ms for a second init # doing what the first one does. # diff --git a/image/build.sh b/image/build.sh index dabe333..f5e8417 100755 --- a/image/build.sh +++ b/image/build.sh @@ -154,8 +154,8 @@ echo "==> $(wc -l < "$SHARE/packages.txt") packages recorded" # its own — but the filesystem in it is: an overlay holds this same ext4, mounted read-write # and grown onto the disk the overlay was made at. A copy of that overlay taken while the guest # writes, or the overlay of a machine that died, is a filesystem interrupted mid-update, and -# without a journal nothing puts its metadata back together at the next mount. It was built -# without one until 2026-09-14, and a copy taken seconds after a grow would not mount: +# without a journal nothing puts its metadata back together at the next mount. Measured +# 2026-09-14: built without one, a copy taken seconds after a grow does not mount — # "structure needs cleaning", a corrupt group descriptor. # # Each overlay replays and writes its own journal; the blocks it writes land in that overlay, diff --git a/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh b/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh index 7fc5320..ad8ccb5 100755 --- a/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh +++ b/image/mkosi.extra/usr/local/lib/spin-base/configure-system.sh @@ -32,8 +32,7 @@ sed -i 's/#PubkeyAuthentication yes/PubkeyAuthentication yes/' /etc/ssh/sshd_con # # Without this line the control plane signs five-minute certificates that nothing on earth # accepts, and every login quietly falls back to a long-lived key in authorized_keys that -# nothing ever rewrites. That was the state of things until 2026-09-11, and it was invisible -# because logins kept working. +# nothing ever rewrites. It is invisible when it happens, because logins keep working. mkdir -p /etc/ssh/sshd_config.d cat <<'EOF' > /etc/ssh/sshd_config.d/10-spin-user-ca.conf TrustedUserCAKeys /etc/ssh/spin_user_ca.pub @@ -116,8 +115,8 @@ esac MOTD chmod 0755 /etc/update-motd.d/00-spin-boot -# pam_motd stays enabled, including for ssh, where it had been commented out. There is -# something worth printing now. +# pam_motd stays enabled, including for ssh, where the distribution ships it commented +# out. There is something worth printing. sed -i 's/^#\(session.*pam_motd.so\)/\1/' /etc/pam.d/sshd diff --git a/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh b/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh index a2c3c0c..3d55c56 100755 --- a/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh +++ b/image/mkosi.extra/usr/local/lib/spin-base/optimize-systemd.sh @@ -37,10 +37,6 @@ MASK_UNITS=( # spin-machine-console.service, and the decision left open there. # * A hot-plugged CPU or memory block never produced an add event, so nothing could # act on one — see 40-spin-hotadd.rules, which is the thing that acts on it. - # - # The numbers here used to compare a boot with the debug initramfs against one without. - # That initramfs was deleted the same day, so the comparison it made cannot be re-run; - # what is above is the one the harness measures now. # Time sync - handled by host systemd-timesyncd.service @@ -139,10 +135,6 @@ MASK_UNITS=( # tmpfiles-setup (19ms + 13ms + 11ms). The directories it would create are already # there: /tmp is in the image at mode 1777 (image/build.sh makes it), /run is a tmpfs # systemd mounts itself before any unit runs, and /dev is devtmpfs with udev on top. - # - # The reason recorded here until 2026-09-10 was that an init outside this repository - # created them, which was both wrong — it named a consumer, and this machine now boots - # root=/dev/vda with no initrd at all — and unfalsifiable from inside this tree. systemd-tmpfiles-setup.service systemd-tmpfiles-setup-dev.service systemd-tmpfiles-setup-dev-early.service diff --git a/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service b/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service index cac7295..62d624b 100644 --- a/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service +++ b/image/mkosi.extra/usr/local/lib/spin-base/rootfs/usr/lib/systemd/system/spin-machine-console.service @@ -13,15 +13,15 @@ # systemd gave up. # # Why this and not the distribution's serial-getty@.service: it carries -# `BindsTo=dev-%i.device`, a .device unit exists only if udev announced it, and this image -# used to mask systemd-udevd. Enabling serial-getty then got +# `BindsTo=dev-%i.device`, and a .device unit exists only if udev announced it. With +# systemd-udevd masked, enabling serial-getty gets # `[DEPEND] Dependency failed for serial-getty@ttyS0.service`, and a drop-in clearing -# BindsTo did not lift it either. This unit has no device dependency to fail. +# BindsTo does not lift it either. This unit has no device dependency to fail. # -# THAT REASON EXPIRED ON 2026-09-10 and left a hole this file is the wrong place to close. -# udev runs now, so serial-getty@ttyS0 works — and systemd-getty-generator instantiates it +# THAT REASON NO LONGER HOLDS, and it leaves a hole this file is the wrong place to close. +# udev runs, so serial-getty@ttyS0 works — and systemd-getty-generator instantiates it # on its own from console=ttyS0, with nothing in optimize-systemd.sh disabling that -# generator. So the split this file is built around no longer exists: production gets a +# generator. So the split this file is built around does not exist: production gets a # login prompt on the console it prints its kernel log to, which is what the paragraph above # says production must not have, and it gets it without anybody enabling anything. # diff --git a/kernel/Dockerfile b/kernel/Dockerfile index 83dcb2c..2e7a9c0 100644 --- a/kernel/Dockerfile +++ b/kernel/Dockerfile @@ -340,10 +340,10 @@ EOT # ============================================ # # The tree this stage writes is the release tree: kernel/vmlinux, at the path a release -# carries it and the path machine.Release reads it from. It used to extract to the root and -# let hack/release do the moving, which meant the same four files had one layout here, a -# second in the tarball and a third wherever a consumer rearranged them into — with a -# translation step between each pair, and each of those a place to be wrong. +# carries it and the path machine.Release reads it from. Extracting to the root and letting +# hack/release do the moving gives the same four files one layout here, a second in the +# tarball and a third wherever a consumer rearranges them into — with a translation step +# between each pair, and each of those a place to be wrong. FROM scratch AS kernel COPY --from=kernel-build /build/vmlinux /kernel/vmlinux COPY --from=kernel-build /build/kernel-config /kernel/kernel-config diff --git a/machine/cmdline.go b/machine/cmdline.go index acd4089..d37cf8b 100644 --- a/machine/cmdline.go +++ b/machine/cmdline.go @@ -130,13 +130,13 @@ func (c Cmdline) String() string { // The tick stops when a CPU has nothing to do, which is what CONFIG_NO_HZ_IDLE // is compiled in for. // - // It used to be turned off here, on the grounds that a short-lived VM never - // amortises the tickless machinery's setup cost and that the timer interrupt - // it saves is cheap on a guest. Both halves were wrong when measured - // (2026-09-09). The interrupt is not cheap: a guest kernel at CONFIG_HZ=1000 + // Turning it off here is the tempting change, on the grounds that a short-lived + // VM never amortises the tickless machinery's setup cost and that the timer + // interrupt it saves is cheap on a guest. Both halves are wrong, measured + // 2026-09-09. The interrupt is not cheap: a guest kernel at CONFIG_HZ=1000 // with the tick forced on burns 1.7% of a core per vCPU doing nothing at all, // because every one of those thousand wakeups a second is a vmexit. And the - // setup cost it was buying back does not show: with the tick left alone, a + // setup cost it would buy back does not show: with the tick left alone, a // machine's boot and a workspace's restore both measured what they did before // — 821-853 ms to boot and build a template, 113-221 ms to a usable workspace, // against 843-846 ms and 124-216 ms with nohz=off. diff --git a/machine/machine.go b/machine/machine.go index b9674cd..1d6106e 100644 --- a/machine/machine.go +++ b/machine/machine.go @@ -207,8 +207,9 @@ type Memory struct { File string // Shared maps that file MAP_SHARED. A VM being frozen into a template needs // this — the pages it dirties must reach the file the restores will read. A - // VM restoring from one passes false, mapping the same file MAP_PRIVATE: it - // sees the template's memory and anything it writes stays private to it. + // VM restoring from one passes false, opening the same file read-only and + // mapping it MAP_PRIVATE: it sees the template's memory and anything it writes + // stays private to it. // That is the whole copy-on-write story, and it is why one template file can // serve many VMs without being copied. Shared bool @@ -642,13 +643,18 @@ func (s Spec) Args() ([]string, error) { } if s.Memory.File != "" { - share := "off" + // A restore opens the file read-only. QEMU otherwise opens it read-write even to + // map it private, so every VM restored from a template could write the template + // the next one restores from — and a QEMU that is not root cannot open it at all, + // the file being its host's and not the VM's. rom=off keeps the guest's RAM + // writable: its writes land in its private copy, which is what share=off meant. + backing := "share=off,readonly=on,rom=off" if s.Memory.Shared { - share = "on" + backing = "share=on" } args = append(args, "-object", - fmt.Sprintf("memory-backend-file,id=%s,size=%dM,mem-path=%s,share=%s", - MemoryBackendID, s.Memory.SizeMB, s.Memory.File, share)) + fmt.Sprintf("memory-backend-file,id=%s,size=%dM,mem-path=%s,%s", + MemoryBackendID, s.Memory.SizeMB, s.Memory.File, backing)) } // S3 and S4 are suspend states this machine cannot come back from and that a diff --git a/machine/machine_test.go b/machine/machine_test.go index c586666..56baa7a 100644 --- a/machine/machine_test.go +++ b/machine/machine_test.go @@ -210,6 +210,37 @@ func TestMemoryFileChangesTheShape(t *testing.T) { } } +// A template's source maps its file shared and writes it; a restore maps the same file +// private and opens it read-only, so no VM restored from a template can change it for the +// next, and a QEMU that does not own the file can still restore from it. +func TestARestoreOpensItsTemplateReadOnly(t *testing.T) { + for _, tc := range []struct { + name string + shared bool + want string + refuses string + }{ + {"the template's source", true, "mem-path=/tmp/pc.ram,share=on", "readonly=on"}, + {"a restore", false, "mem-path=/tmp/pc.ram,share=off,readonly=on,rom=off", "rom=on"}, + } { + t.Run(tc.name, func(t *testing.T) { + s := spec(t) + s.Memory.File, s.Memory.Shared = "/tmp/pc.ram", tc.shared + args, err := s.Args() + if err != nil { + t.Fatal(err) + } + line := strings.Join(args, " ") + if !strings.Contains(line, tc.want) { + t.Errorf("the memory backend is not %q: %s", tc.want, line) + } + if strings.Contains(line, tc.refuses) { + t.Errorf("the memory backend says %q: %s", tc.refuses, line) + } + }) + } +} + // The whole point of the fingerprint: two machines may exchange templates only // if the binary, the kernel, the initrd and the shape all agree. Each of these // changes is one a caller could make without noticing. @@ -315,9 +346,9 @@ func TestFingerprintCoversTheDevicesPresentAtRestore(t *testing.T) { // A NIC is on the command line, so it is present when state is loaded and a machine // with one cannot restore from a template frozen without one. // - // This case said the opposite until 2026-09-09 — that a NIC must not change the + // The opposite is the tempting assertion — that a NIC must not change the // fingerprint, because "it is cold-plugged after a restore, so a VM would never find - // its template". That was true of disks and asserted of both. + // its template". That is true of disks, and does not carry across to NICs. withNIC := base withNIC.NICs = []NIC{{TapFD: 3, MAC: "52:54:00:00:00:01"}} got, err := withNIC.Fingerprint() diff --git a/machine/qemu_accepts_test.go b/machine/qemu_accepts_test.go index 695174d..8849f57 100644 --- a/machine/qemu_accepts_test.go +++ b/machine/qemu_accepts_test.go @@ -138,6 +138,17 @@ func TestQEMUAcceptsEveryArgument(t *testing.T) { if err := os.Truncate(memFile, 512<<20); err != nil { t.Fatal(err) } + // A published template, which a restoring QEMU may read and not write. + sealedMemFile := filepath.Join(t.TempDir(), "sealed-memory") + if err := os.WriteFile(sealedMemFile, nil, 0o600); err != nil { + t.Fatal(err) + } + if err := os.Truncate(sealedMemFile, 512<<20); err != nil { + t.Fatal(err) + } + if err := os.Chmod(sealedMemFile, 0o444); err != nil { + t.Fatal(err) + } base := func() Spec { return Spec{ @@ -177,6 +188,16 @@ func TestQEMUAcceptsEveryArgument(t *testing.T) { s.IncomingDefer = true return s }, + }, { + // Restored from a file it may not write: a published template, which belongs to + // the host and not to the VM. Root would open it anyway, so under root this case + // says nothing; the gate runs as a user. + name: "restore target, read-only template", + spec: func(s Spec) Spec { + s.Memory.File = sealedMemFile + s.IncomingDefer = true + return s + }, }, { // A memory ceiling, which is what adds virtio-mem and its second memory // backend, and a vCPU ceiling, which is what adds maxcpus=. diff --git a/qemu/Dockerfile b/qemu/Dockerfile index 43d04f8..598519f 100644 --- a/qemu/Dockerfile +++ b/qemu/Dockerfile @@ -392,9 +392,9 @@ RUN for b in /opt/qemu/bin/qemu-system-x86_64 /opt/qemu/bin/qemu-img \ # # Two binaries and five files on a base that exists only so a human can open a shell in it # and look around. There are no libraries to install: that is the whole difference static -# linking makes here. This stage used to be an Ubuntu with a hand-written apt list that had -# to be kept in step with what the builder linked against, and the way it was found to be -# wrong was a CI run. +# linking makes here. An Ubuntu base with a hand-written apt list would have to be kept in +# step with what the builder links against, and the way that is found to be wrong is a CI +# run. FROM alpine:${ALPINE_VERSION} AS runtime COPY --from=builder /opt/qemu/bin/qemu-system-x86_64 /usr/local/bin/qemu-system-x86_64 diff --git a/qemu/devices.mak b/qemu/devices.mak index 5e8d1fe..29445ec 100644 --- a/qemu/devices.mak +++ b/qemu/devices.mak @@ -11,14 +11,14 @@ # upstream shipped depends on whether some earlier build happened to configure. This file # is copied in and read on every build. # -# ## An allowlist, since 2026-09-10 +# ## An allowlist, not a denylist # -# This file used to include upstream's default.mak and switch off about forty symbols. That -# is a denylist, and it loses by construction: it can only remove what somebody thought to +# The obvious alternative is to include upstream's default.mak and switch off some forty +# symbols. That is a denylist, and it loses by construction: it can only remove what somebody thought to # name, and every QEMU release adds devices that arrive switched on. The binary it produced -# could instantiate ich9-ahci, ide-hd, sb16, isa-fdc, isa-parallel, VGA, virtio-vga, -# vfio-pci, intel-iommu and amd-iommu — audited on the shipped binary with `-device help`, -# not read off this file. +# binary it produces can instantiate ich9-ahci, ide-hd, sb16, isa-fdc, isa-parallel, VGA, +# virtio-vga, vfio-pci, intel-iommu and amd-iommu — audited on a shipped binary with +# `-device help`, not read off a file like this one. # # Now the build is configured `--without-default-devices`, which makes meson run # scripts/minikconf.py with `--allnoconfig` instead of `--defconfig`: nothing is on unless