diff --git a/NOTICE b/NOTICE index 556c19f..448a3ed 100644 --- a/NOTICE +++ b/NOTICE @@ -9,7 +9,8 @@ https://github.com/bonzini/qboot, commit 8ca302e86d685fa05b16e2b208888243da319941, under GPL-2.0 (see qemu/qboot/COPYING). write-pointer.patch modifies fw_cfg.c, include/fw_cfg.h and tables.c to implement WRITE_POINTER, DMA writes and error/bounds checks; pam.patch modifies hwsetup.c to -set q35's PAM registers in three configuration writes. The firmware built with +set q35's PAM registers in three configuration writes; mtrr.patch modifies main.c +to enable the MTRRs, write-back by default and the 32-bit PCI hole uncacheable. The firmware built with them ships in releases as qemu/qboot.bin, the patches beside it. The Apache-2.0 licence does not apply to those patches or COPYING. diff --git a/boot/pagecache_test.go b/boot/pagecache_test.go index f532aa1..dee8bf5 100644 --- a/boot/pagecache_test.go +++ b/boot/pagecache_test.go @@ -95,6 +95,7 @@ func TestPageCache(t *testing.T) { {"reclaim", func(t *testing.T) { cacheSteps(t, out, reps, fileMB, reclaimProbe) }}, {"reclaimers", func(t *testing.T) { cacheSteps(t, out, reps, fileMB, reclaimersProbe) }}, {"cow", func(t *testing.T) { cacheSteps(t, out, reps, fileMB, cowProbe) }}, + {"pmem", func(t *testing.T) { pmemCompare(t, out, reps) }}, } { if only.MatchString(sub.name) { t.Run(sub.name, sub.run) diff --git a/boot/pmem_exp_test.go b/boot/pmem_exp_test.go new file mode 100644 index 0000000..5f5942f --- /dev/null +++ b/boot/pmem_exp_test.go @@ -0,0 +1,320 @@ +// SPDX-License-Identifier: Apache-2.0 + +package boot_test + +// EXPERIMENT (exp/pmem-base, not for merge): the base image on virtio-pmem with DAX, against the +// qcow2 chain, for the sandbox survey's fifth experiment. Run as TestPageCache's "pmem" subtest, +// on the lab's qemu_run (a QEMU with virtio-pmem-pci) and kernel_run (FS_DAX), variant instead. + +import ( + "bufio" + "encoding/json" + "fmt" + "net" + "os" + "os/exec" + "path/filepath" + "regexp" + "strconv" + "strings" + "syscall" + "testing" + "time" + + "github.com/spin-stack/spin-machine/boot" +) + +// pmemWorkload reads the base image's /usr twice - what a build reads of the toolchain - and +// says at each step what the guest caches. On the qcow2 chain the reads fill the guest's page +// cache, which is the guest's RAM, which is QEMU's anonymous memory; on pmem with DAX they map the +// host's page cache of the one base file, shared by every VM, and the guest caches none of it. +const pmemWorkload = `#!/bin/sh +say() { echo "PMEM $1 $(awk '/^Cached:/{c=$2} /^MemFree:/{m=$2} END{print c, m}' /proc/meminfo) $2" > /dev/ttyS0; sleep 4; } +ms() { awk '{ printf "%d", $1 * 1000 }' /proc/uptime; } +usr() { find /usr -xdev -type f -size -2M 2>/dev/null | head -40000 | xargs cat > /dev/null 2>&1; } +echo "PMEM-NOTE $(findmnt -no SOURCE,FSTYPE,OPTIONS / | tr '\n' ' '); $(findmnt -no SOURCE,OPTIONS /mnt/oldroot 2>/dev/null)" > /dev/ttyS0 +# Whether the pmem region itself is slow, and how the guest maps it: a 256 MB read of the raw +# device, the region in /proc/iomem, and the uncached and write-combining ranges PAT holds. +mountpoint -q /sys/kernel/debug || mount -t debugfs debugfs /sys/kernel/debug +echo "PMEM-NOTE mtrr $(cat /proc/mtrr 2>&1 | tr '\n' ' '); $(dmesg | grep -i -m3 'mtrr' | tr -s ' ' | tr '\n' ' ')" > /dev/ttyS0 +echo "PMEM-NOTE raw $(dd if=/dev/pmem0 of=/dev/null bs=1M count=256 2>&1 | tail -1); iomem $(grep -i -E 'pmem|persistent' /proc/iomem | tr -s ' ' | tr '\n' ' '); pat $(grep -i '0x00000001[0-3]' /sys/kernel/debug/x86/pat_memtype_list 2>/dev/null | head -4 | tr '\n' ' '); pat_enabled $(grep -c . /sys/kernel/debug/x86/pat_memtype_list 2>/dev/null) entries, $(grep -o 'x86/PAT.*' /dev/null; dmesg | grep -i -m2 'PAT' | tr '\n' ' ')" > /dev/ttyS0 +sync; echo 3 > /proc/sys/vm/drop_caches +say 0 0 +t=$(ms); usr; say 1 $(( $(ms) - t )) +t=$(ms); usr; say 2 $(( $(ms) - t )) +echo 3 > /proc/sys/vm/drop_caches; sleep 6; say 3 0 +echo PMEM-DONE > /dev/ttyS0 +` + +var pmemSteps = []string{"idle", "/usr read", "/usr read again", "guest dropped its cache"} + +// overlayInit is PID 1 on the pmem root: the writable layer on vda under an overlayfs whose lower +// layer is the DAX root, pivoted into, and then systemd. +const overlayInit = `#!/bin/sh +fail() { echo "PMEM-FAILED overlay-init: $*" > /dev/ttyS0; exec /bin/sh; } +mount -t ext4 /dev/vda /mnt/upper || fail "mounting vda" +mkdir -p /mnt/upper/u /mnt/upper/w +mount -t overlay overlay -o lowerdir=/,upperdir=/mnt/upper/u,workdir=/mnt/upper/w /mnt/root || fail "overlayfs" +mkdir -p /mnt/root/mnt/oldroot +cd /mnt/root && pivot_root . mnt/oldroot || fail "pivot_root" +exec /sbin/init "$@" +` + +var pmemLine = regexp.MustCompile(`PMEM (\d) (\d+) (\d+) (\d+)`) + +type pmemStep struct { + ms, guestCache, guestFree, rssAnon, rssFile, baseCache float64 +} + +// pmemCompare boots the workload on the qcow2 chain and on the pmem base, reps times each, +// interleaved, and logs each step's p50. +func pmemCompare(t *testing.T, out string, reps int) { + // A run that failed between its mount and its umount left the mount behind (run + // 36807084946): unmounted here, before anything of this run is made. + mustRun(t, "sudo", "sh", "-c", `findmnt -rno TARGET | grep /TestPageCachepmem | while read -r m; do umount "$m"; done; true`) + dir := t.TempDir() + base := filepath.Join(out, "image", "rootfs.qcow2") + qemuImg := filepath.Join(out, "bin", "qemu-img") + + // Both variants boot the same files: an overlay of the base with the workload and the + // overlay-init in it, which the qcow2 variant boots as it is and the pmem variant flattens + // into a raw image - a copy, the base is never written - padded to 2 MiB. + edit := without() + edit.setup = `mkdir -p "$MNT/mnt/upper" "$MNT/mnt/root" && chmod 0755 "$MNT/sbin/overlay-init"` + edit.files["/usr/local/sbin/pmem.sh"] = pmemWorkload + edit.files["/sbin/overlay-init"] = overlayInit + edit.files["/etc/systemd/system/pmem.service"] = strings.ReplaceAll(cacheUnit, "pagecache.sh", "pmem.sh") + edit.links = map[string]string{"/etc/systemd/system/multi-user.target.wants/pmem.service": "../pmem.service"} + lower := filepath.Join(dir, "lower.qcow2") + mustRun(t, qemuImg, "create", "-f", "qcow2", "-F", "qcow2", "-b", base, lower) + editOverlay(t, lower, edit) + raw := filepath.Join(dir, "base.raw") + mustRun(t, qemuImg, "convert", "-O", "raw", lower, raw) + fi, err := os.Stat(raw) + if err != nil { + t.Fatal(err) + } + if err := os.Truncate(raw, (fi.Size()+(2<<20)-1)&^((2<<20)-1)); err != nil { + t.Fatal(err) + } + // The same tree as erofs, uncompressed so DAX can map it: built from the raw image mounted + // read-only, by mkfs.erofs in a container, since the runner's host has no erofs-utils. + mnt := t.TempDir() + mustRun(t, "sudo", "mount", "-o", "loop,ro", raw, mnt) + t.Cleanup(func() { _ = exec.Command("sudo", "umount", mnt).Run() }) // before TempDir's RemoveAll + mustRun(t, "docker", "run", "--rm", "-v", mnt+":/src:ro", "-v", dir+":/out", "alpine:3.22", + "sh", "-c", "apk add -q erofs-utils && mkfs.erofs /out/base.erofs /src >/dev/null && mkfs.erofs -E noinline_data /out/noinline.erofs /src >/dev/null") + erofs := filepath.Join(dir, "base.erofs") + noinline := filepath.Join(dir, "noinline.erofs") + for _, img := range []string{erofs, noinline} { + mustRun(t, "sudo", "sh", "-c", fmt.Sprintf(`s=$(stat -c %%s %[1]s); truncate -s $(( (s + 2097151) / 2097152 * 2097152 )) %[1]s && chmod 0644 %[1]s`, img)) + } + + // Run 36811489671 read /dev/pmem0 at 8.3 MB/s whatever the filesystem: how QEMU maps the + // file is the question now, so ext4 only, in each mapping. + // Run 36816626039: QEMU maps the region as RAM in a KVM memslot, in every mapping, and it + // still reads at 8.3 MB/s. Whether the guest maps it uncached is what nopat says: with PAT + // off the guest cannot ask for anything but write-back. + // Run 36818224412: the guest's PAT held the whole region uncached-minus, and nopat made it + // 1.1 GB/s. Linux turns a write-back request into UC- where the MTRRs do not say write-back, + // and the MTRRs are the firmware's: the default firmware against SeaBIOS, with /proc/mtrr. + variants := []string{"pmem ext4 DAX, share=on,readonly=on", "pmem ext4 DAX seabios, share=on,readonly=on", "pmem ext4 DAX nopat, share=on,readonly=on"} + got := map[string][][]pmemStep{} + var notes = map[string]string{} + for range reps { + for _, v := range variants { + steps, note := pmemBoot(t, out, v, lower, raw, erofs, noinline) + got[v] = append(got[v], steps) + notes[v] = note + } + } + var b strings.Builder + fmt.Fprintf(&b, "\n/usr read twice in a 2048 MB guest, p50 over %d boots; MB but the I/O's ms.\n", reps) + fmt.Fprintf(&b, "host: QEMU's anonymous resident memory (the guest's RAM, private to this VM) and file-backed (pmem's mapping, shared), and the base's pages in the host's cache\n\n") + for _, v := range variants { + fmt.Fprintf(&b, "%s\n%-26s %8s %8s %8s %9s %9s %9s\n", v, "STEP", "I/O ms", "G.CACHE", "G.FREE", "RSS ANON", "RSS FILE", "BASE") + for i, name := range pmemSteps { + var col [6][]float64 + for _, run := range got[v] { + s := run[i] + for j, x := range []float64{s.ms, s.guestCache, s.guestFree, s.rssAnon, s.rssFile, s.baseCache} { + col[j] = append(col[j], x) + } + } + fmt.Fprintf(&b, "%-26s", name) + for j, c := range col { + p, _ := boot.Percentile(c, 0.5) + w := 8 + if j >= 3 { + w = 9 + } + fmt.Fprintf(&b, " %*.0f", w, p) + } + fmt.Fprintln(&b) + } + fmt.Fprintf(&b, " %s\n\n", notes[v]) + } + t.Log(b.String()) +} + +// pmemBoot runs the workload once on variant v and measures each step on the host. +func pmemBoot(t *testing.T, out, v, lower, raw, erofs, noinline string) ([]pmemStep, string) { + t.Helper() + dir := t.TempDir() + qemuImg := filepath.Join(out, "bin", "qemu-img") + sockDir, err := os.MkdirTemp("/tmp", "pm") + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { _ = os.RemoveAll(sockDir) }) // a socket QEMU made + qmpSock := filepath.Join(sockDir, "q.sock") + args := []string{"boot", "--release", out, "--memory", "2048", "--cpus", "2", "--console", "file:/dev/stdout", "--qmp", qmpSock} + cached := raw + fstype := "ext4" + _, opts, _ := strings.Cut(v, ", ") + extra := "" + if strings.Contains(v, "nopat") { + extra = " nopat" + } + if strings.Contains(v, "seabios") { + args = append(args, "--bios", "bios-256k.bin") + } + switch v { + case "pmem + erofs DAX": + cached, fstype = erofs, "erofs" + case "pmem + erofs DAX, no inline data": + cached, fstype = noinline, "erofs" + } + if v == "qcow2 chain" { + overlay := filepath.Join(dir, "overlay.qcow2") + mustRun(t, qemuImg, "create", "-f", "qcow2", "-F", "qcow2", "-b", lower, overlay) + args = append(args, "--disk", overlay, "--append", "init=/sbin/init") + cached = filepath.Join(out, "image", "rootfs.qcow2") + } else { + upper := filepath.Join(dir, "upper.raw") + if err := os.WriteFile(upper, nil, 0o644); err != nil { + t.Fatal(err) + } + if err := os.Truncate(upper, 2<<30); err != nil { + t.Fatal(err) + } + mustRun(t, filepath.Join(out, "bin", "mkfs.ext4"), "-q", "-F", upper) + args = append(args, "--pmem", cached, "--pmem-opts", opts, "--disk", upper, "--disk-format", "raw", "--root", "/dev/pmem0", + "--append", "init=/sbin/overlay-init ro rootfstype="+fstype+" rootflags=dax=always"+extra) + } + cmd := exec.Command(filepath.Join(out, "bin", "spin-machine"), args...) + cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} + stdout, err := cmd.StdoutPipe() + if err != nil { + t.Fatal(err) + } + if err := cmd.Start(); err != nil { + t.Fatalf("launching the machine: %v", err) + } + kill := func() { + if pgid, err := syscall.Getpgid(cmd.Process.Pid); err == nil { + _ = syscall.Kill(-pgid, syscall.SIGKILL) + } + } + timer := time.AfterFunc(time.Duration(envInt(t, "CACHE_TIMEOUT_S", 300))*time.Second, kill) + defer func() { timer.Stop(); kill(); _ = cmd.Wait() }() + + pid := cmd.Process.Pid + steps := make([]pmemStep, len(pmemSteps)) + seen, note := 0, "" + var console strings.Builder + sc := bufio.NewScanner(stdout) + sc.Buffer(make([]byte, 64<<10), 1<<20) + for sc.Scan() { + line := sc.Text() + console.WriteString(line + "\n") + if strings.Contains(line, "PMEM-DONE") { + break + } + if strings.Contains(line, "PMEM-FAILED") { + t.Fatalf("%s: %s\n%s", v, line, tail([]byte(console.String()), 1500)) + } + if _, n, ok := strings.Cut(line, "PMEM-NOTE "); ok { + note += n + "; " + continue + } + m := pmemLine.FindStringSubmatch(line) + if m == nil { + continue + } + i, _ := strconv.Atoi(m[1]) + s := &steps[i] + s.guestCache, s.guestFree, s.ms = kb(m[2]), kb(m[3]), num(m[4]) + // How many times the guest left for QEMU to emulate an access, idle and after the + // first read: a region KVM does not map as memory is one exit per load. + if i <= 1 { + note += fmt.Sprintf("step %d kvm %s; ", i, kvmExits(pid)) + } + if i == 1 { + note += qemuSays(t, qmpSock) + "; " + } + if s.rssAnon, s.rssFile, err = rssSplit(pid); err != nil { + t.Fatal(err) + } + if s.baseCache, _, err = resident(cached); err != nil { + t.Fatal(err) + } + seen++ + } + if seen != len(steps) { + t.Fatalf("%s: the guest reported %d of %d steps; console tail:\n%s", v, seen, len(steps), tail([]byte(console.String()), 1500)) + } + return steps, note +} + +// kvmExits is the VM's exit counters from KVM's debugfs, which only root reads. +func kvmExits(pid int) string { + out, _ := exec.Command("sudo", "sh", "-c", fmt.Sprintf(`cd /sys/kernel/debug/kvm/%d-* 2>/dev/null && for f in exits mmio_exits io_exits halt_exits; do printf "%%s=%%s " $f "$(cat $f 2>/dev/null)"; done`, pid)).Output() + return strings.TrimSpace(string(out)) +} + +// qemuSays is QEMU's own account of the pmem region: the flattened memory tree's lines for it, +// which say ram or i/o, and the memory devices. +func qemuSays(t *testing.T, socket string) string { + t.Helper() + conn, err := net.Dial("unix", socket) + if err != nil { + return "no QMP: " + err.Error() + } + defer func() { _ = conn.Close() }() // a diagnostic connection + _ = conn.SetDeadline(time.Now().Add(30 * time.Second)) + q := &qmp{conn: conn, enc: json.NewEncoder(conn), dec: json.NewDecoder(conn)} + var greeting struct{ QMP *struct{} } + if err := q.dec.Decode(&greeting); err != nil { + return "no QMP greeting: " + err.Error() + } + q.do(t, "qmp_capabilities", nil) + var says []string + for _, cmd := range []string{"info mtree -f", "info memory-devices"} { + var text string + _ = json.Unmarshal(q.do(t, "human-monitor-command", map[string]any{"command-line": cmd}), &text) + for l := range strings.Lines(text) { + if strings.Contains(strings.ToLower(l), "pmem") { + says = append(says, strings.Join(strings.Fields(l), " ")) + } + } + } + return "qemu: " + strings.Join(says, " | ") +} + +// rssSplit is the process pid's anonymous and file-backed resident memory in MB. +func rssSplit(pid int) (anon, file float64, err error) { + raw, err := os.ReadFile(fmt.Sprintf("/proc/%d/status", pid)) + if err != nil { + return 0, 0, err + } + for l := range strings.Lines(string(raw)) { + if v, ok := strings.CutPrefix(l, "RssAnon:"); ok { + anon = kb(strings.Fields(v)[0]) + } + if v, ok := strings.CutPrefix(l, "RssFile:"); ok { + file = kb(strings.Fields(v)[0]) + } + } + return anon, file, nil +} diff --git a/cmd/spin-machine/main.go b/cmd/spin-machine/main.go index 6bdb5fc..4e8196b 100644 --- a/cmd/spin-machine/main.go +++ b/cmd/spin-machine/main.go @@ -201,6 +201,8 @@ type machineFlags struct { root string profile bool extra string + pmem string + pmemOpt string } func (o *machineFlags) register(fs *flag.FlagSet) { @@ -238,6 +240,8 @@ func (o *machineFlags) register(fs *flag.FlagSet) { fs.StringVar(&o.root, "root", "/dev/vda", "block device the kernel mounts as root; empty to stay on the initrd") fs.BoolVar(&o.profile, "profile", false, "boot with initcall profiling: silent console, full ring buffer") fs.StringVar(&o.extra, "append", "", "extra kernel command line arguments") + fs.StringVar(&o.pmem, "pmem", "", "EXPERIMENT: a raw filesystem image mapped read-only as /dev/pmem0 (virtio-pmem)") + fs.StringVar(&o.pmemOpt, "pmem-opts", "", "EXPERIMENT: the memory-backend-file options for --pmem (default share=on,readonly=on)") } // scratch is whether the disk is the base image under an overlay QEMU makes and @@ -313,6 +317,8 @@ func (o *machineFlags) spec() (machine.Spec, error) { DirectOverBacking: o.directOverBacking, }} } + s.Pmem = o.pmem + s.PmemOpts = o.pmemOpt c := machine.DefaultCmdline() c.Init = o.init diff --git a/machine/machine.go b/machine/machine.go index 17b1566..a23c66f 100644 --- a/machine/machine.go +++ b/machine/machine.go @@ -25,6 +25,7 @@ package machine import ( "bufio" + "cmp" "crypto/sha256" "encoding/hex" "encoding/json" @@ -73,6 +74,9 @@ const ( // every slot below this one is where it was, so this did not renumber a // single existing device. Fifteen NICs was not a number anything needed. slotMem = 0x1e + + // EXPERIMENT (exp/pmem-base, not for merge): Spec.Pmem's device. + slotPmem = 0x1b ) // maxDisks and maxNICs bound the fixed slot ranges above. Exceeding either is a @@ -554,6 +558,14 @@ type Spec struct { // may be given have one fingerprint, and raising the count strands no checkpoint. HotplugDisks int + // Pmem, EXPERIMENT (exp/pmem-base, not for merge): a raw filesystem image mapped into the + // guest read-only as /dev/pmem0 through virtio-pmem, shared with every other VM that maps + // it. Its size is a multiple of 2 MiB. + Pmem string + // PmemOpts, EXPERIMENT: how the memory-backend-file maps Pmem; share=on,readonly=on when + // empty. + PmemOpts string + // VsockCID, when non-zero, gives the machine a vhost-vsock device with that // context id. It is how anything inside the guest is reached: this machine // has no serial port for a caller to drive and no network it is required to @@ -703,7 +715,7 @@ func (s Spec) Shape() Shape { Accel: accel, CPU: cpu, SMP: smpArg(s.BootCPUs, s.MaxCPUs), - Memory: memoryArg(s.Memory.SizeMB, s.Memory.MaxMB), + Memory: memoryArg(s.Memory.SizeMB, s.maxMB()), } } @@ -794,6 +806,28 @@ func smpArg(bootCPUs, maxCPUs int) string { return fmt.Sprintf("%d", bootCPUs) } +// pmemMB is Spec.Pmem's size in MiB; 0 without one, or when it cannot be read, which QEMU then +// fails on. +func (s Spec) pmemMB() int { + if s.Pmem == "" { + return 0 + } + fi, err := os.Stat(s.Pmem) + if err != nil { + return 0 + } + return int(fi.Size() >> 20) +} + +// maxMB is -m's maxmem: the growth ceiling, and above it room for Spec.Pmem, which is a memory +// device too. +func (s Spec) maxMB() int { + if p := s.pmemMB(); p > 0 { + return max(s.Memory.MaxMB, s.Memory.SizeMB) + p + } + return s.Memory.MaxMB +} + // memoryArg formats -m. // // No slots=, and that is the change: slots are ACPI DIMM sockets, and this @@ -1115,6 +1149,12 @@ func (s Spec) appendDevices(args []string) []string { VirtioMemID, memGrowthID, virtioModern, slotMem)) } + if s.Pmem != "" { + args = append(args, + "-object", fmt.Sprintf("memory-backend-file,id=pmem0-mem,mem-path=%s,size=%dM,%s", qemuOpt(s.Pmem), s.pmemMB(), cmp.Or(s.PmemOpts, "share=on,readonly=on")), + "-device", fmt.Sprintf("virtio-pmem-pci,id=pmem0,memdev=pmem0-mem,%s,addr=0x%x", virtioModern, slotPmem)) + } + // The controller disks this machine is given while it runs are added to. See // Spec.HotplugDisks: the root complex takes no device_add, and a SCSI bus does. if s.HotplugDisks > 0 { diff --git a/qemu/Dockerfile b/qemu/Dockerfile index 9237b08..530c56c 100644 --- a/qemu/Dockerfile +++ b/qemu/Dockerfile @@ -459,7 +459,7 @@ RUN git clone --quiet --branch "${QBOOT_VERSION}" https://github.com/bonzini/qbo git -C upstream archive "${QBOOT_COMMIT}" > upstream.tar && \ echo "${QBOOT_ARCHIVE_SHA256} upstream.tar" | sha256sum -c - -COPY qemu/qboot/write-pointer.patch qemu/qboot/pam.patch /build/ +COPY qemu/qboot/write-pointer.patch qemu/qboot/pam.patch qemu/qboot/mtrr.patch /build/ # Upstream's meson.build flags, with -Os fixed rather than taken from a build type. RUN set -eux; \ @@ -467,6 +467,7 @@ RUN set -eux; \ tar -xf upstream.tar -C src; \ patch -p1 --fuzz=0 -d src < write-pointer.patch; \ patch -p1 --fuzz=0 -d src < pam.patch; \ + patch -p1 --fuzz=0 -d src < mtrr.patch; \ cd src; \ for f in *.c *.S; do \ gcc -Os -m32 -march=i386 -mregparm=3 -fno-stack-protector \ @@ -482,12 +483,13 @@ RUN set -eux; \ # fail here - it fails as a guest that prints nothing - so the size is checked; and the patches, # because a patch that applied to nothing still leaves a tree that compiles: without the first # the firmware hangs every machine that has vmgenid, without the second it is slower and says -# nothing. +# nothing, and without the third every device memory a guest maps reads uncached. RUN set -eux; \ size=$(stat -c%s /build/qboot.bin); \ test "$size" -eq 65536 || { echo "qboot.bin is $size bytes, not 65536" >&2; exit 1; }; \ grep -q 'CMD_WRITE_PTR' src/tables.c && grep -q 'fw_cfg_write_file' src/fw_cfg.c || { echo "the WRITE_POINTER patch is not in the tree that was compiled" >&2; exit 1; }; \ grep -q '0x33333330' src/hwsetup.c || { echo "the PAM patch is not in the tree that was compiled" >&2; exit 1; }; \ + grep -q 'setup_mtrr();' src/main.c || { echo "the MTRR patch is not in the tree that was compiled" >&2; exit 1; }; \ sha256sum /build/qboot.bin FROM ${ALPINE_IMAGE} AS runtime diff --git a/qemu/devices.mak b/qemu/devices.mak index f164355..4986e29 100644 --- a/qemu/devices.mak +++ b/qemu/devices.mak @@ -88,6 +88,9 @@ CONFIG_VHOST_VSOCK=y # virtio-mem-pci: how a machine grows past its boot memory. ACPI DIMM slots are not used, # which is why the machine string carries no slots=. CONFIG_VIRTIO_MEM=y +# EXPERIMENT (exp/pmem-base, not for merge): virtio-pmem-pci, a base image mapped into the +# guest with DAX instead of read through virtio-blk and the guest's page cache. +CONFIG_VIRTIO_PMEM=y # vmgenid: how a restored guest learns it was restored, so its random pool is reseeded. CONFIG_ACPI_VMGENID=y # virtio-scsi-pci and scsi-hd: disks that arrive while the machine runs (Spec.HotplugDisks). diff --git a/qemu/qboot/README.md b/qemu/qboot/README.md index 35c0ec2..cf0dbe6 100644 --- a/qemu/qboot/README.md +++ b/qemu/qboot/README.md @@ -23,12 +23,23 @@ byte. The values are immediates: once PAM0 is ram, 0xf0000-0x100000 reads zeroes `setup_hw` has shadowed the BIOS, so a table in `.rodata` would be read back as zeroes. i440fx keeps the loop; this machine is q35 and nothing here boots the other. +## MTRRs + +Stock qboot never touches the MTRRs, so the guest boots with them disabled ("MTRRs disabled by +BIOS"). Linux then maps every range it is asked to map write-back but does not know as RAM +uncached-minus: a virtio-pmem region read at 8.3 MB/s, against 1.1 GB/s under SeaBIOS, which +programs them (lab runs 36818224412 and 36818721809, 2026-10-01). `mtrr.patch` does what SeaBIOS +does, on the boot CPU only - Linux gives its own MTRR state to the CPUs it starts: enabled, +write-back by default, and the 32-bit PCI hole, from the top of low RAM to 4 GiB, uncacheable in +naturally aligned power-of-two ranges. Fixed-range MTRRs stay off. If the hole does not fit the +variable ranges the CPU has, they are left disabled rather than caching MMIO. + ## Build `task qemu:build` builds it, in the `qboot` stage of `qemu/Dockerfile`, and it lands beside the SeaBIOS blobs as `_output/qemu/qboot.bin`, where a release, `task qemu:fetch` and CI all find it. The stage clones the pinned commit, verifies the SHA-256 of the tar `git archive` -writes for it, applies `write-pointer.patch` and then `pam.patch` with no fuzz, and compiles with upstream's +writes for it, applies `write-pointer.patch`, `pam.patch` and `mtrr.patch` with no fuzz, and compiles with upstream's meson.build flags plus `-Os`. It asserts on what came out: 65536 bytes, the ROM window it is linked for, and the patch's code in the tree that was compiled. diff --git a/qemu/qboot/mtrr.patch b/qemu/qboot/mtrr.patch new file mode 100644 index 0000000..8a2d6c2 --- /dev/null +++ b/qemu/qboot/mtrr.patch @@ -0,0 +1,95 @@ +diff --git a/main.c b/main.c +index afa2200..1a156f6 100644 +--- a/main.c ++++ b/main.c +@@ -77,6 +77,82 @@ static void extract_e820(void) + e820_seg = ((uintptr_t) e820) >> 4; + } + ++/* ++ * MTRRs: enabled, write-back by default, and the 32-bit PCI hole - from the top of low ++ * RAM to 4 GiB - uncacheable, as SeaBIOS sets them. Left disabled, Linux maps every ++ * range it memremaps write-back but does not know as RAM - a virtio-pmem region, for ++ * one - uncached-minus, and reads it at 8 MB/s. Only this CPU: Linux gives its own ++ * state to the APs it starts. Fixed-range MTRRs stay off, so below 1 MiB is the ++ * default type too. ++ */ ++#define MSR_MTRRcap 0xfe ++#define MSR_MTRRdefType 0x2ff ++#define MTRR_PHYSBASE(n) (0x200 + 2 * (n)) ++#define MTRR_PHYSMASK(n) (0x201 + 2 * (n)) ++#define MTRR_TYPE_UC 0 ++#define MTRR_TYPE_WB 6 ++#define MTRR_DEF_ENABLE 0x800 ++#define MTRR_MASK_VALID 0x800 ++ ++static inline void cpuid(uint32_t leaf, uint32_t *a, uint32_t *b, uint32_t *c, uint32_t *d) ++{ ++ asm volatile("cpuid" : "=a"(*a), "=b"(*b), "=c"(*c), "=d"(*d) : "0"(leaf), "2"(0)); ++} ++ ++static inline uint64_t rdmsr(uint32_t msr) ++{ ++ uint32_t lo, hi; ++ asm volatile("rdmsr" : "=a"(lo), "=d"(hi) : "c"(msr)); ++ return ((uint64_t)hi << 32) | lo; ++} ++ ++static inline void wrmsr(uint32_t msr, uint64_t val) ++{ ++ asm volatile("wrmsr" : : "c"(msr), "a"((uint32_t)val), "d"((uint32_t)(val >> 32))); ++} ++ ++static void setup_mtrr(void) ++{ ++ uint32_t a, b, c, d; ++ uint64_t phys_mask, base = lowmem, top = 1ull << 32; ++ int vcnt, n = 0; ++ ++ cpuid(1, &a, &b, &c, &d); ++ if (!(d & (1 << 12)) || !(d & (1 << 5))) /* MTRR, MSR */ ++ return; ++ vcnt = rdmsr(MSR_MTRRcap) & 0xff; ++ cpuid(0x80000000, &a, &b, &c, &d); ++ phys_mask = (1ull << 36) - 1; ++ if (a >= 0x80000008) { ++ cpuid(0x80000008, &a, &b, &c, &d); ++ phys_mask = (1ull << (a & 0xff)) - 1; ++ } ++ ++ wrmsr(MSR_MTRRdefType, 0); ++ for (n = 0; n < vcnt; n++) { ++ wrmsr(MTRR_PHYSBASE(n), 0); ++ wrmsr(MTRR_PHYSMASK(n), 0); ++ } ++ /* [lowmem, 4G) in naturally aligned powers of two. */ ++ for (n = 0; base < top && n < vcnt; n++) { ++ uint64_t size = top - base; ++ while (size & (size - 1)) ++ size &= size - 1; ++ while (base & (size - 1)) ++ size >>= 1; ++ wrmsr(MTRR_PHYSBASE(n), base | MTRR_TYPE_UC); ++ wrmsr(MTRR_PHYSMASK(n), (~(size - 1) & phys_mask) | MTRR_MASK_VALID); ++ base += size; ++ } ++ /* Not all of the hole fits: leave them off rather than cache MMIO. */ ++ if (base < top) { ++ for (n = 0; n < vcnt; n++) ++ wrmsr(MTRR_PHYSMASK(n), 0); ++ return; ++ } ++ wrmsr(MSR_MTRRdefType, MTRR_DEF_ENABLE | MTRR_TYPE_WB); ++} ++ + int __attribute__ ((section (".text.startup"))) main(void) + { + bool have_pci; +@@ -98,6 +174,7 @@ int __attribute__ ((section (".text.startup"))) main(void) + fw_cfg_setup(); + extract_acpi(); + extract_e820(); ++ setup_mtrr(); + setup_mptable(); + extract_smbios(); + boot_from_fwcfg(); diff --git a/versions.yaml b/versions.yaml index a099df4..cf9c363 100644 --- a/versions.yaml +++ b/versions.yaml @@ -152,7 +152,7 @@ the machine's BIOS, built in qemu/Dockerfile's qboot stage. archive is the sha256 of what `git archive` writes for the commit with debian's git, which nothing but a build can compute, so a bump is by hand: the commit, a build, the archive's sum from its failure, - write-pointer.patch and pam.patch rebased onto it, NOTICE and qemu/qboot/README.md restated, and + write-pointer.patch, pam.patch and mtrr.patch rebased onto it, NOTICE and qemu/qboot/README.md restated, and `task boot:firmware` again. - name: actionlint