Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions boot/Taskfile.yml
Original file line number Diff line number Diff line change
Expand Up @@ -69,6 +69,15 @@ tasks:
cmds:
- SPIN_LOGIND_TEST=1 go test ./boot/ -run '^TestLogindSessions$' -count=1 -v -timeout 5m

lockdown:
desc: >-
Check the guest kernel runs lockdown at confidentiality and the BPF LSM, and has no
/dev/mem or /proc/kcore, in a disposable guest. Needs a built release; uses KVM when
available, otherwise TCG.
deps: [':tools']
cmds:
- SPIN_LOCKDOWN_TEST=1 go test ./boot/ -run '^TestTheGuestKernelIsLockedDown$' -count=1 -v -timeout 5m

initcalls:
desc: >-
Where the kernel's own boot goes, initcall by initcall, as a p50 over REPS boots.
Expand Down
102 changes: 102 additions & 0 deletions boot/console_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,102 @@
// SPDX-License-Identifier: Apache-2.0

package boot_test

import (
"bytes"
"context"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"time"

"github.com/spin-stack/spin-machine/machine"
)

// checkOnConsole boots a disposable 1 GiB guest of rel on root, whose unit prints ok or failed on
// the serial console, and fails the test unless ok comes first. It returns what the console said.
// Under TCG where there is no /dev/kvm: what it checks is behaviour, not time.
func checkOnConsole(t *testing.T, out string, rel *machine.Release, root *rawRoot, ok, failed string) string {
t.Helper()
spec := rel.Spec()
spec.BootCPUs = 2
spec.Memory.SizeMB = 1024
spec.Disks = []machine.Disk{{Path: root.path, Format: "raw"}}
spec.Serial = "stdio"
c := machine.DefaultCmdline()
c.Root = "/dev/vda"
c.Init = "/sbin/init"
spec.Cmdline = c
args, err := spec.Args()
if err != nil {
t.Fatal(err)
}
if _, err := os.Stat("/dev/kvm"); err != nil {
spec.QEMU = filepath.Join(out, "bin/qemu-system-x86_64-tcg")
for i := range args {
if args[i] == "-accel" {
args[i+1] = "tcg"
}
if strings.HasPrefix(args[i], "host,migratable=on") {
args[i] = "max" + strings.TrimPrefix(args[i], "host")
}
}
t.Log("TCG: checking functionality only")
}
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
cmd := exec.CommandContext(ctx, spec.QEMU, args...)
var stderr bytes.Buffer
cmd.Stderr = &stderr
stdout, err := cmd.StdoutPipe()
if err != nil {
t.Fatal(err)
}
if err := cmd.Start(); err != nil {
t.Fatal(err)
}
defer func() {
cancel()
_ = cmd.Wait() // killed by the cancel above; the console is the result
}()
var console bytes.Buffer
buf := make([]byte, 4096)
for {
n, err := stdout.Read(buf)
console.Write(buf[:n])
if strings.Contains(console.String(), ok) {
return console.String()
}
if err != nil || strings.Contains(console.String(), failed) {
t.Fatalf("guest check failed: %v\n%s\n%s", err, &console, &stderr)
}
}
}

// oneshotAtBoot has root run script as a oneshot unit three seconds after boot, its output on
// the console.
func oneshotAtBoot(t *testing.T, root *rawRoot, name, script, env string) {
t.Helper()
content, err := os.ReadFile(filepath.Join("testdata", script))
if err != nil {
t.Fatal(err)
}
root.write("/"+script, string(content))
root.write("/etc/systemd/system/"+name+".service", `[Unit]
Description=A check in a disposable guest
After=multi-user.target
[Service]
Type=oneshot
`+env+`
ExecStart=/bin/sh /`+script+`
StandardOutput=journal+console
StandardError=journal+console
`)
root.write("/etc/systemd/system/"+name+".timer", `[Timer]
OnBootSec=3s
AccuracySec=100ms
`)
root.link("/etc/systemd/system/timers.target.wants/"+name+".timer", "/etc/systemd/system/"+name+".timer")
}
32 changes: 32 additions & 0 deletions boot/lockdown_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
// SPDX-License-Identifier: Apache-2.0

package boot_test

import (
"os"
"testing"

"github.com/spin-stack/spin-machine/machine"
)

// The guest's root is not the kernel's: the kernel runs lockdown at confidentiality and the BPF
// LSM, and has no /dev/mem and no /proc/kcore. A consumer that enforces policy in the guest
// with BPF LSM programs relies on root being unable to read or rewrite the kernel they live in;
// kernel/Dockerfile fails a build whose config lost these, and this asks the kernel that boots.
func TestTheGuestKernelIsLockedDown(t *testing.T) {
if os.Getenv("SPIN_LOCKDOWN_TEST") != "1" {
t.Skip("set SPIN_LOCKDOWN_TEST=1 to check the guest kernel's lockdown in a built image")
}
out := releaseTree(t)
rel, err := machine.OpenRelease(out)
if err != nil {
t.Fatal(err)
}
base, err := rel.Rootfs()
if err != nil {
t.Fatal(err)
}
root := newRawRoot(t, out, base)
oneshotAtBoot(t, root, "lockdown-check", "lockdown-check.sh", "")
t.Log(checkOnConsole(t, out, rel, root, "LOCKDOWN_CHECK_OK", "LOCKDOWN_CHECK_FAILED"))
}
89 changes: 4 additions & 85 deletions boot/logind_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -3,14 +3,8 @@
package boot_test

import (
"bytes"
"context"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"time"

"github.com/spin-stack/spin-machine/machine"
)
Expand Down Expand Up @@ -45,86 +39,11 @@ func TestLogindSessions(t *testing.T) {
if os.Getenv("SPIN_LOGIND_NO_SEATS") == "1" {
root.remove("/etc/systemd/system/systemd-logind-varlink.socket.d/10-seats.conf")
}
for _, name := range []string{"logind-check.sh", "logind-session.sh"} {
content, err := os.ReadFile(filepath.Join("testdata", name))
if err != nil {
t.Fatal(err)
}
root.write("/"+name, string(content))
}
root.write("/etc/systemd/system/logind-check.service", `[Unit]
Description=Check on-demand sessions in a disposable guest
After=multi-user.target
[Service]
Type=oneshot
Environment=LOGIND_EXPECT=`+expectedState+`
ExecStart=/bin/sh /logind-check.sh
StandardOutput=journal+console
StandardError=journal+console
`)
root.write("/etc/systemd/system/logind-check.timer", `[Timer]
OnBootSec=3s
AccuracySec=100ms
`)
root.link("/etc/systemd/system/timers.target.wants/logind-check.timer", "/etc/systemd/system/logind-check.timer")
spec := rel.Spec()
spec.BootCPUs = 2
spec.Memory.SizeMB = 1024
spec.Disks = []machine.Disk{{Path: root.path, Format: "raw"}}
spec.Serial = "stdio"
c := machine.DefaultCmdline()
c.Root = "/dev/vda"
c.Init = "/sbin/init"
spec.Cmdline = c
args, err := spec.Args()
if err != nil {
t.Fatal(err)
}
if _, err := os.Stat("/dev/kvm"); err != nil {
spec.QEMU = filepath.Join(out, "bin/qemu-system-x86_64-tcg")
for i := range args {
if args[i] == "-accel" {
args[i+1] = "tcg"
}
if strings.HasPrefix(args[i], "host,migratable=on") {
args[i] = "max" + strings.TrimPrefix(args[i], "host")
}
}
t.Log("TCG: checking functionality only")
}
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
cmd := exec.CommandContext(ctx, spec.QEMU, args...)
var stderr bytes.Buffer
cmd.Stderr = &stderr
stdout, err := cmd.StdoutPipe()
content, err := os.ReadFile("testdata/logind-session.sh")
if err != nil {
t.Fatal(err)
}
if err := cmd.Start(); err != nil {
t.Fatal(err)
}
waited := false
stop := func() {
cancel()
if !waited {
_ = cmd.Wait()
waited = true
}
}
defer stop()
var console bytes.Buffer
buf := make([]byte, 4096)
for {
n, err := stdout.Read(buf)
console.Write(buf[:n])
if strings.Contains(console.String(), "LOGIND_CHECK_OK") {
t.Log(console.String())
return
}
if err != nil || strings.Contains(console.String(), "LOGIND_CHECK_FAILED") {
stop()
t.Fatalf("guest session check failed: %v\n%s\n%s", err, &console, &stderr)
}
}
root.write("/logind-session.sh", string(content))
oneshotAtBoot(t, root, "logind-check", "logind-check.sh", "Environment=LOGIND_EXPECT="+expectedState)
t.Log(checkOnConsole(t, out, rel, root, "LOGIND_CHECK_OK", "LOGIND_CHECK_FAILED"))
}
24 changes: 24 additions & 0 deletions boot/testdata/lockdown-check.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
#!/bin/sh
# SPDX-License-Identifier: Apache-2.0
set -eu
trap 'echo LOCKDOWN_CHECK_FAILED' EXIT

# The LSMs the kernel runs, in its own words, and the lockdown level it enforces: the brackets
# mark the one in force.
echo "lsm: $(cat /sys/kernel/security/lsm)"
echo "lockdown: $(cat /sys/kernel/security/lockdown)"
# capability is always there, whatever CONFIG_LSM says: the kernel puts it first among the rest.
test "$(cat /sys/kernel/security/lsm)" = "lockdown,capability,bpf"
grep -q '\[confidentiality\]' /sys/kernel/security/lockdown

# Root cannot read the kernel's memory, nor lower the level.
test ! -e /dev/mem
test ! -e /proc/kcore
if echo none > /sys/kernel/security/lockdown 2>/dev/null; then
echo "root lowered lockdown"
exit 1
fi
grep -q '\[confidentiality\]' /sys/kernel/security/lockdown

trap - EXIT
echo LOCKDOWN_CHECK_OK
15 changes: 15 additions & 0 deletions kernel/Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -230,6 +230,21 @@ RUN <<EOT

grep -q "CONFIG_EXT4_FS=y" .config || (echo "ERROR: CONFIG_EXT4_FS not enabled (the base image cannot be mounted)!" ; exit 1)

# The guest defends spin's security from its own root (spin's F2b): the supervisor loads BPF
# LSM programs that refuse an agent what its mark does not cover and refuse anybody
# detaching them, and lockdown keeps root from reading or writing the kernel those
# programs and the supervisor's keys live in. Without any one of these a root that wants
# to, gets around them in silence. DEVMEM and PROC_KCORE are the two reads of the kernel's
# memory, gone rather than left to lockdown alone.
for opt in SECURITY SECURITYFS SECURITY_NETWORK SECURITY_PATH BPF_LSM SECURITY_LOCKDOWN_LSM \
SECURITY_LOCKDOWN_LSM_EARLY LOCK_DOWN_KERNEL_FORCE_CONFIDENTIALITY; do
grep -q "^CONFIG_${opt}=y" .config || { echo "ERROR: CONFIG_${opt} is not enabled after olddefconfig!"; exit 1; }
done
grep -q '^CONFIG_LSM="lockdown,bpf"' .config || { echo "ERROR: the LSMs are not lockdown and bpf!"; grep '^CONFIG_LSM=' .config; exit 1; }
for opt in DEVMEM PROC_KCORE MODULES KEXEC; do
! grep -q "^CONFIG_${opt}=y" .config || { echo "ERROR: CONFIG_${opt} lets root reach the kernel's memory!"; exit 1; }
done

# No virtual consoles. This machine's QEMU ships no VGA at all and its console is one
# 16550 on ttyS0, so CONFIG_VT builds 64 tty devices that nothing opens and that udev
# walks on every boot — a quarter of the 266 devices it coldplugs. olddefconfig drops
Expand Down
29 changes: 26 additions & 3 deletions kernel/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -38,9 +38,32 @@ and 5.0 MB of image — 4.55 MB of which is the `.BTF` section, which `strip -s`
it is allocated and the guest reads it back out of its own image. `ftrace: allocating 41727
entries` is the new work that shows up in the log.

`BPF_LSM` is deliberately not enabled: it needs `CONFIG_SECURITY`, and what it buys is
enforcing access policy inside a VM that holds one workload — a boundary drawn inside the
boundary this machine already is.
## Lockdown and the BPF LSM

The guest's root is not the kernel's. The kernel runs two LSMs, `lockdown` and `bpf`
(`CONFIG_LSM="lockdown,bpf"`):

- **Lockdown** is forced to *confidentiality* from the first instruction, and no write to
`/sys/kernel/security/lockdown` lowers it. Root can then neither rewrite the running kernel
(kexec, `/dev/port`, MSRs, ACPI table overrides) nor read it (kprobes, `bpf_probe_read_kernel`,
perf on the kernel).
- **`/dev/mem` and `/proc/kcore`** are not built at all, rather than left to lockdown.
- **The BPF LSM** is there so a consumer can enforce policy in the guest with programs it loads
before handing the machine to its tenant. Lockdown is what makes such policy hold: without
it, root reads or rewrites the kernel those programs live in.

What enforcing the policy is for, and how, is the consumer's. That this kernel does both is
what `kernel/Dockerfile` checks after olddefconfig, and `task boot:lockdown` asks a booted guest.

Lockdown at confidentiality also takes away the tenant's kernel tracing: kprobes,
`bpf_probe_read_kernel` and perf on the kernel. User-space tracing and networking eBPF are
untouched. systemd notices at boot: it logs `use of bpf to read kernel RAM is restricted` and
carries on, and logins are unaffected (`task boot:logind`).

Measured 2026-10-01, `task boot:initcalls` with `SPIN_KERNEL_B` set to the kernel without these
options: no cost that can be told from noise. Over 20 interleaved boots each, kernel start to
`Freeing unused kernel image` had a p50 of 196.3 ms with them and 208.3 ms without, on a host at
load 6-10, where the p95 was 342 ms.

## Changing it

Expand Down
16 changes: 11 additions & 5 deletions kernel/config-7.3-rc5-x86_64
Original file line number Diff line number Diff line change
Expand Up @@ -2316,9 +2316,8 @@ CONFIG_HW_RANDOM=y
CONFIG_HW_RANDOM_VIA=y
CONFIG_HW_RANDOM_VIRTIO=y
# CONFIG_HW_RANDOM_XIPHERA is not set
CONFIG_DEVMEM=y
# CONFIG_DEVMEM is not set
CONFIG_NVRAM=y
CONFIG_DEVPORT=y
CONFIG_HPET=y
# CONFIG_HPET_MMAP is not set
# CONFIG_HANGCHECK_TIMER is not set
Expand Down Expand Up @@ -3229,7 +3228,7 @@ CONFIG_FAT_DEFAULT_IOCHARSET="iso8859-1"
# Pseudo filesystems
#
CONFIG_PROC_FS=y
CONFIG_PROC_KCORE=y
# CONFIG_PROC_KCORE is not set
CONFIG_PROC_SYSCTL=y
CONFIG_PROC_PAGE_MONITOR=y
# CONFIG_PROC_CHILDREN is not set
Expand Down Expand Up @@ -3367,11 +3366,18 @@ CONFIG_PROC_MEM_ALWAYS_FORCE=y
# CONFIG_PROC_MEM_FORCE_PTRACE is not set
# CONFIG_PROC_MEM_NO_FORCE is not set
# CONFIG_MSEAL_SYSTEM_MAPPINGS is not set
# CONFIG_SECURITY is not set
# CONFIG_SECURITYFS is not set
CONFIG_SECURITY=y
CONFIG_SECURITYFS=y
CONFIG_SECURITY_NETWORK=y
CONFIG_SECURITY_PATH=y
CONFIG_SECURITY_LOCKDOWN_LSM=y
CONFIG_SECURITY_LOCKDOWN_LSM_EARLY=y
CONFIG_LOCK_DOWN_KERNEL_FORCE_CONFIDENTIALITY=y
CONFIG_BPF_LSM=y
# CONFIG_INTEL_TXT is not set
# CONFIG_STATIC_USERMODEHELPER is not set
CONFIG_DEFAULT_SECURITY_DAC=y
CONFIG_LSM="lockdown,bpf"

#
# Kernel hardening options
Expand Down