diff --git a/cmd/host-agent-real/main.go b/cmd/host-agent-real/main.go index e639bb76..30e7598d 100644 --- a/cmd/host-agent-real/main.go +++ b/cmd/host-agent-real/main.go @@ -41,6 +41,7 @@ import ( "github.com/onmoose/os/internal/hostagent/brainlaunch" "github.com/onmoose/os/internal/hostagent/controlplane" "github.com/onmoose/os/internal/hostagent/cpupdate" + "github.com/onmoose/os/internal/hostagent/sshaccess" "github.com/onmoose/os/internal/profile" "github.com/onmoose/os/internal/protocol" "github.com/onmoose/os/internal/version" @@ -80,6 +81,25 @@ func main() { a, cleanup := buildAgent() defer cleanup() + // After a Debian major the /etc tidy-up keeps the sshd drop-in but drops the + // unit's enable links (BUILD.md # 1b, rule 4), and leaves a marker. Only + // then is sshd turned back on for the accounts the drop-in names. On any + // other start host-agent leaves sshd's run state alone. + if _, err := os.Stat(sshaccess.MajorTidiedMarker); err == nil { + if sm, ok := a.SSH.(*sshaccess.Manager); ok { + if on, err := sm.EnsureOnAtStart(); err != nil { + slog.Warn("could not turn sshd on after a Debian-major tidy-up; trying again at the next start", "err", err) + } else { + if on { + slog.Info("sshd turned on after a Debian-major tidy-up: accounts have SSH on") + } + if err := os.Remove(sshaccess.MajorTidiedMarker); err != nil { + slog.Warn("could not remove the tidy-up marker", "err", err) + } + } + } + } + // The brain's launch config is built once and used twice: to launch the // brain at boot, and as the base of every control-plane update. Reusing it // is the point — an updated brain has to be identical to a first-boot one diff --git a/dev/cloud/build-bundle.sh b/dev/cloud/build-bundle.sh index e90de688..98637da4 100755 --- a/dev/cloud/build-bundle.sh +++ b/dev/cloud/build-bundle.sh @@ -26,7 +26,9 @@ # never trust the throwaway signer; # 4. a wrong key: a bundle-shaped check against an unrelated CA is refused; # 5. a rehearsal of dev/release/sign-bundle.sh with a throwaway "release" CA, -# so the sign job's script runs on every build, not only on a release. +# so the sign job's script runs on every build, not only on a release; +# 6. the image's account and group ids, read from the sysusers.d file the +# build generates, match dev/os-lock/cloud-accounts.lock. # # Writes to OUTDIR: moose-cloud.raucb, image-etc/{system.conf,keyring.pem} and # bundle-info.txt, and a size table to $GITHUB_STEP_SUMMARY when set. @@ -96,6 +98,7 @@ rauc_run "$out" ' # The image'"'"'s own RAUC config and keyring, read back out of the slot. unsquashfs -cat bundle/rootfs.img etc/rauc/system.conf > image-etc/system.conf unsquashfs -cat bundle/rootfs.img etc/rauc/keyring.pem > image-etc/keyring.pem + unsquashfs -cat bundle/rootfs.img usr/lib/sysusers.d/moose-image-accounts.conf > image-etc/accounts.conf compatible="$(sed -n "s/^compatible=//p" image-etc/system.conf | head -n1)" [ -n "$compatible" ] || { echo "the slot'"'"'s /etc/rauc/system.conf names no compatible" >&2; exit 1; } grep -qx "check-purpose=codesign" image-etc/system.conf || { echo "the slot'"'"'s system.conf does not ask for check-purpose=codesign" >&2; exit 1; } @@ -145,6 +148,23 @@ echo "check 1: the slot carries the keyring this build staged (${mode})" cmp -s "$out/image-etc/keyring.pem" "$staged" || { echo "the slot's /etc/rauc/keyring.pem is not the ${mode} keyring this build staged" >&2; exit 1; } echo "ok: $(openssl x509 -in "$out/image-etc/keyring.pem" -noout -subject)" +echo "check 6: the image's account ids match dev/os-lock/cloud-accounts.lock" +# A box keeps an account's uid and gid for life: its /etc/passwd is in the +# /etc upper layer, and its files on the state partition are owned by number. +# So an image whose packages gave an account another id would leave a box's +# files owned by the wrong account (BUILD.md # 1b, rule 3). The lock is edited +# by hand: a new account is added, a changed id is a bug to fix in the build. +awk '$1 == "g" { print "g", $2, $3 } $1 == "u" { split($3, id, ":"); print "u", $2, id[1], id[2] }' \ + "$out/image-etc/accounts.conf" | LC_ALL=C sort > "$out/image-etc/cloud-accounts.lock" +lock="${REPO_ROOT}/dev/os-lock/cloud-accounts.lock" +if ! diff -u <(grep -v '^#' "$lock" 2>/dev/null | LC_ALL=C sort) "$out/image-etc/cloud-accounts.lock" > "$out/accounts.diff"; then + echo "the image's accounts differ from dev/os-lock/cloud-accounts.lock ('-' the lock, '+' the image):" >&2 + cat "$out/accounts.diff" >&2 + echo "A new account: add its line to the lock. A changed id: the image must keep the old id, because every box keeps it." >&2 + exit 1 +fi +echo "ok: $(grep -c . "$out/image-etc/cloud-accounts.lock") accounts and groups, ids as locked" + echo "check 5: rehearse the sign job's signing with a throwaway release CA" mkdir -p "$out/rehearsal/image-etc" sed 's/^path=.*/path=keyring.pem/' "$out/image-etc/system.conf" > "$out/rehearsal/image-etc/system.conf" diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index 57db7e3e..d08428a5 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -151,7 +151,57 @@ fail() { # serial diag, then kill it and keep the run artifacts. exit 1 } +# Every file in the /etc upper layer must be one a Debian-major tidy-up knows +# (BUILD.md # 1b, rule 4): on the keep list (/usr/lib/moose/etc-keep.list, and +# this lane's own list in etc-keep.d/), an account file, a pinned file, or a +# link sshd's run state makes, which host-agent makes again at start. A file +# none of these covers would be lost at a major without anyone choosing that, +# so it fails the boot here first. +etc_upper_check() { + local up=/state/etc/upper pats p pat hit bad="" f + pats="$(cat /usr/lib/moose/etc-keep.list /usr/lib/moose/etc-keep.d/*.list 2>/dev/null | sed -e 's/#.*//' -e 's/[[:space:]]*$//' -e '/^$/d')" + pats="$pats +passwd +group +shadow +gshadow +passwd- +group- +shadow- +gshadow- +subuid- +subgid- +.pwd.lock +docker/daemon.json +login.defs +systemd/system/multi-user.target.wants/ssh.service +systemd/system/sshd.service +rc[0-6S].d/[SK][0-9][0-9]ssh" + while IFS= read -r p; do + [ -n "$p" ] || continue + hit="" + while IFS= read -r pat; do + # shellcheck disable=SC2053 + if [[ $p == $pat || $p == $pat/* ]]; then hit=1; break; fi + done <<<"$pats" + [ -n "$hit" ] || bad="$bad $p" + done < <(cd "$up" && find . -mindepth 1 ! -type d -printf '%P\n' 2>/dev/null) + [ -z "$bad" ] || fail "the /etc upper layer holds files no tidy-up rule covers (add them to /usr/lib/moose/etc-keep.list or give them a rule):$bad" + # Every account and group the image made is on the box, with the image's + # id (BUILD.md # 1b, rule 3; systemd-sysusers adds a missing one at boot). + f=/usr/lib/sysusers.d/moose-image-accounts.conf + [ -s "$f" ] || fail "no $f in the slot" + while read -r kind name id _; do + case "$kind" in + u) [ "$(getent passwd "$name" | cut -d: -f3)" = "${id%%:*}" ] || bad="$bad user:$name(${id%%:*}/$(getent passwd "$name" | cut -d: -f3))" ;; + g) [ "$(getent group "$name" | cut -d: -f3)" = "$id" ] || bad="$bad group:$name($id/$(getent group "$name" | cut -d: -f3))" ;; + esac + done < "$f" + [ -z "$bad" ] || fail "accounts the image made are missing on the box or have another id (image/box):$bad" + echo "cloud-assertions: /etc upper layer: $(cd "$up" && find . -mindepth 1 ! -type d | wc -l) files, all covered by a tidy-up rule; the image's $(grep -c '^[ug] ' "$f") accounts and groups are on the box with their ids" +} ok() { + etc_upper_check # Last gate, every boot (#561): the slot is read-only, so anything that # still writes to it fails. Catch it whether it failed a unit or only # logged. Container logs are left out: an app's own read-only filesystem @@ -197,6 +247,12 @@ if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then systemctl is-active -q host-agent.service && fail "os-revert: host-agent runs on the slot it was meant to be kept off" systemctl list-timers --all --no-pager 2>/dev/null | grep -q moose-os-trial.timer || fail "os-revert: moose-os-trial.timer is not scheduled on the trial boot" echo "cloud-assertions: os-revert: on slot B, host-agent cannot start, trial marker present; grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' '); waiting for the safety net to reboot the box" + # The revert direction of a Debian-major tidy-up (BUILD.md # 1b, rule 4): + # record a major one above this slot's, as a newer major would have, so + # slot A must tidy the upper layer when the safety net takes the box back. + rec_major="$(( $( . /usr/lib/os-release && echo "${VERSION_ID%%.*}" ) + 1 ))" + echo "$rec_major" > /state/etc/.moose-debian-major && sync + echo "cloud-assertions: os-revert: recorded Debian $rec_major on the state partition before the revert" set_os_stage 3 sleep 300 fail "os-revert: the safety net never rebooted the box off the broken slot. $boot_env. grubenv now: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ') timer: $(systemctl list-timers --all --no-pager 2>&1 | grep moose-os-trial) service: $(journalctl -u moose-os-trial.service -b --no-pager 2>&1 | tail -5 | tr '\n' ' ')" @@ -337,6 +393,21 @@ grep -qx "${BOOTED}_OK=1" <<<"$grubenv_now" || layout_fail "grubenv does not mar for u in emergency.service rescue.service; do systemctl cat "$u" 2>/dev/null | grep -q 'systemctl --no-block reboot' || layout_fail "$u has no reboot drop-in" done +# A slot that hangs (#486 point 3, BUILD.md # 1b # As built): systemd feeds a +# hardware watchdog, so a hung PID 1 or kernel resets the box. This lane is +# QEMU q35, like a Hetzner Cloud VM, so it has the ICH9 TCO watchdog +# (iTCO_wdt) that a real box has. +wd_dev="$(systemctl show -p WatchdogDevice --value)" +wd_sec="$(systemctl show -p RuntimeWatchdogUSec --value)" +[ -n "$wd_dev" ] && [ -e /sys/class/watchdog/watchdog0 ] \ + || layout_fail "no hardware watchdog in use (WatchdogDevice='$wd_dev', /sys/class/watchdog: $(ls /sys/class/watchdog 2>/dev/null | tr '\n' ' '))" +[ "$wd_sec" = 1min ] || layout_fail "RuntimeWatchdogUSec is '$wd_sec', want 1min" +# systemd logs this before journald runs, so it is in the kernel log (the +# probe of run 37358065628 found it there and not under _PID=1). +wd_log="$(dmesg 2>/dev/null; journalctl -k -b --no-pager -o cat 2>/dev/null; journalctl -b _PID=1 --no-pager -o cat 2>/dev/null)" +grep -q 'Using hardware watchdog' <<<"$wd_log" \ + || layout_fail "systemd does not feed the hardware watchdog (no 'Using hardware watchdog' in the boot's log)" +echo "cloud-assertions: layout: systemd feeds the hardware watchdog $wd_dev ($(cat /sys/class/watchdog/watchdog0/identity 2>/dev/null)) with a $wd_sec timeout" dmesg 2>/dev/null | grep -q 'moose-state: bind mounts done' || journalctl -k -b --no-pager 2>/dev/null | grep -q 'moose-state: bind mounts done' \ || layout_fail "no 'moose-state: bind mounts done' in the kernel log" # state-setup looked for the state partition on the boot disk only. @@ -2419,6 +2490,7 @@ os-update|os-revert) echo "hostkey=$(ssh-keygen -lf /etc/ssh/ssh_host_ed25519_key.pub 2>/dev/null | awk '{print $2}')" echo "machineid=$(cat /etc/machine-id)" echo "data=$(cat "/home/${owner}/os-update-data.txt" 2>/dev/null)" + echo "groups=$(id -nG "$owner" 2>/dev/null | tr ' ' '\n' | sort | tr '\n' ' ')" } # 4. os-revert, last stage: slot B died before userspace (#575). The @@ -2447,6 +2519,32 @@ os-update|os-revert) [ -n "$session_cookie" ] || fail "$MODE: no moose_session cookie from the SSO landing" owner="$(json_str_of "$(full_get /api/v1/me "$apex" "$session_cookie" 2>/dev/null || true)" username)" [ -n "$owner" ] && id -u "$owner" >/dev/null 2>&1 || fail "$MODE: the SSO owner '$owner' has no host account" + if :; then + # SSH on for the owner, so the faked major below must bring sshd + # back on slot B (BUILD.md # 1b, rule 4: the tidy-up keeps the + # drop-in, drops the enable links, and host-agent enables sshd once). + # A known password first, for the elevation gate, as the ssh boot + # does (harness setup, not a product path). + printf '%s:%s\n' "$owner" 'moose-cloud-lane-owner-pw' | chpasswd || fail "$MODE: could not set a known password for '$owner'" + el="$(full_send POST /api/v1/auth/elevate "$apex" "$session_cookie" '{"password":"moose-cloud-lane-owner-pw"}' 2>/dev/null)" + grep -q ' 200' <<<"$(status_of "$el")" || fail "$MODE: elevate as the owner failed: status='$(status_of "$el")'" + rm -f /run/moose-os-update-key /run/moose-os-update-key.pub + ssh-keygen -t ed25519 -N '' -C 'moose-os-update' -f /run/moose-os-update-key >/dev/null 2>&1 || fail "$MODE: ssh-keygen failed" + on="$(full_send PUT /api/v1/me/ssh "$apex" "$session_cookie" \ + "{\"enabled\":true,\"keys\":[{\"public_key\":\"$(tr -d '\n' < /run/moose-os-update-key.pub)\",\"label\":\"os-update\"}]}" 2>/dev/null)" + grep -q ' 200' <<<"$(status_of "$on")" || fail "$MODE: turning SSH on for the owner failed: status='$(status_of "$on")'" + [ "$(systemctl is-enabled ssh.service 2>/dev/null)" = enabled ] || fail "$MODE: ssh.service is not enabled after the owner turned SSH on" + if [ "$MODE" = os-revert ]; then + # An admin turns sshd off by hand while the owner keeps SSH on + # in the drop-in. The tidy-up on the way back must not turn it + # on again (no marker for host-agent). + systemctl disable --now ssh.service >/dev/null 2>&1 || fail "$MODE: could not turn sshd off by hand" + [ "$(systemctl is-enabled ssh.service 2>/dev/null)" != enabled ] || fail "$MODE: ssh.service still enabled after disable" + echo "cloud-assertions: $MODE: SSH on for '$owner' in the drop-in, sshd turned off by hand before the faked major" + else + echo "cloud-assertions: $MODE: SSH on for '$owner' (ssh.service enabled) before the faked major" + fi + fi head -c 16 /dev/urandom | od -An -tx1 | tr -d ' \n' > "/home/${owner}/os-update-data.txt" chown "$owner" "/home/${owner}/os-update-data.txt" mkdir -p "$OS_STATE_DIR" && chmod 700 "$OS_STATE_DIR" @@ -2546,6 +2644,16 @@ UNIT # 1c. THE SWITCH. An open window: host-agent puts slot B first and # reboots. The next stage runs on slot B. boot_gate + if [ "$MODE" = os-update ]; then + # A faked Debian major (BUILD.md # 1b, rule 4): the state partition + # says the box last ran Debian 12, so slot B must tidy the /etc + # upper layer before it mounts it. An admin's own edit goes to the + # attic; the owner, the password hash, the host keys, machine-id and + # the sudo membership must come through (stage 2 checks them). + echo 12 > /state/etc/.moose-debian-major + echo "# an admin's own edit (os-update boot)" > /etc/moose-test-hand-edit.conf + sync + fi date +%s > "$OS_STATE_DIR/switch-at" set_os_stage 2 write_os_target "$os_sum" "00:00-23:59" @@ -2567,6 +2675,28 @@ UNIT [ "$(/usr/lib/moose/host-agent-real --version | awk '{print $2}')" = "$os_ver" ] || fail "os-update: slot B's host-agent is not $os_ver" want_outcome=good; want_state=current; want_ver="$os_ver" t_good="$(ha_log | grep 'the new slot is healthy and marked good' | tail -1 | awk '{print $1}')" + slot_major="$( . /usr/lib/os-release && echo "${VERSION_ID%%.*}" )" + [ "$(cat /state/etc/.moose-debian-major 2>/dev/null)" = "$slot_major" ] \ + || fail "os-update: the state partition records Debian '$(cat /state/etc/.moose-debian-major 2>/dev/null)', want $slot_major after the tidy-up" + # dmesg first, as the layout checks read the moose-state lines: the + # journal's kernel log was empty here once (run 37374562648). + tidy_log="$(dmesg 2>/dev/null; journalctl -k -b --no-pager -o cat 2>/dev/null)" + grep -q "moose-state: tidied /etc for Debian 12 to $slot_major" <<<"$tidy_log" \ + || fail "os-update: slot B did not tidy the /etc upper layer for the faked major: $(grep 'moose-state' <<<"$tidy_log" | tail -5 | tr '\n' ' ')" + [ ! -e /etc/moose-test-hand-edit.conf ] || fail "os-update: the admin's own /etc edit survived the faked major" + attic_edit="$(ls -d /state/etc/attic/*-debian-12-to-"$slot_major"/moose-test-hand-edit.conf 2>/dev/null | head -1)" + [ -n "$attic_edit" ] || fail "os-update: the admin's own /etc edit is not in the attic: $(ls /state/etc/attic 2>&1 | tr '\n' ' ')" + # sshd comes back: the drop-in was kept, its enable links were not, and + # host-agent enabled the unit once because of the tidy-up marker. + grep -qE "^AllowUsers .*\b${owner}\b" /etc/ssh/sshd_config.d/moose-allowed.conf 2>/dev/null \ + || fail "os-update: the sshd drop-in does not name '$owner' after the faked major: $(cat /etc/ssh/sshd_config.d/moose-allowed.conf 2>&1 | tr '\n' ' ')" + for _i in $(seq 1 60); do [ "$(systemctl is-enabled ssh.service 2>/dev/null)" = enabled ] && break; sleep 1; done + [ "$(systemctl is-enabled ssh.service 2>/dev/null)" = enabled ] \ + || fail "os-update: ssh.service is '$(systemctl is-enabled ssh.service 2>&1)' after the faked major, want enabled: $(ha_log | grep -i 'sshd\|tidy' | tail -3 | tr '\n' ' ')" + ha_log | grep -q "sshd turned on after a Debian-major tidy-up" || fail "os-update: host-agent did not log turning sshd on after the tidy-up" + [ ! -e /state/etc/.moose-major-tidied ] || fail "os-update: host-agent did not remove the tidy-up marker" + echo "cloud-assertions: os-update: SSHD BACK OK (drop-in names '$owner', ssh.service enabled again by host-agent after the tidy-up, marker removed)" + echo "cloud-assertions: os-update: MAJOR TIDY OK (faked Debian 12 to $slot_major: the upper layer was rebuilt, the admin's edit is in $(dirname "$attic_edit"), $(grep -o 'kept [0-9]* files' <<<"$tidy_log" | tail -1))" echo "cloud-assertions: os-update: SWITCH OK (booted slot B, marked good, grubenv ORDER='B A' B_OK=1 B_TRY=0); measured: switch to marked good $(awk -v a="$(cat "$OS_STATE_DIR/switch-at")" -v b="$t_good" 'BEGIN{printf "%.0f", b-a}') s, the reboot included" note_pat="updated its system to $os_ver" else @@ -2578,6 +2708,21 @@ UNIT [ "$(grubvar B_OK)" = 0 ] && [ "$(grubvar A_OK)" = 1 ] || fail "os-revert: grubenv after the revert: $(grub-editenv /efi/grub/grubenv list | tr '\n' ' ')" [ "$(/usr/lib/moose/host-agent-real --version | awk '{print $2}')" = "$base_ver" ] || fail "os-revert: back on slot A, host-agent is not $base_ver" want_outcome=reverted; want_state=held; want_ver="$base_ver" + # The revert direction of the tidy-up: slot B recorded a newer major, + # so slot A tidied the upper layer on the way back. + slot_major="$( . /usr/lib/os-release && echo "${VERSION_ID%%.*}" )" + [ "$(cat /state/etc/.moose-debian-major 2>/dev/null)" = "$slot_major" ] \ + || fail "os-revert: the state partition records Debian '$(cat /state/etc/.moose-debian-major 2>/dev/null)' back on slot A, want $slot_major" + tidy_log="$(dmesg 2>/dev/null; journalctl -k -b --no-pager -o cat 2>/dev/null)" + grep -q "moose-state: tidied /etc for Debian $(( slot_major + 1 )) to $slot_major: .* sshd was off" <<<"$tidy_log" \ + || fail "os-revert: slot A did not tidy the upper layer on the way back (or saw sshd on): $(grep 'moose-state' <<<"$tidy_log" | tail -4 | tr '\n' ' ')" + ls -d /state/etc/attic/*-debian-$(( slot_major + 1 ))-to-"$slot_major" >/dev/null 2>&1 || fail "os-revert: no attic entry for the tidy-up on the way back" + # sshd was off by hand, so no marker, and host-agent left it off. + [ ! -e /state/etc/.moose-major-tidied ] || fail "os-revert: the tidy-up left the sshd marker although sshd was off" + grep -qE "^AllowUsers .*\b${owner}\b" /etc/ssh/sshd_config.d/moose-allowed.conf 2>/dev/null || fail "os-revert: the sshd drop-in lost '$owner' on the way back" + sleep 5 + [ "$(systemctl is-enabled ssh.service 2>/dev/null)" != enabled ] || fail "os-revert: ssh.service is enabled again after the tidy-up, but the admin had turned it off" + echo "cloud-assertions: os-revert: MAJOR TIDY BACK OK (Debian $(( slot_major + 1 )) to $slot_major on slot A, attic entry made, sshd left off as the admin set it)" echo "cloud-assertions: os-revert: REVERT OK (slot B never came up; back on slot A, B marked bad, A good); measured: switch to back on slot A $(( $(date +%s) - $(cat "$OS_STATE_DIR/switch-at") )) s, two reboots and the safety net included" note_pat="did not work, so moose went back" fi diff --git a/dev/cloud/mkosi.extra/etc/systemd/system.conf.d/10-moose-watchdog.conf b/dev/cloud/mkosi.extra/etc/systemd/system.conf.d/10-moose-watchdog.conf index be8bf105..d948ede1 100644 --- a/dev/cloud/mkosi.extra/etc/systemd/system.conf.d/10-moose-watchdog.conf +++ b/dev/cloud/mkosi.extra/etc/systemd/system.conf.d/10-moose-watchdog.conf @@ -1,6 +1,7 @@ -# A slot that hangs must still revert (UPDATES.md # 1): where the machine has a -# hardware watchdog, systemd feeds it, and a hung PID 1 gets a reset. On a -# machine with none (the QEMU lane; Hetzner is not known yet, NEXT.md # A/B OS -# image point 3), systemd logs that and carries on. +# A slot that hangs must still revert (UPDATES.md # 1): systemd feeds the +# hardware watchdog, and a hung PID 1 or kernel gets a reset. Hetzner Cloud VMs +# (QEMU q35) have the ICH9 TCO watchdog, which the generic kernel's iTCO_wdt +# drives; the reset comes about twice this value after the last feed +# (BUILD.md # 1b # As built). The boot lane checks systemd uses it. [Manager] RuntimeWatchdogSec=60s diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/etc-keep.list b/dev/cloud/mkosi.extra/usr/lib/moose/etc-keep.list new file mode 100644 index 00000000..51bc35ac --- /dev/null +++ b/dev/cloud/mkosi.extra/usr/lib/moose/etc-keep.list @@ -0,0 +1,26 @@ +# The files in the /etc upper layer that a Debian-major tidy-up keeps as they +# are (BUILD.md # 1b, rule 4; state-setup step 3a). One shell glob per line, +# relative to /etc; a directory keeps everything under it. Everything else in +# the upper layer moves to the attic at a major, except the account files +# (merged) and the pinned daemon.json and login.defs (taken again from the slot +# when that keeps the box's remap). The boot lane fails a boot whose upper +# layer holds a file this list, the account files, the pinned files and the +# links sshd's run state makes do not cover (cloud-assertions.sh), so a new +# file moose writes into /etc must be added here, or given a regeneration rule. +# +# The box's identity. +machine-id +ssh/ssh_host_* +# What users set: SSH access (the drop-in is the enabled set) and keys, and +# the time zone. +ssh/sshd_config.d/moose-allowed.conf +ssh/moose-authorized-keys +localtime +# The box's remap range, pinned at first boot (rule 2). +subuid +subgid +# Appliance only (#564): the LUKS recovery key, the data-drive marker and the +# network connections, WiFi included. +moose/secrets +moose/data-drive.enrolled +NetworkManager/system-connections diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/state-setup b/dev/cloud/mkosi.extra/usr/lib/moose/state-setup index 325ecfb4..d74a3e08 100755 --- a/dev/cloud/mkosi.extra/usr/lib/moose/state-setup +++ b/dev/cloud/mkosi.extra/usr/lib/moose/state-setup @@ -9,7 +9,9 @@ # and grows the state partition on a later boot when the disk grew. # 2. The state partition is checked and mounted at /state, and its ext4 is # grown to fill the partition. -# 3. /etc becomes an overlay: the slot's /etc below, /state/etc/upper above. +# 3. When the slot's Debian major is not the one the box last booted, the +# /etc upper layer is tidied first (rule 4 below). Then /etc becomes an +# overlay: the slot's /etc below, /state/etc/upper above. # 4. On the first boot only, the four pinned files are copied up and never # follow the image again. # 5. The bind mounts from the state partition. When a state directory does @@ -114,6 +116,189 @@ mount -t ext4 -o rw,noatime "$dev" "$STATE" || die "mount $dev on $STATE failed" /usr/lib/systemd/systemd-growfs "$STATE" >/dev/null 2>&1 || die "systemd-growfs $STATE failed" log "state partition $dev mounted on $STATE" +# --- 3a. a Debian major (BUILD.md # 1b, rule 4). A file in the upper layer +# hides the slot's copy, so after a move to another Debian major it would stay +# at the old major's version under the new major's packages. So when the +# slot's major is not the one recorded on the state partition, the upper layer +# is rebuilt before the overlay is mounted, from three things: +# - the files moose keeps (/usr/lib/moose/etc-keep.list, and any list in +# /usr/lib/moose/etc-keep.d/), copied as they are: the box's identity and +# its users' settings; +# - the account files, merged: the slot's own entries, plus the box's +# entries the slot does not have, with group members joined; +# - the pinned files, taken again from the slot when that keeps the box's +# remap as it was, and kept as they are otherwise. +# Everything else (an admin's own edits, files a package wrote at run time) +# moves to an attic on the state partition, where it can still be read. It +# runs in both directions, so a revert to the older major is tidied the same +# way. The old upper layer moves to the attic whole, so nothing is lost. +MAJOR_FILE="$STATE/etc/.moose-debian-major" +# Set by a swap that took place; host-agent reads it once (below). +TIDIED_MARKER="$STATE/etc/.moose-major-tidied" +# Debian testing and sid have no VERSION_ID. Then this step is skipped: no +# tidy-up and no record. It never stops a boot. +slot_major="$( . /usr/lib/os-release 2>/dev/null; v="${VERSION_ID:-}"; echo "${v%%.*}" )" || slot_major="" + +keep_patterns() { + local f + for f in /usr/lib/moose/etc-keep.list /usr/lib/moose/etc-keep.d/*.list; do + [ -f "$f" ] || continue + sed -e 's/#.*//' -e 's/[[:space:]]*$//' -e '/^$/d' "$f" + done +} + +# merge_accounts FILE SLOT BOX: the slot's file with the box's entries the slot +# lacks. passwd: the slot's line wins (the slot's packages own it). shadow and +# gshadow: the box's password field wins, so a password set on the box stays. +# group and gshadow: members are joined. +merge_accounts() { + awk -F: -v OFS=: -v kind="$1" ' + function join(a, b, n, i, m, out, seen) { + out = ""; split("", seen) + n = split(a "," b, m, ",") + for (i = 1; i <= n; i++) if (m[i] != "" && !(m[i] in seen)) { seen[m[i]] = 1; out = out (out == "" ? "" : ",") m[i] } + return out + } + FNR == NR { box[$1] = $0; order[++nb] = $1; next } + { + name = $1; slot[name] = 1 + if (name in box) { + split(box[name], b, ":") + if (kind == "passwd" && b[3] != $3) printf "moose-state: account %s has uid %s on the box and %s in the slot\n", name, b[3], $3 > "/dev/stderr" + if (kind == "group" && b[3] != $3) printf "moose-state: group %s has gid %s on the box and %s in the slot\n", name, b[3], $3 > "/dev/stderr" + if (kind == "shadow") $2 = b[2] + if (kind == "group") $4 = join($4, b[4]) + if (kind == "gshadow") { $2 = b[2]; $3 = join($3, b[3]); $4 = join($4, b[4]) } + } + print + } + END { for (i = 1; i <= nb; i++) if (!(order[i] in slot)) print box[order[i]] } + ' "$3" "$2" +} + +# tidy_build FROM TO: build the new upper layer at upper.moose-new. It never +# dies: it returns non-zero on any failure, and the caller boots with the +# upper layer as it was. A bug here must not stop the box from booting, on +# either slot, because both slots run this same step. +tidy_build() { + local up="$STATE/etc/upper" new="$STATE/etc/upper.moose-new" pat p f line + rm -rf "$new" && mkdir -p "$new" && chmod 0755 "$new" || return 1 + # The files moose keeps, as they are (whiteouts and opaque dirs included). + # A pattern with no match is skipped: nullglob drops a glob, and the test + # drops a plain path the box does not have. + ( + cd "$up" || exit 1 + shopt -s nullglob dotglob + for pat in $(keep_patterns); do + for p in $pat; do + [ -e "$p" ] || [ -L "$p" ] || continue + cp -a --parents "$p" "$new/" || exit 1 + done + done + ) || { log "tidy-up: copying the kept files failed"; return 1; } + # The account files, merged. + for f in passwd group shadow gshadow; do + [ -f "$up/$f" ] || continue + merge_accounts "$f" "/etc/$f" "$up/$f" > "$new/$f" 2> "$TMPDIR/merge.log" \ + || { log "tidy-up: merging /etc/$f failed"; return 1; } + while IFS= read -r line; do log "${line#moose-state: }"; done < "$TMPDIR/merge.log" + chown --reference="/etc/$f" "$new/$f" && chmod --reference="/etc/$f" "$new/$f" || return 1 + done + # The pinned files (rule 2). subuid and subgid are the box's remap range: + # they are on the keep list. daemon.json and login.defs are taken again from + # the slot when that keeps the box's remap, and kept as they are otherwise. + mkdir -p "$new/docker" || return 1 + if [ "$(grep -o '"userns-remap"[^,}]*' /etc/docker/daemon.json 2>/dev/null)" = "$(grep -o '"userns-remap"[^,}]*' "$up/docker/daemon.json" 2>/dev/null)" ]; then + cp -p /etc/docker/daemon.json "$new/docker/daemon.json" || return 1 + else + cp -a "$up/docker/daemon.json" "$new/docker/daemon.json" || return 1 + log "tidy-up: the slot's daemon.json would change this box's remap, so the box's copy stays" + fi + if grep -Eq '^SUB_UID_COUNT[[:space:]]+0$' /etc/login.defs && grep -Eq '^SUB_GID_COUNT[[:space:]]+0$' /etc/login.defs; then + cp -p /etc/login.defs "$new/login.defs" || return 1 + else + cp -a "$up/login.defs" "$new/login.defs" || return 1 + log "tidy-up: the slot's login.defs does not set SUB_UID_COUNT and SUB_GID_COUNT to 0, so the box's copy stays" + fi +} + +# tidy_swap FROM TO: the old upper layer goes to the attic and the new one +# takes its place. A power cut between the two renames is finished on the next +# boot (below); one before the first rename starts the tidy-up again. +tidy_swap() { + local up="$STATE/etc/upper" new="$STATE/etc/upper.moose-new" attic + attic="$STATE/etc/attic/$(date -u +%Y%m%dT%H%M%SZ)-debian-$1-to-$2" + mkdir -p "$(dirname "$attic")" || return 1 + # Was sshd on? Its enable link is in the old layer only when the box turned + # it on (the image ships it off). Read before the swap moves the layer. + local sshd_on="" + [ -L "$up/systemd/system/multi-user.target.wants/ssh.service" ] && sshd_on=1 + mv "$up" "$attic" || return 1 + if ! mv "$new" "$up"; then + # Put the old layer back. If even that fails, the copy at + # upper.moose-new is kept: the caller never removes it while upper is + # missing, and the next boot finishes the swap from it. + [ -d "$up" ] || mv "$attic" "$up" || log "tidy-up: could not put the old upper layer back from $attic" + return 1 + fi + rm -rf "$STATE/etc/work" + mkdir -p "$STATE/etc/work" + # For host-agent: sshd's enable links went to the attic (BUILD.md # 1b, + # rule 4), so it turns sshd back on once, at its next start. Only when sshd + # was on: an admin who turned it off by hand keeps it off. A marker that is + # already there stays: it comes from a swap a power cut stopped before the + # record, and this repeat sees the tidied layer, which has no sshd link. + if [ -n "$sshd_on" ]; then + touch "$TIDIED_MARKER" 2>/dev/null || log "tidy-up: could not leave $TIDIED_MARKER for host-agent" + fi + log "tidied /etc for Debian $1 to $2: kept $(find "$up" ! -type d | wc -l) files, sshd was $([ -n "$sshd_on" ] && echo on || echo off), the old upper layer is in $attic" +} + +# A tidy-up a power cut stopped: between the two renames, finish it; before +# them, the half-built copy goes and the tidy-up runs again. +if [ -d "$STATE/etc/upper.moose-new" ]; then + if [ -d "$STATE/etc/upper" ]; then + rm -rf "$STATE/etc/upper.moose-new" + else + mv "$STATE/etc/upper.moose-new" "$STATE/etc/upper" + log "finished a tidy-up of /etc that a reboot cut short" + fi +fi +box_major="$(cat "$MAJOR_FILE" 2>/dev/null || true)" +record_major=1 +if [ -z "$slot_major" ]; then + record_major="" + log "the slot's /usr/lib/os-release has no VERSION_ID; no Debian-major tidy-up of /etc on this boot" +elif [ -n "$box_major" ] && [ "$box_major" != "$slot_major" ] && [ -e "$STATE/etc/.moose-pinned" ] && [ -d "$STATE/etc/upper" ]; then + if tidy_build "$box_major" "$slot_major" && tidy_swap "$box_major" "$slot_major"; then + : + else + # The box boots with the upper layer as it was, and the next boot + # tries again, because the old major stays recorded. The half-built + # copy is removed only while the old layer is in place; with no upper + # layer it is all the box has, and the next boot's check above uses it. + if [ -d "$STATE/etc/upper" ]; then + rm -rf "$STATE/etc/upper.moose-new" + else + # Both renames back failed. Booting would mount an empty upper + # layer: no users, host keys or machine-id. Stop instead; the next + # boot moves upper.moose-new into place (the check above). + die "tidy-up: no /etc upper layer after a failed swap" + fi + record_major="" + log "tidy-up of /etc for Debian $box_major to $slot_major FAILED; booting with the upper layer as it was" + fi +fi +# Recorded last, so a tidy-up a power cut stopped runs again. A box without the +# record (a new box, or one made before the tidy-up shipped) gets the slot's. +# A failed write (a full state partition) is logged and the boot goes on. +if [ -n "$record_major" ] && [ "$box_major" != "$slot_major" ]; then + if ! { mkdir -p "$STATE/etc" && echo "$slot_major" > "$MAJOR_FILE.new" && mv -f "$MAJOR_FILE.new" "$MAJOR_FILE"; }; then + log "could not record Debian major $slot_major in $MAJOR_FILE; the next boot tries again" + fi +fi +log "Debian major of the slot: $slot_major; recorded on the state partition: $(cat "$MAJOR_FILE" 2>/dev/null)" + # --- 3. /etc as an overlay. Every change the box makes in /etc (users, # passwords, SSH host keys, machine-id, the rendered sshd drop-ins) lands in # the upper layer. A file the box never touched still comes from the slot. diff --git a/dev/cloud/mkosi.postinst.chroot b/dev/cloud/mkosi.postinst.chroot index c13bcff4..2e4d21f1 100755 --- a/dev/cloud/mkosi.postinst.chroot +++ b/dev/cloud/mkosi.postinst.chroot @@ -278,3 +278,19 @@ ln -sf /etc/machine-id /var/lib/dbus/machine-id # journal catalog) run on every boot, which is what an image that can swap # under /etc needs anyway. systemctl --root=/ mask systemd-repart.service ldconfig.service systemd-update-done.service >/dev/null + +# Every account and group the image's packages made, as a sysusers.d file +# (BUILD.md # 1b, rule 3). The box's /etc/passwd and /etc/group are in the +# /etc upper layer and hide the slot's, so an account a later image's packages +# make at build time would be missing on the box. systemd-sysusers runs on +# every boot (systemd-update-done is masked above) and adds any of these the +# box lacks, with the image's ids. It changes nothing on a box that has them. +# The build checks the ids against dev/os-lock/cloud-accounts.lock +# (dev/cloud/build-bundle.sh), because a box keeps an account's id for life. +{ + echo "# Generated at build time from the image's /etc/group and /etc/passwd" + echo "# (dev/cloud/mkosi.postinst.chroot). Do not edit." + awk -F: '{ printf "g %s %s\n", $1, $3 }' /etc/group + awk -F: '{ printf "u %s %s:%s %s %s %s\n", $1, $3, $4, ($5 == "" ? "-" : "\"" $5 "\""), $6, $7 }' /etc/passwd + awk -F: '$4 != "" { n = split($4, m, ","); for (i = 1; i <= n; i++) printf "m %s %s\n", m[i], $1 }' /etc/group +} > /usr/lib/sysusers.d/moose-image-accounts.conf diff --git a/dev/cloud/test/bootstrap.sh b/dev/cloud/test/bootstrap.sh index 640daa29..44109f46 100755 --- a/dev/cloud/test/bootstrap.sh +++ b/dev/cloud/test/bootstrap.sh @@ -33,7 +33,7 @@ WIRING="${CLOUD_DIR}/mkosi.extra.wiring" # shared production wiring (ExtraTree o PKGMNGR="${TEST_DIR}/mkosi.pkgmngr" CP_BUNDLE="${REPO_ROOT}/.dev/control-plane" CANARY="${WORK}/.cloud-boot-ready" -CANARY_VERSION="v29" # bump when staging/mkosi.conf/repart changes require a clean rebuild +CANARY_VERSION="v30" # bump when staging/mkosi.conf/repart changes require a clean rebuild # A change to the OS package lock (#560) must rebuild too, so all three lock # files are part of the canary. The resolved list is in it as well: a re-run # after only the list changed must not exit early and skip os_lock_check below. @@ -134,6 +134,15 @@ mkdir -p "$EXTRA/usr/local/bin" "$EXTRA/etc/systemd/system" cp "${CLOUD_DIR}/cloud-assertions.sh" "$EXTRA/usr/local/bin/cloud-assertions.sh" chmod 0755 "$EXTRA/usr/local/bin/cloud-assertions.sh" cp "${TEST_DIR}/moose-cloud-assertions.service" "$EXTRA/etc/systemd/system/" +# What this lane writes into /etc at run time: host-agent drop-ins that point +# it at the in-guest update targets. They go on a keep list of their own, so a +# Debian-major tidy-up (BUILD.md # 1b, rule 4) keeps them across the os-update +# boot's faked major, and the end-of-boot check of the upper layer accepts them. +mkdir -p "$EXTRA/usr/lib/moose/etc-keep.d" +cat > "$EXTRA/usr/lib/moose/etc-keep.d/boot-lane.list" <<'EOF' +# The boot lane's own writes into /etc (dev/cloud/test/bootstrap.sh). +systemd/system/host-agent.service.d +EOF # The OS update trial's safety net fires after 90 s here instead of 15 min # (#563), so the os-revert boot does not sit out a quarter of an hour. A diff --git a/dev/os-lock/cloud-accounts.lock b/dev/os-lock/cloud-accounts.lock new file mode 100644 index 00000000..f9eebdc3 --- /dev/null +++ b/dev/os-lock/cloud-accounts.lock @@ -0,0 +1,73 @@ +# The account and group ids of the hosted image (BUILD.md # 1b, rule 3). +# A box keeps them for life, so dev/cloud/build-bundle.sh fails a build whose +# image differs. Edited by hand: add a new account's line; never change an id. +# "g NAME GID" and "u NAME UID GID", sorted. +g _ssh 101 +g adm 4 +g audio 29 +g backup 34 +g bin 2 +g cdrom 24 +g clock 994 +g daemon 1 +g dialout 20 +g dip 30 +g disk 6 +g docker 991 +g fax 21 +g floppy 25 +g games 60 +g input 996 +g irc 39 +g kmem 15 +g kvm 993 +g list 38 +g lp 7 +g mail 8 +g man 12 +g messagebus 997 +g news 9 +g nogroup 65534 +g operator 37 +g plugdev 46 +g proxy 13 +g render 992 +g root 0 +g sasl 45 +g sgx 995 +g shadow 42 +g src 40 +g staff 50 +g sudo 27 +g sys 3 +g systemd-journal 999 +g systemd-network 998 +g tape 26 +g tty 5 +g users 100 +g utmp 43 +g uucp 10 +g video 44 +g voice 22 +g www-data 33 +u _apt 42 65534 +u backup 34 34 +u bin 2 2 +u daemon 1 1 +u games 5 60 +u irc 39 39 +u list 38 38 +u lp 7 7 +u mail 8 8 +u man 6 12 +u messagebus 997 997 +u news 9 9 +u nobody 65534 65534 +u proxy 13 13 +u root 0 0 +u sshd 990 65534 +u sync 4 65534 +u sys 3 3 +u systemd-network 998 998 +u uucp 10 10 +u www-data 33 33 diff --git a/docs/architecture.md b/docs/architecture.md index 7778d62a..17d7d8a7 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -202,7 +202,18 @@ So this doc isn't read as a claim about the finished product: `os_version`/`os_slot` and stream A's decision, and raises one admin notification per outcome. Proven by the `os-update` and `os-revert` boots under both firmwares, and `os-revert` also proves that GRUB skips a slot - whose kernel panics before userspace (#575). Not built: the OS part of the private control plane's + whose kernel panics before userspace (#575). **A Debian major tidies the + `/etc` upper layer** (`BUILD.md` # 1b, rule 4): `state-setup` keeps the + files on `/usr/lib/moose/etc-keep.list`, merges the account files, takes the + pinned files again from the slot when the remap stays, and moves the rest to + an attic, in both directions; the `os-update` boot fakes a major, and every + boot fails on an upper-layer file no rule covers. The image's accounts are a + generated `sysusers.d` file whose ids the build checks against + `dev/os-lock/cloud-accounts.lock`, and host-agent turns sshd back on once, + at its first start after a tidy-up, when the drop-in names an account + (`sshaccess.EnsureOnAtStart`, gated by `/state/etc/.moose-major-tidied`). systemd + feeds the hardware watchdog a Hetzner VM has (ICH9 TCO), checked on every + boot. Not built: the OS part of the private control plane's answer (described in `docs/progress/host-agent-os-update.md`), so no production box moves its OS yet, and the appliance layout (#564). **The OS package lock is built (#560)** for the hosted image: `dev/os-lock/` holds a diff --git a/docs/dev/hosted-boot-proof.md b/docs/dev/hosted-boot-proof.md index 3d2c3052..763ed5e3 100644 --- a/docs/dev/hosted-boot-proof.md +++ b/docs/dev/hosted-boot-proof.md @@ -32,7 +32,7 @@ Net: a provisioned box logs both milestones, binds `:443`, and serves every `_TRY=1` in the grubenv GRUB left for this boot (#575; the `layout: grubenv at boot:` line, logged by `moose-test-grubenv.service` before host-agent can mark the slot good), no `NvVars` file on the ESP, and the reboot drop-ins on emergency and rescue. It also prints the measured numbers for the disk budget (`BUILD.md` # 1b # Disk budget) in one `layout: measured:` line: the squashfs size read from slot A's superblock, the ESP's use, and the state partition's use. A red boot here is the image or the initramfs, not the control plane: look for `moose-state:` lines on the serial console (the initramfs setup logs to the kernel log), or a `Kernel panic`/reboot right after them. A box that never reaches the serial verdict and shows the UEFI shell (or `BdsDxe: failed to load`) did not find GRUB on the ESP: check the ESP's FAT (runs 36925538434 and 36927908164 had a FAT32 with too few clusters, fixed by `SectorSize=512` in `dev/cloud/mkosi.conf`). **The build checks the slot budget too**: the job summary has an "OS slot budget" table from `dev/cloud/slotbudget`, and the lean build fails when the squashfs in slot A is over 60% of the slot. **The build makes the OS update bundle too** (#562, `BUILD.md` # 1b # The bundle): the step "Build the OS update bundle" runs `dev/cloud/build-bundle.sh`, which prints `check 1` to `check 5` and an `ok:` line for each, and adds an "OS update bundle" size table to the job summary. A red `check 1` means the slot carries another keyring than the build staged (look at the `rauc keyring:` line of the image build). A red `check 3` on a release run means the image would trust the throwaway signer, which must never ship. "unsuitable certificate purpose" means a signer without the codeSigning purpose, or a `system.conf` without `check-purpose=codesign`. "no release root CA" or "no release signer" on a run that publishes the OS means the maintainer's setup in `rauc-signing.md` is not done yet. **Every boot checks which control plane it runs** (#566, `BUILD.md` # Versioning), in step 5a of `cloud-assertions.sh`: the running `moose-brain` and `moose-ui` containers must run the image IDs the image recorded in `/usr/lib/moose/control-plane.env`. The line is `control plane baked: `, and a released bake also prints `layout: control plane baked from ghcr:` with the two pinned refs. The build job's summary shows the same record ("Control plane baked into the image"), and the boot-proof build fails when its record differs from the image that ships. A run that publishes only the OS bakes `released`; to see that path on a branch, add `-f control_plane=released`. A red "cannot resolve ghcr.io/onmoose/brain:vX.Y.Z" in the image build means `CONTROL_PLANE_VERSION` names a control-plane release that is not on ghcr. **Every boot also ends with a gate** (`ok()`): no unit may have failed, and no host process may have hit `Read-only file system`, because the slot is read-only and a write the inventory missed must fail here rather than on a box. +**Every boot checks the A/B layout** (#561, `BUILD.md` # 1b) first, in step 1b of `cloud-assertions.sh`. Its serial lines start `cloud-assertions: layout:`. It wants 5 partitions after first boot (ESP, BIOS boot, slot A, slot B, state) with a 128 MiB ESP and 1 GiB slots, a state partition that reaches the end of the 24G disk with an ext4 that fills it, `/` a read-only squashfs booted from slot A by GRUB, with root-owned files (`/usr/bin/sudo` is `0 4755`) (`rauc.slot=A`, `panic=10`, `BOOT_IMAGE=/usr/lib/moose/boot/vmlinuz` on the command line), `/etc` as the overlay on `/state/etc/upper`, every bind mount on the state partition (`/home` from `srv/moose/home`, `/var/lib/moose` a bind mount of its own), `/var/tmp` a tmpfs, the four pinned files, the SSH host keys and `machine-id` in the upper layer, `rauc status` booted from slot A, the RAUC keyring `/etc/rauc/keyring.pem` present and coming from the slot, not the upper layer, with `check-purpose=codesign` in `system.conf` (#562; the `layout: rauc keyring from the slot:` line names it: a throwaway root on a run that publishes no OS image), `A_OK=1` in the grubenv, `_TRY=1` in the grubenv GRUB left for this boot (#575; the `layout: grubenv at boot:` line, logged by `moose-test-grubenv.service` before host-agent can mark the slot good), no `NvVars` file on the ESP, the reboot drop-ins on emergency and rescue, and a hardware watchdog that systemd feeds with a 1 min timeout (the `layout: systemd feeds the hardware watchdog` line; this lane is QEMU `q35`, so it has the ICH9 TCO watchdog a Hetzner VM has). It also prints the measured numbers for the disk budget (`BUILD.md` # 1b # Disk budget) in one `layout: measured:` line: the squashfs size read from slot A's superblock, the ESP's use, and the state partition's use. A red boot here is the image or the initramfs, not the control plane: look for `moose-state:` lines on the serial console (the initramfs setup logs to the kernel log), or a `Kernel panic`/reboot right after them. A box that never reaches the serial verdict and shows the UEFI shell (or `BdsDxe: failed to load`) did not find GRUB on the ESP: check the ESP's FAT (runs 36925538434 and 36927908164 had a FAT32 with too few clusters, fixed by `SectorSize=512` in `dev/cloud/mkosi.conf`). **The build checks the slot budget too**: the job summary has an "OS slot budget" table from `dev/cloud/slotbudget`, and the lean build fails when the squashfs in slot A is over 60% of the slot. **The build makes the OS update bundle too** (#562, `BUILD.md` # 1b # The bundle): the step "Build the OS update bundle" runs `dev/cloud/build-bundle.sh`, which prints `check 1` to `check 6` and an `ok:` line for each, and adds an "OS update bundle" size table to the job summary. A red `check 1` means the slot carries another keyring than the build staged (look at the `rauc keyring:` line of the image build). A red `check 3` on a release run means the image would trust the throwaway signer, which must never ship. "unsuitable certificate purpose" means a signer without the codeSigning purpose, or a `system.conf` without `check-purpose=codesign`. "no release root CA" or "no release signer" on a run that publishes the OS means the maintainer's setup in `rauc-signing.md` is not done yet. **Every boot checks which control plane it runs** (#566, `BUILD.md` # Versioning), in step 5a of `cloud-assertions.sh`: the running `moose-brain` and `moose-ui` containers must run the image IDs the image recorded in `/usr/lib/moose/control-plane.env`. The line is `control plane baked: `, and a released bake also prints `layout: control plane baked from ghcr:` with the two pinned refs. The build job's summary shows the same record ("Control plane baked into the image"), and the boot-proof build fails when its record differs from the image that ships. A run that publishes only the OS bakes `released`; to see that path on a branch, add `-f control_plane=released`. A red "cannot resolve ghcr.io/onmoose/brain:vX.Y.Z" in the image build means `CONTROL_PLANE_VERSION` names a control-plane release that is not on ghcr. **Every boot also ends with a gate** (`ok()`): no unit may have failed, and no host process may have hit `Read-only file system`, because the slot is read-only and a write the inventory missed must fail here rather than on a box. The gate also checks the `/etc` upper layer (`BUILD.md` # 1b, rules 3 and 4): every file in it must be on a keep list (`/usr/lib/moose/etc-keep.list`, or the lane's own `etc-keep.d/boot-lane.list` for its host-agent drop-ins), an account file, a pinned file or an sshd link, and every account in the image's `moose-image-accounts.conf` must be on the box with the same id. A red "holds files no tidy-up rule covers" means moose or a test wrote a new file into `/etc`: add it to the keep list, or give it a rule. **The `os-update` boot fakes a Debian major**: before the switch it records Debian 12 on the state partition and plants an admin's edit in `/etc`. On slot B it wants the `moose-state: tidied /etc for Debian 12 to 13` kernel log line, the major recorded as 13, the edit in the attic and gone from `/etc`, and the owner's password hash, host key, `machine-id` and groups unchanged (`MAJOR TIDY OK`). The owner turned SSH on before the switch, so slot B also wants the drop-in kept and `ssh.service` enabled again by host-agent (`SSHD BACK OK`). **The `os-revert` boot fakes the revert direction:** slot B records Debian 14 before the safety net reboots it, and slot A must tidy 14 to 13, with sshd, turned off by hand before the switch, left off (`MAJOR TIDY BACK OK`). **The build checks the account ids** (`check 6` of `build-bundle.sh`): the image's accounts must match `dev/os-lock/cloud-accounts.lock`. A red check prints the difference; add a new account's line by hand, and fix the build when an id moved. **Every boot checks the userns-remap** (#530, `BUILD.md` # User-namespace remap) before its scenario runs. The serial log shows the two `docker info` lines (`security options: [... "name=userns" ...]` and `storage driver: overlay2, root dir: /var/lib/docker/1000000.1000000`) and then `userns-remap on (...)`. A red boot at this step means the image lost part of the remap setup, or the control plane no longer splits across it (proxy and brain in the host user namespace, Caddy and `moose-ui` remapped). The `access` boot also installs the `imageuser` fixture and checks that it runs remapped as its images' own users. The `remap` boots check the other tiers, one app each (the row above). Their serial lines start `cloud-assertions: remap [remap]:` and `cloud-assertions: remap [remap-reboot]:`, one per tier, and each names what it saw, so a red one says which tier and which field was wrong. diff --git a/docs/progress/README.md b/docs/progress/README.md index 2fecd582..9983414c 100644 --- a/docs/progress/README.md +++ b/docs/progress/README.md @@ -315,3 +315,4 @@ Oldest first; append new entries to the bottom. | [host-agent-os-update.md](host-agent-os-update.md) — Closes #563, a slice of #486, after [hosted-ab-layout.md](hosted-ab-layout.md) and [rauc-bundle.md](rauc-bundle.md). **host-agent applies OS updates on hosted.** The update-target answer gains an optional `os` list; host-agent picks the next release (never skipping a minor, at most one minor back, never below the control plane's floor), downloads the bundle, checks its sha256 against the answer before RAUC sees it, installs it into the other slot ahead of the window (`activate-installed=false`), switches inside it after stream B (checking the grubenv reads `OK=1 TRY=0`), and on the next boot marks the slot good once the brain answers or reboots back; an image timer reboots a slot whose host-agent never started. Only the first boot after a switch is on trial. Stream A is reported on the version and update-target reads, and admins get one notification per outcome. New `os-update` and `os-revert` boots prove both paths under UEFI and BIOS, in every full run. Gaps: the private control plane does not send the `os` part yet (wire described for the maintainer), QEMU only, appliance waits for #564 | done | | [uefi-grub-try.md](uefi-grub-try.md) — Closes #575, a slice of #486, after [host-agent-os-update.md](host-agent-os-update.md). **GRUB's try flag now survives under UEFI in the boot lane.** The cause was the harness, not GRUB or the image: it ran OVMF with no writable VARS store (Ubuntu 24.04 renamed the files), so OVMF saved its variables to an `NvVars` file on the ESP at every boot and lost GRUB's `save_env` write (traced in runs 37061482623, 37063254219 and 37064755329: GRUB wrote and read back `A_TRY=1`, the initramfs read `A_TRY=0`). The harness now needs a CODE and VARS pair. Every boot checks GRUB saved `_TRY=1` and the ESP has no `NvVars`, under both firmwares, and `os-revert` gains a last stage: slot B made active again crashes its kernel in the initramfs (a test-only hook), and GRUB must skip it. The trial timer stays on the marker, not the grubenv. No image change, no disk cost. **Gaps:** QEMU only, a real UEFI provider box not checked for `NvVars`; the appliance medium lane has the same OVMF fallback (#564) | done | | [bake-released-control-plane.md](bake-released-control-plane.md) — Part of #566 (kept open until the released path is green), a slice of #486, after [control-plane-version-line.md](control-plane-version-line.md). **An OS-only release bakes the last released control plane:** the brain and UI of `v`, resolved to digests once and pulled from ghcr by digest (`make control-plane-released`, `dev/release/ghcr-resolve.sh`). A control-plane release, or one that bumps both files, bakes the pair it builds and releases; a PR, a lock bump or a dispatch bakes a build of its commit (`-f control_plane=released` checks the release path ahead). The image records the pair in `/usr/lib/moose/control-plane.env`, and every boot checks the running brain and UI against it. The released run (37345392679) baked 0.15.0 and passed the six original gate boots under both firmwares; `os-update` and `os-revert` fail with it because the 0.15.0 brain predates #563, so the next OS release must also release the control plane. | +| [ab-hang-and-major.md](ab-hang-and-major.md) — A slice of #486, after [hosted-ab-layout.md](hosted-ab-layout.md) and [host-agent-os-update.md](host-agent-os-update.md). Settles points 3 and 5 of `NEXT.md` # A/B OS image. **A slot that hangs:** real Hetzner cx23 VMs (Intel and AMD hosts) are QEMU `q35` with the ICH9 TCO watchdog; the image's generic kernel drives it (`iTCO_wdt`), and with PID 1 frozen the VM reset after 118 s. No image change; every boot checks systemd feeds it. Gaps: the reset comes after about twice the 60 s timeout, and a hang in GRUB or the initramfs is not covered. **A Debian major:** when the slot's major differs from the one the state partition records, in either direction, the initramfs tidies the `/etc` upper layer: the keep list (`etc-keep.list`) stays, the account files are merged, the pinned files are taken again from the slot when the remap stays, the rest goes to an attic; a failure never stops a boot. host-agent turns sshd back on once after a tidy-up when the drop-in names an account. The image's accounts are a generated `sysusers.d` file, with ids checked against `dev/os-lock/cloud-accounts.lock`. The `os-update` boot fakes a major, and every boot fails on an upper-layer file no rule covers. It must ship in a Debian 13 release before the first 14 release | done | diff --git a/docs/progress/ab-hang-and-major.md b/docs/progress/ab-hang-and-major.md new file mode 100644 index 00000000..28c919e8 --- /dev/null +++ b/docs/progress/ab-hang-and-major.md @@ -0,0 +1,97 @@ +# A slot that hangs, and a Debian major across the /etc overlay + +- **Status:** done +- **Date:** 2026-10-05 +- **Specs touched:** `docs/specs/BUILD.md`, `docs/specs/UPDATES.md`, `docs/specs/NEXT.md`, `docs/specs/DECISIONS.md`, `docs/architecture.md`, `docs/dev/hosted-boot-proof.md` + +A slice of #486. It settles points 3 and 5 of `NEXT.md` # A/B OS image, which [ab-os-update-design.md](ab-os-update-design.md) opened and [hosted-ab-layout.md](hosted-ab-layout.md) narrowed: whether a Hetzner VM has a watchdog, so a hung slot still reboots and reverts, and what happens to the `/etc` upper layer when a box moves to another Debian major. + +## What was done + +### Point 3: Hetzner Cloud has a hardware watchdog + +Checked on real Hetzner Cloud VMs, not in QEMU. Three `cx23` servers (the smallest box), stock Debian 13, two locations on AMD EPYC-Rome hosts and one on an Intel Skylake host, all booting legacy BIOS. Each lived a few minutes and was deleted with its SSH key; the API then listed no server, key, primary IP or snapshot made by the probe. The cost was about €0.035 (three started hours of a cx23). + +- **The device.** The VM is QEMU's `q35` machine. Its ICH9 chipset (`8086:2918`, the LPC bridge at `00:1f.0`) has a TCO watchdog timer. There is no `i6300esb` and no ACPI `WDAT` table. The same chipset is there on the Intel and the AMD hosts. +- **Stock Debian shows nothing.** Hetzner's Debian image runs the `-cloud` kernel, which has no `iTCO_wdt` module: no `/dev/watchdog`, `wdctl` finds no device, `WatchdogDevice=` is empty. +- **The moose kernel drives it.** With `linux-image-amd64`, the kernel moose ships, `lpc_ich` and `iTCO_wdt` load on their own: `Found a ICH9 TCO device (Version=2, TCOBASE=0x0660)`, `initialized. heartbeat=30 sec (nowayout=0)`, `/dev/watchdog0` with identity `iTCO_wdt`. With `RuntimeWatchdogSec=60s`, as the image sets, systemd logs `Using hardware watchdog 'iTCO_wdt', version 2, device /dev/watchdog0` and `Watchdog running with a hardware timeout of 1min.` The box stayed up 150 s with systemd feeding it. +- **It fires.** Opened once and never fed (`echo 1 > /dev/watchdog`, 30 s timeout, `watchdog did not stop!`): the Intel VM reset after about 50 to 65 s, the AMD VM after 59.5 s. The journal of the boot that reset ends with no shutdown. +- **A real hang is covered.** With systemd feeding it at 60 s, PID 1 was frozen with `ptrace` (`PTRACE_SEIZE` and `PTRACE_INTERRUPT`, state `t (tracing stop)`, no sysrq). The VM reset 118.3 s after the freeze, with no shutdown in the journal. +- **The boot lane has it too.** It is `q35` as well. A probe run (37358065628, `unseeded`, both firmwares) logged the same `iTCO_wdt` lines, so `BUILD.md`'s "none in the QEMU lane" was wrong. + +What changed: + +- **No image change.** The kernel already has the driver and `10-moose-watchdog.conf` already sets 60 s. Only its comment changed. +- **Every boot checks it** (`cloud-assertions.sh`, layout): `WatchdogDevice` set, `/sys/class/watchdog/watchdog0` there, `RuntimeWatchdogUSec=1min`, and systemd's `Using hardware watchdog` line in the boot's log. systemd logs that line before journald runs, so it is in the kernel log, not under `_PID=1`. +- **No hang proof in CI** (the maintainer's call). The Hetzner probe is the proof; the boot lane only checks the device is fed. +- **Two gaps, in `BUILD.md` # 1b # As built.** The reset comes after about twice the timeout (the TCO timer fires on its second expiry), so about 2 minutes on a box. A hang in GRUB or the initramfs is not covered, only a crash through `panic=10`; the watchdog is not started earlier because a long `e2fsck` or `systemd-growfs` of a big state partition would reset the box again and again. +- **The appliance fallback** is recorded for #564: `softdog` only where a machine has no hardware watchdog. It covers a hung PID 1, not a kernel locked up with interrupts off, and nothing before systemd. + +### Point 5: a Debian major tidies the /etc upper layer + +The maintainer chose option A (`DECISIONS.md` 2026-10-05). + +- **`state-setup` step 3a** (`dev/cloud/mkosi.extra/usr/lib/moose/state-setup`). It reads the slot's Debian major from `/usr/lib/os-release` and the box's from `/state/etc/.moose-debian-major`. When they differ, before `/etc` is mounted, it builds a new upper layer beside the old one: + - the files on `/usr/lib/moose/etc-keep.list` (and any `etc-keep.d/*.list`), copied with `cp -a --parents`, whiteouts included; + - `passwd`, `group`, `shadow`, `gshadow` merged by `awk`: the slot's lines in order, then the box's lines the slot lacks. In `shadow` and `gshadow` the box's password wins; in `group` and `gshadow` the members are joined. An id that differs between box and slot is logged; + - `daemon.json` from the slot when its `userns-remap` is the box's, `login.defs` from the slot when it still sets `SUB_UID_COUNT 0` and `SUB_GID_COUNT 0`; the box's copy otherwise, with a log line. + Then the old upper layer is renamed into `/state/etc/attic/