From 1c4de346b92ecb20c3399c7ab2f0be584fa22219 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 16:36:39 +0100 Subject: [PATCH 01/18] host-agent applies OS updates: install, switch in the window, trial boot, revert (#563) The update-target answer gains an OS list; host-agent installs the next release into the other slot with RAUC after checking the bundle's sha256 against the answer, switches inside the window, and marks the new slot good once the brain is healthy. A trial that fails reboots to the old slot; an image timer reboots a slot whose host-agent never started. Two new boots prove both paths under UEFI and BIOS. --- .github/workflows/ci-cloud-image.yml | 47 +- api/openapi.json | 101 ++- api/openapi.yaml | 73 +- cmd/brain/main.go | 5 + cmd/brain/osupdate.go | 88 +++ cmd/brain/osupdate_test.go | 70 ++ cmd/host-agent-real/main.go | 10 +- cmd/host-agent-real/osupdate.go | 108 +++ cmd/host-agent-real/osupdate_appliance.go | 10 + cmd/host-agent-real/osupdate_hosted.go | 10 + cmd/host-agent-real/updatetarget.go | 19 +- cmd/host-agent-real/updatetargetreport.go | 53 ++ cmd/host-agent/updatetarget.go | 35 + dev/cloud/cloud-assertions.sh | 259 ++++++- dev/cloud/mkosi.extra/etc/rauc/system.conf | 8 +- .../etc/systemd/system/moose-os-trial.service | 10 + .../etc/systemd/system/moose-os-trial.timer | 13 + .../mkosi.extra/usr/lib/moose/os-trial-check | 20 + dev/cloud/mkosi.postinst.chroot | 7 + dev/cloud/run-cloud-tests.sh | 78 ++- dev/cloud/test/bootstrap.sh | 13 +- dev/cloud/test/build-os-test-bundle.sh | 98 +++ dev/cloud/test/moose-cloud-assertions.service | 1 + internal/api/system.go | 8 +- internal/api/systemupdate.go | 50 ++ internal/hostagent/agent.go | 17 + internal/hostagent/jobs.go | 54 ++ internal/hostagent/osupdate/osupdate.go | 638 ++++++++++++++++++ internal/hostagent/osupdate/osupdate_test.go | 351 ++++++++++ internal/hostagent/osupdate/rauc.go | 142 ++++ internal/hostagent/updatetarget/http.go | 16 + internal/hostagent/updatetarget/loop.go | 136 +++- internal/hostagent/updatetarget/os.go | 170 +++++ internal/hostagent/updatetarget/os_test.go | 214 ++++++ internal/hostagent/updatetarget/state.go | 22 + internal/hostagent/updatetarget/target.go | 5 + internal/notify/notify.go | 48 ++ internal/notify/notify_test.go | 18 + internal/protocol/host.go | 105 ++- internal/store/notify.go | 13 + internal/store/notify_test.go | 13 + web-ui/src/generated/openapi.ts | 27 +- 42 files changed, 3139 insertions(+), 44 deletions(-) create mode 100644 cmd/brain/osupdate.go create mode 100644 cmd/brain/osupdate_test.go create mode 100644 cmd/host-agent-real/osupdate.go create mode 100644 cmd/host-agent-real/osupdate_appliance.go create mode 100644 cmd/host-agent-real/osupdate_hosted.go create mode 100644 dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.service create mode 100644 dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.timer create mode 100755 dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check create mode 100755 dev/cloud/test/build-os-test-bundle.sh create mode 100644 internal/hostagent/osupdate/osupdate.go create mode 100644 internal/hostagent/osupdate/osupdate_test.go create mode 100644 internal/hostagent/osupdate/rauc.go create mode 100644 internal/hostagent/updatetarget/os.go create mode 100644 internal/hostagent/updatetarget/os_test.go diff --git a/.github/workflows/ci-cloud-image.yml b/.github/workflows/ci-cloud-image.yml index 5fe6ac17..c768811c 100644 --- a/.github/workflows/ci-cloud-image.yml +++ b/.github/workflows/ci-cloud-image.yml @@ -560,7 +560,7 @@ jobs: BOOTS_INPUT: ${{ inputs.boots }} run: | set -euo pipefail - full="unseeded seeded access update ssh remap" + full="unseeded seeded access update ssh remap os-update os-revert" if [ "$EVENT" = "pull_request" ]; then # The compare API needs only `contents: read`, which this job has. The # PR files API would need `pull-requests: read`, and asking for it @@ -587,11 +587,24 @@ jobs: for b in $list; do case "$b" in unseeded|seeded|frozen) case " $chain " in *" $b "*) ;; *) chain="${chain:+$chain }$b" ;; esac ;; - access|update|ssh|remap) case " ${groups[*]-} " in *" $b "*) ;; *) groups+=("$b") ;; esac ;; + access|update|ssh|remap|os-update|os-revert) case " ${groups[*]-} " in *" $b "*) ;; *) groups+=("$b") ;; esac ;; bios) echo "::error::'bios' is no longer a boot: every boot runs under both firmwares"; exit 1 ;; - *) echo "::error::unknown boot '$b' (known: unseeded seeded frozen access update ssh remap)"; exit 1 ;; + *) echo "::error::unknown boot '$b' (known: unseeded seeded frozen access update ssh remap os-update os-revert)"; exit 1 ;; esac done + # A run that publishes the OS bakes only the release root into the + # boot-proof image, so the throwaway-signed test bundle cannot install + # (#563). os-update then runs its refusal half only, and os-revert, + # which needs an install, is left out. + if [ "$SHOULD_PUBLISH_OS" = true ]; then + kept=(); for g in "${groups[@]}"; do [ "$g" = os-revert ] || kept+=("$g"); done; groups=("${kept[@]}") + echo "an OS release run: os-update runs its refusal half only, os-revert is left out" + fi + # The OS update boots need the test bundle; the build makes it only + # when one of them runs (about 1.5 min). + os_test=false + case " ${groups[*]} " in *" os-update "*|*" os-revert "*) os_test=true ;; esac + echo "os_test=${os_test}" >> "$GITHUB_OUTPUT" if [ -n "$chain" ]; then groups=("$chain" "${groups[@]}"); fi [ "${#groups[@]}" -gt 0 ] || { echo "::error::the boot list '${list}' names no boot"; exit 1; } matrix="$(printf '%s\n' "${groups[@]}" | jq -Rnc '[inputs] as $g | {include: [$g[] as $b | ("uefi", "bios") as $f | {boots: $b, firmware: $f}]}')" @@ -673,6 +686,26 @@ jobs: qemu-img convert -f raw -O qcow2 .dev/cloud-boot/moose-cloud.raw .dev/cloud-boot/moose-cloud-boot.qcow2 ls -l .dev/cloud-boot/moose-cloud-boot.qcow2 + # The test-only OS bundle the os-update and os-revert boots install + # (#563): the boot-proof slot repacked fast (gzip level 1) with a + # host-agent one patch release up, signed with the throwaway key. Only + # when one of those boots runs. + - name: Build the OS test bundle + if: ${{ steps.boots.outputs.os_test == 'true' }} + env: + GO: ${{ steps.go.outputs.go-bin }} + run: sudo -E ./dev/cloud/test/build-os-test-bundle.sh .dev/cloud-boot/moose-cloud.raw .dev/cloud-boot/os-test + - name: Upload the OS test bundle + if: ${{ steps.boots.outputs.os_test == 'true' }} + uses: actions/upload-artifact@v4 + with: + overwrite: true + name: cloud-image-os-test-bundle + path: .dev/cloud-boot/os-test + compression-level: 0 + retention-days: 3 + if-no-files-found: error + # The squashfs inside is already xz, so zipping it again only costs time. # Three days is enough to re-run a failed boot or publish job. - name: Upload the boot-proof image @@ -799,6 +832,12 @@ jobs: with: name: cloud-image-boot-qcow2 path: .dev/cloud-boot + - name: Download the OS test bundle + if: ${{ matrix.boots == 'os-update' || matrix.boots == 'os-revert' }} + uses: actions/download-artifact@v4 + with: + name: cloud-image-os-test-bundle + path: .dev/cloud-boot/os-test # MOOSE_CLOUD_QCOW2 makes the lane boot the downloaded image and build # nothing. The boot list reaches the script through an env var, never @@ -808,6 +847,8 @@ jobs: MOOSE_CLOUD_QCOW2: ${{ github.workspace }}/.dev/cloud-boot/moose-cloud-boot.qcow2 MOOSE_CLOUD_BOOTS: ${{ matrix.boots }} MOOSE_CLOUD_FIRMWARES: ${{ matrix.firmware }} + MOOSE_CLOUD_OS_BUNDLE_DIR: ${{ github.workspace }}/.dev/cloud-boot/os-test + MOOSE_CLOUD_OS_REFUSE_ONLY: ${{ env.SHOULD_PUBLISH_OS }} GO: ${{ steps.go.outputs.go-bin }} run: make test-cloud-qemu diff --git a/api/openapi.json b/api/openapi.json index 25b25f0c..dc57c0b3 100644 --- a/api/openapi.json +++ b/api/openapi.json @@ -2888,6 +2888,96 @@ ], "type": "object" }, + "OSOutcomeDTO": { + "additionalProperties": false, + "properties": { + "at": { + "type": "string" + }, + "from": { + "type": "string" + }, + "id": { + "type": "string" + }, + "outcome": { + "enum": [ + "good", + "reverted" + ], + "type": "string" + }, + "version": { + "type": "string" + } + }, + "required": [ + "id", + "outcome", + "version", + "at" + ], + "type": "object" + }, + "OSReleaseDTO": { + "additionalProperties": false, + "properties": { + "bundle_sha256": { + "type": "string" + }, + "bundle_url": { + "type": "string" + }, + "version": { + "type": "string" + } + }, + "required": [ + "version", + "bundle_url", + "bundle_sha256" + ], + "type": "object" + }, + "OSUpdateDTO": { + "additionalProperties": false, + "properties": { + "detail": { + "type": "string" + }, + "last": { + "$ref": "#/components/schemas/OSOutcomeDTO" + }, + "running": { + "type": "string" + }, + "slot": { + "type": "string" + }, + "state": { + "enum": [ + "unsupported", + "none", + "refused", + "current", + "installing", + "installed", + "waiting", + "rebooting", + "held", + "failed" + ], + "type": "string" + }, + "target": { + "$ref": "#/components/schemas/OSReleaseDTO" + } + }, + "required": [ + "state" + ], + "type": "object" + }, "Parse-custom-overlayRequest": { "additionalProperties": false, "properties": { @@ -3555,6 +3645,12 @@ "host_agent_version": { "type": "string" }, + "os_slot": { + "type": "string" + }, + "os_version": { + "type": "string" + }, "ui_image": { "type": "string" }, @@ -3656,6 +3752,9 @@ "from": { "type": "string" }, + "os": { + "$ref": "#/components/schemas/OSUpdateDTO" + }, "profile": { "type": "string" }, @@ -6125,7 +6224,7 @@ "description": "Error" } }, - "summary": "What this box is running: brain version and commit, host-agent version, UI image" + "summary": "What this box is running: brain version and commit, host-agent version, UI image, OS version and slot" } }, "/api/v1/users": { diff --git a/api/openapi.yaml b/api/openapi.yaml index 9d288981..9aeb61c4 100644 --- a/api/openapi.yaml +++ b/api/openapi.yaml @@ -2048,6 +2048,71 @@ components: - summary - read type: object + OSOutcomeDTO: + additionalProperties: false + properties: + at: + type: string + from: + type: string + id: + type: string + outcome: + enum: + - good + - reverted + type: string + version: + type: string + required: + - id + - outcome + - version + - at + type: object + OSReleaseDTO: + additionalProperties: false + properties: + bundle_sha256: + type: string + bundle_url: + type: string + version: + type: string + required: + - version + - bundle_url + - bundle_sha256 + type: object + OSUpdateDTO: + additionalProperties: false + properties: + detail: + type: string + last: + $ref: "#/components/schemas/OSOutcomeDTO" + running: + type: string + slot: + type: string + state: + enum: + - unsupported + - none + - refused + - current + - installing + - installed + - waiting + - rebooting + - held + - failed + type: string + target: + $ref: "#/components/schemas/OSReleaseDTO" + required: + - state + type: object Parse-custom-overlayRequest: additionalProperties: false properties: @@ -2523,6 +2588,10 @@ components: type: string host_agent_version: type: string + os_slot: + type: string + os_version: + type: string ui_image: type: string version: @@ -2595,6 +2664,8 @@ components: type: string from: type: string + os: + $ref: "#/components/schemas/OSUpdateDTO" profile: type: string running: @@ -4109,7 +4180,7 @@ paths: schema: $ref: "#/components/schemas/ErrorModel" description: Error - summary: "What this box is running: brain version and commit, host-agent version, UI image" + summary: "What this box is running: brain version and commit, host-agent version, UI image, OS version and slot" /api/v1/users: get: operationId: list-users diff --git a/cmd/brain/main.go b/cmd/brain/main.go index f6875fea..0bd95f33 100644 --- a/cmd/brain/main.go +++ b/cmd/brain/main.go @@ -81,6 +81,7 @@ func main() { if err := os.MkdirAll(cfg.stateDir, 0o755); err != nil { fatal("create state dir", "err", err) } + writeFloorFile(cfg.stateDir) st, err := store.Open(filepath.Join(cfg.stateDir, "moose.db")) if err != nil { @@ -295,6 +296,10 @@ func main() { checkAgentVersion(pollCtx, host, healthMgr, auditor, notifier, bus) go versionCheckPollLoop(pollCtx, host, healthMgr, auditor, notifier, bus, cfg.healthPollPeriod) + // Stream A outcomes (#563): one admin notification per OS update, read + // off host-agent's update-target report on the same cadence. + go osOutcomeLoop(pollCtx, host, st, notifier, cfg.healthPollPeriod) + // Locus-C brain-DB integrity check (HEALTH.md # Detector catalog): PRAGMA // integrity_check at boot + every 6h, reconciling brain-db-corrupt. Runs // entirely on its own goroutine — the boot run is inside the loop, never diff --git a/cmd/brain/osupdate.go b/cmd/brain/osupdate.go new file mode 100644 index 00000000..eaf60388 --- /dev/null +++ b/cmd/brain/osupdate.go @@ -0,0 +1,88 @@ +package main + +import ( + "context" + "log/slog" + "os" + "path/filepath" + "time" + + "github.com/onmoose/os/internal/notify" + "github.com/onmoose/os/internal/protocol" +) + +// This file is the brain's part of stream A (UPDATES.md # 1, #563): it tells +// host-agent which OS floor this control plane needs, and it tells admins what +// an OS update did. host-agent does the update itself. + +// floorFileName is where the brain writes minimumAgentVersion, in its state +// directory. host-agent reads it to refuse an OS target below the running +// control plane's floor (UPDATES.md # 1): booting a host-agent the brain +// refuses to work with would leave a box nobody can reach. +const floorFileName = "minimum-host-agent" + +// writeFloorFile writes the floor. A failure is a warning: host-agent then has +// no floor to check, which is the state of a box before #563. +func writeFloorFile(stateDir string) { + path := filepath.Join(stateDir, floorFileName) + if err := os.WriteFile(path, []byte(minimumAgentVersion+"\n"), 0o644); err != nil { + slog.Warn("could not write the control-plane floor for host-agent; it will not check OS targets against it", "err", err, "dir", stateDir) + } +} + +// osOutcomeReader is the slice of the host client the outcome check needs. +type osOutcomeReader interface { + SystemUpdateTarget(ctx context.Context) (protocol.UpdateTarget, error) +} + +// notificationLookup says whether a notification was ever raised for a key. +type notificationLookup interface { + HasNotification(dedupKey string) (bool, error) +} + +// osOutcomeNotifier is the slice of the notifier the check needs. +type osOutcomeNotifier interface { + OSUpdateOutcome(outcomeID, outcome, version, from string) +} + +// osOutcomeMaxAge bounds which outcomes still get a notification. host-agent +// keeps the last outcome for good, and notifications are pruned after 90 days, +// so without a bound an old outcome would be announced again after its row was +// pruned. A box whose brain was down for a week after an update loses that one +// notification; the outcome is still on the update-target read. +const osOutcomeMaxAge = 7 * 24 * time.Hour + +// checkOSOutcome raises one admin notification per OS update outcome. +func checkOSOutcome(ctx context.Context, host osOutcomeReader, seen notificationLookup, n osOutcomeNotifier, now time.Time) { + t, err := host.SystemUpdateTarget(ctx) + if err != nil || t.OS == nil || t.OS.Last == nil || t.OS.Last.ID == "" { + return + } + last := t.OS.Last + at, err := time.Parse(time.RFC3339, last.At) + if err != nil || now.Sub(at) > osOutcomeMaxAge { + return + } + done, err := seen.HasNotification(notify.OSUpdateDedupKey(last.ID)) + if err != nil { + slog.Warn("os update: could not check for an earlier notification; trying on the next poll", "err", err) + return + } + if done { + return + } + n.OSUpdateOutcome(last.ID, last.Outcome, last.Version, last.From) + slog.Info("os update: notified admins of the outcome", "os", last.Version) +} + +// osOutcomeLoop runs checkOSOutcome on the health-poll cadence. +func osOutcomeLoop(ctx context.Context, host osOutcomeReader, seen notificationLookup, n osOutcomeNotifier, interval time.Duration) { + for { + checkOSOutcome(ctx, host, seen, n, time.Now()) + select { + case <-ctx.Done(): + return + case <-time.After(interval): + } + } +} diff --git a/cmd/brain/osupdate_test.go b/cmd/brain/osupdate_test.go new file mode 100644 index 00000000..b5d73208 --- /dev/null +++ b/cmd/brain/osupdate_test.go @@ -0,0 +1,70 @@ +package main + +import ( + "context" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/onmoose/os/internal/protocol" +) + +type fakeOutcomeHost struct{ t protocol.UpdateTarget } + +func (f fakeOutcomeHost) SystemUpdateTarget(context.Context) (protocol.UpdateTarget, error) { + return f.t, nil +} + +type fakeSeen map[string]bool + +func (f fakeSeen) HasNotification(k string) (bool, error) { return f[k], nil } + +type fakeOutcomeNotifier struct{ raised []string } + +func (f *fakeOutcomeNotifier) OSUpdateOutcome(id, outcome, version, from string) { + f.raised = append(f.raised, id+" "+outcome+" "+version+" "+from) +} + +func TestCheckOSOutcome(t *testing.T) { + now := time.Date(2026, 10, 3, 9, 0, 0, 0, time.UTC) + last := &protocol.OSOutcome{ID: "os-0.15.1-1", Outcome: "reverted", Version: "0.15.1", From: "0.15.0", At: "2026-10-03T03:20:00Z"} + host := fakeOutcomeHost{protocol.UpdateTarget{OS: &protocol.OSUpdate{State: "held", Last: last}}} + + n := &fakeOutcomeNotifier{} + checkOSOutcome(context.Background(), host, fakeSeen{}, n, now) + if len(n.raised) != 1 || n.raised[0] != "os-0.15.1-1 reverted 0.15.1 0.15.0" { + t.Fatalf("want one notification, got %v", n.raised) + } + + n = &fakeOutcomeNotifier{} + checkOSOutcome(context.Background(), host, fakeSeen{"os-update:os-0.15.1-1": true}, n, now) + if len(n.raised) != 0 { + t.Fatalf("an outcome already notified was raised again: %v", n.raised) + } + + n = &fakeOutcomeNotifier{} + checkOSOutcome(context.Background(), host, fakeSeen{}, n, now.Add(8*24*time.Hour)) + if len(n.raised) != 0 { + t.Fatalf("an old outcome was raised: %v", n.raised) + } + + n = &fakeOutcomeNotifier{} + checkOSOutcome(context.Background(), fakeOutcomeHost{}, fakeSeen{}, n, now) + if len(n.raised) != 0 { + t.Fatalf("no OS part must raise nothing: %v", n.raised) + } +} + +func TestWriteFloorFile(t *testing.T) { + dir := t.TempDir() + writeFloorFile(dir) + b, err := os.ReadFile(filepath.Join(dir, floorFileName)) + if err != nil { + t.Fatal(err) + } + if strings.TrimSpace(string(b)) != minimumAgentVersion { + t.Fatalf("got %q", b) + } +} diff --git a/cmd/host-agent-real/main.go b/cmd/host-agent-real/main.go index fe6286ec..e639bb76 100644 --- a/cmd/host-agent-real/main.go +++ b/cmd/host-agent-real/main.go @@ -138,7 +138,15 @@ func main() { // prompt differ, and both come from the build-tagged updateTargetSource. // Started last for the same reason the poll is: nothing about booting waits // on it. - stopTarget := startUpdateTarget(brainCfg, a, poller) + // Stream A (#563). Boot decides first whether this boot is an OS trial, + // whatever the update loop does below: a trial has to be decided even on + // a box whose update target is unusable. + osApp := osUpdateApplier(osUpdateDeps{agent: a, brainCfg: brainCfg}) + if osApp != nil { + osApp.Boot(context.Background()) + a.OS = osApp + } + stopTarget := startUpdateTarget(brainCfg, a, poller, osApp) defer stopTarget() slog.Info("host-agent-real listening", "sock", sockPath) diff --git a/cmd/host-agent-real/osupdate.go b/cmd/host-agent-real/osupdate.go new file mode 100644 index 00000000..2d19b94b --- /dev/null +++ b/cmd/host-agent-real/osupdate.go @@ -0,0 +1,108 @@ +package main + +import ( + "context" + "errors" + "fmt" + "log/slog" + "os" + "path/filepath" + "time" + + "github.com/onmoose/os/internal/hostagent" + "github.com/onmoose/os/internal/hostagent/brainlaunch" + "github.com/onmoose/os/internal/hostagent/cpupdate" + "github.com/onmoose/os/internal/hostagent/osupdate" + "github.com/onmoose/os/internal/hostagent/updatetarget" + "github.com/onmoose/os/internal/version" +) + +// Stream A's settings (#563). Not build-tagged, so `go vet` and `go test` +// see them; only the call that builds the applier is (osupdate_hosted.go). +const ( + // envOSURLPrefix is the expected start of every bundle URL. The default + // is the GitHub Releases path of this repo (updatetarget.DefaultOSURLPrefix). + // The boot lane points it at a server inside the guest. + envOSURLPrefix = "MOOSE_UPDATE_OS_URL_PREFIX" + // envOSTrialTimeout bounds a trial boot's wait for the brain, as a Go + // duration. The default is osupdate.DefaultTrialTimeout. + envOSTrialTimeout = "MOOSE_OS_TRIAL_TIMEOUT" +) + +// osURLPrefix is the bundle URL prefix in force. +func osURLPrefix() string { + if v := os.Getenv(envOSURLPrefix); v != "" { + return v + } + return updatetarget.DefaultOSURLPrefix +} + +// osTrialTimeout is the trial bound in force. An unreadable value warns and +// falls back: a wrong bound only makes a trial shorter or longer. +func osTrialTimeout() time.Duration { + v := os.Getenv(envOSTrialTimeout) + if v == "" { + return osupdate.DefaultTrialTimeout + } + d, err := time.ParseDuration(v) + if err != nil || d <= 0 { + slog.Warn("os update: the trial timeout is not readable; using the default", "err", err, "window", v) + return osupdate.DefaultTrialTimeout + } + return d +} + +// osUpdateDeps is what the applier is built from. +type osUpdateDeps struct { + agent *hostagent.Agent + brainCfg brainlaunch.Config +} + +// agentJobs adapts the agent's job lock to the applier's Jobs seam, and maps +// the lock's refusal to osupdate.ErrBusy. +type agentJobs struct{ a *hostagent.Agent } + +func (j agentJobs) StartJob(kind string, d time.Duration, fn func(ctx context.Context) error) (string, error) { + job, err := j.a.StartJob(kind, d, fn) + if hostagent.IsJobRunning(err) { + return "", fmt.Errorf("%w: %v", osupdate.ErrBusy, err) + } + if err != nil { + return "", err + } + return job.ID, nil +} + +// newOSApplier builds the applier. The health check is the brain's /healthz on +// its container address, the same probe the control-plane update uses +// (cpupdate), retried until the trial's deadline: on a fresh boot the brain +// container may not have an address yet. +func newOSApplier(d osUpdateDeps) *osupdate.Applier { + docker := cpupdate.NewCLIDocker() + prober := cpupdate.HTTPProber{} + healthy := func(ctx context.Context) error { + var last error + for { + ip, err := docker.ContainerIP(ctx, d.brainCfg.ContainerName) + if err == nil { + if err = prober.WaitServing(ctx, fmt.Sprintf("http://%s:8080/healthz", ip)); err == nil { + return nil + } + } + last = err + select { + case <-ctx.Done(): + return errors.Join(ctx.Err(), last) + case <-time.After(2 * time.Second): + } + } + } + return &osupdate.Applier{ + RAUC: osupdate.CLIRAUC{}, + Jobs: agentJobs{d.agent}, + Version: version.Version, + FloorFile: filepath.Join(d.brainCfg.StateDir, "minimum-host-agent"), + Healthy: healthy, + TrialTimeout: osTrialTimeout(), + } +} diff --git a/cmd/host-agent-real/osupdate_appliance.go b/cmd/host-agent-real/osupdate_appliance.go new file mode 100644 index 00000000..2b52be9b --- /dev/null +++ b/cmd/host-agent-real/osupdate_appliance.go @@ -0,0 +1,10 @@ +//go:build !hosted + +package main + +import "github.com/onmoose/os/internal/hostagent/osupdate" + +// osUpdateApplier is nil on the appliance build: its image is not in the A/B +// layout yet (#564), so there is no other slot to install into. The update +// loop then reports stream A as unsupported. +func osUpdateApplier(osUpdateDeps) *osupdate.Applier { return nil } diff --git a/cmd/host-agent-real/osupdate_hosted.go b/cmd/host-agent-real/osupdate_hosted.go new file mode 100644 index 00000000..a54e25ed --- /dev/null +++ b/cmd/host-agent-real/osupdate_hosted.go @@ -0,0 +1,10 @@ +//go:build hosted + +package main + +import "github.com/onmoose/os/internal/hostagent/osupdate" + +// osUpdateApplier builds stream A's applier on the hosted build, the one +// profile whose image is in the A/B layout today (BUILD.md # 1b). The +// appliance gets it with #564. +func osUpdateApplier(r osUpdateDeps) *osupdate.Applier { return newOSApplier(r) } diff --git a/cmd/host-agent-real/updatetarget.go b/cmd/host-agent-real/updatetarget.go index e0d246e6..cd7534bf 100644 --- a/cmd/host-agent-real/updatetarget.go +++ b/cmd/host-agent-real/updatetarget.go @@ -7,6 +7,7 @@ import ( "github.com/onmoose/os/internal/hostagent" "github.com/onmoose/os/internal/hostagent/brainlaunch" + "github.com/onmoose/os/internal/hostagent/osupdate" "github.com/onmoose/os/internal/hostagent/relmanifest" "github.com/onmoose/os/internal/hostagent/updatetarget" ) @@ -20,7 +21,7 @@ import ( // validation, the window, the apply, the failure handling — is one loop shared by // both profiles. A second copy of any of it, per profile, is what UPDATES.md # 8 // means by "we only build it once". -func startUpdateTarget(brainCfg brainlaunch.Config, a *hostagent.Agent, poller *relmanifest.Poller) func() { +func startUpdateTarget(brainCfg brainlaunch.Config, a *hostagent.Agent, poller *relmanifest.Poller, osApp *osupdate.Applier) func() { window, windowFrom := updateWindow() // What the box is running, read fresh on every socket read. Built here // whatever happens below: a box that will not update itself still has to be @@ -53,6 +54,7 @@ func startUpdateTarget(brainCfg brainlaunch.Config, a *hostagent.Agent, poller * // box could not build. autoApply is left false on purpose: nothing // on this box will apply anything, whatever profile it is. profile: buildProfile, + os: osReporter(osApp), } return func() {} } @@ -64,10 +66,16 @@ func startUpdateTarget(brainCfg brainlaunch.Config, a *hostagent.Agent, poller * Repos: repositories(), // The box's own window. An answer that names one wins over it, so this // is the fallback the loop uses until a source has an opinion. - Window: window, - WindowFrom: windowFrom, - AutoApply: src.AutoApply, - Profile: src.Profile, + Window: window, + WindowFrom: windowFrom, + AutoApply: src.AutoApply, + Profile: src.Profile, + OSURLPrefix: osURLPrefix(), + } + // Set only when there is an applier: a nil *Applier in the interface + // field would not read as nil. + if osApp != nil { + loop.OS = osApp } // The socket read (GET /v1/system/update-target) is served from this loop's // snapshot. Set before Run, so a read that arrives during the first tick @@ -80,6 +88,7 @@ func startUpdateTarget(brainCfg brainlaunch.Config, a *hostagent.Agent, poller * windowFrom: windowFrom, profile: src.Profile, autoApply: src.AutoApply, + os: osReporter(osApp), } ctx, cancel := context.WithCancel(context.Background()) go loop.Run(ctx) diff --git a/cmd/host-agent-real/updatetargetreport.go b/cmd/host-agent-real/updatetargetreport.go index 785d6520..93526704 100644 --- a/cmd/host-agent-real/updatetargetreport.go +++ b/cmd/host-agent-real/updatetargetreport.go @@ -3,6 +3,7 @@ package main import ( "time" + "github.com/onmoose/os/internal/hostagent/osupdate" "github.com/onmoose/os/internal/hostagent/updatetarget" "github.com/onmoose/os/internal/protocol" ) @@ -39,6 +40,57 @@ type updateTargetReport struct { // build-tagged source that built the loop. profile string autoApply bool + // os reads stream A's facts that the loop does not hold: the running OS + // release and slot, and the last outcome, which outlive a reboot. Nil on + // a box that cannot update its OS. + os osFacts +} + +// osFacts is the slice of osupdate.Applier the report reads. +type osFacts interface { + Running() (version, slot string) + Last() *protocol.OSOutcome + Peek(rel updatetarget.OSRelease) (state, detail string, ok bool) +} + +// osReporter returns the applier as osFacts, or a nil interface for a nil +// applier, so a box with no applier reports no OS facts rather than calling a +// nil pointer. +func osReporter(a *osupdate.Applier) osFacts { + if a == nil { + return nil + } + return a +} + +// readOS assembles stream A's part of the report. +func (r updateTargetReport) readOS() *protocol.OSUpdate { + if r.os == nil { + return &protocol.OSUpdate{State: protocol.OSUpdateUnsupported, Detail: "this host-agent cannot update the OS"} + } + out := &protocol.OSUpdate{State: protocol.OSUpdateNone, Last: r.os.Last()} + out.Running, out.Slot = r.os.Running() + if r.loop == nil { + out.Detail = "this box has no update loop, so it will not update its OS" + return out + } + s := r.loop.Snapshot().OS + if s.State != "" { + out.State, out.Detail = s.State, s.Detail + } + if s.Target != nil { + // The loop decides once per tick; the job it started may have ended + // since. Only the states a job moves between are refreshed: current, + // refused, held and the rest are the loop's to say. + switch s.State { + case protocol.OSUpdateInstalling, protocol.OSUpdateInstalled, protocol.OSUpdateWaiting, protocol.OSUpdateFailed: + if st, d, ok := r.os.Peek(*s.Target); ok { + out.State, out.Detail = st, d + } + } + out.Target = &protocol.OSRelease{Version: s.Target.Version, BundleURL: s.Target.BundleURL, BundleSHA256: s.Target.BundleSHA256} + } + return out } // Read assembles the report. It never fails: every way this can go wrong is a @@ -46,6 +98,7 @@ type updateTargetReport struct { // say so in a form the dashboard can render. func (r updateTargetReport) Read() protocol.UpdateTarget { out := protocol.UpdateTarget{ + OS: r.readOS(), From: r.from, Window: r.window.String(), WindowFrom: r.windowFrom, diff --git a/cmd/host-agent/updatetarget.go b/cmd/host-agent/updatetarget.go index 7cf33429..4f5b864d 100644 --- a/cmd/host-agent/updatetarget.go +++ b/cmd/host-agent/updatetarget.go @@ -45,6 +45,8 @@ func fakeUpdateTarget() protocol.UpdateTarget { PublishedAt: "2026-09-01T10:00:00Z", } + out.OS = fakeOSUpdate() + want := os.Getenv("MOOSE_FAKE_UPDATE_TARGET") switch want { case "", protocol.UpdateTargetNone: @@ -71,3 +73,36 @@ func fakeUpdateTarget() protocol.UpdateTarget { } return out } + +// fakeOSUpdate is stream A's canned part (#563). The fake has no slots, so the +// honest default is "unsupported". MOOSE_FAKE_OS_UPDATE picks another state, +// so a dashboard can be built against the shapes a hosted box sends. +func fakeOSUpdate() *protocol.OSUpdate { + want := os.Getenv("MOOSE_FAKE_OS_UPDATE") + if want == "" || want == protocol.OSUpdateUnsupported { + return &protocol.OSUpdate{State: protocol.OSUpdateUnsupported, Detail: "the fake host-agent has no OS slots"} + } + out := &protocol.OSUpdate{ + State: want, + Running: "0.15.0", + Slot: "A", + Target: &protocol.OSRelease{ + Version: "0.15.1", + BundleURL: "https://github.com/onmoose/os/releases/download/v0.15.1/moose-v0.15.1-amd64.raucb", + BundleSHA256: "5555555555555555555555555555555555555555555555555555555555555555", + }, + } + switch want { + case protocol.OSUpdateCurrent: + out.Running = "0.15.1" + out.Last = &protocol.OSOutcome{ID: "os-0.15.1-1790000000", Outcome: protocol.OSOutcomeGood, Version: "0.15.1", From: "0.15.0", At: "2026-10-03T03:12:00Z"} + case protocol.OSUpdateHeld: + out.Detail = "this release was already tried tonight; the next window tries again" + out.Last = &protocol.OSOutcome{ID: "os-0.15.1-1790000000", Outcome: protocol.OSOutcomeReverted, Version: "0.15.1", From: "0.15.0", At: "2026-10-03T03:20:00Z"} + case protocol.OSUpdateFailed, protocol.OSUpdateRefused: + out.Detail = "the bundle's sha256 is 0000..., not the 5555... the update target names; refusing it" + case protocol.OSUpdateNone: + out.Target = nil + } + return out +} diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index 3237c592..aff5f2a1 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -170,6 +170,36 @@ ok() { echo "cloud-assertions: starting boot-proof checks (mode=${MODE})" +# The slot GRUB booted (#563). Every boot but the OS update boots runs on slot A; +# those two move the box between slots, one stage per boot, and keep the stage +# on the state partition. +BOOTED="$(sed -n 's/.*rauc\.slot=\([AB]\).*/\1/p' /proc/cmdline)" +OS_STATE_DIR=/var/lib/moose/test/os +os_stage() { cat "$OS_STATE_DIR/stage" 2>/dev/null || echo 1; } +set_os_stage() { mkdir -p "$OS_STATE_DIR" && echo "$1" > "$OS_STATE_DIR/stage" && sync; } +case "$MODE" in +os-update|os-revert) ;; +*) [ "$BOOTED" = A ] || fail "layout: booted slot '$BOOTED', want A (only the OS update boots leave slot A)" ;; +esac + +# The broken slot of the os-revert boot (#563). host-agent is kept from starting +# on it (a drop-in the first stage planted), which is the case where nothing but +# the image's own safety net can take the box back. So this stage checks only +# that, and waits: every other check of this script would fail on a box with no +# host-agent, which is the point of the scenario. moose-os-trial.timer (3 min in +# this image) must reboot the box; the next stage runs on slot A. +if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then + [ "$BOOTED" = B ] || fail "os-revert: stage 2 should boot slot B, booted '$BOOTED' (did GRUB skip the new slot?)" + [ -e /var/lib/moose/os-update/trial-B ] || fail "os-revert: no trial marker for slot B on its trial boot" + for _ in $(seq 1 20); do systemctl is-active -q host-agent.service && break; sleep 1; done + systemctl is-active -q host-agent.service && fail "os-revert: host-agent runs on the slot it was meant to be kept off" + systemctl list-timers --all --no-pager 2>/dev/null | grep -q moose-os-trial.timer || fail "os-revert: moose-os-trial.timer is not scheduled on the trial boot" + echo "cloud-assertions: os-revert: on slot B, host-agent cannot start, trial marker present; waiting for the safety net to reboot the box" + set_os_stage 3 + sleep 900 + fail "os-revert: the safety net never rebooted the box off the broken slot" +fi + # --- 1. no control-plane unit has failed. # NOTE: we deliberately do NOT gate on `systemctl is-system-running == running`: # this script runs as a boot-transaction unit (WantedBy=multi-user.target), so @@ -207,8 +237,9 @@ slot_b_bytes="$(lsblk -bnr -o PARTN,SIZE "$root_disk" | awk '$1==4{print $2}')" [ "$esp_bytes" = 134217728 ] || layout_fail "the ESP is $esp_bytes bytes, want 128 MiB" [ "$slot_a_bytes" = 1073741824 ] && [ "$slot_b_bytes" = 1073741824 ] \ || layout_fail "slots are A=$slot_a_bytes B=$slot_b_bytes bytes, want 1 GiB each" -[ "$root_src" = /dev/disk/by-partuuid/20202020-2020-4020-8020-202020202020 ] || [ "$(lsblk -no PARTUUID "$root_src")" = 20202020-2020-4020-8020-202020202020 ] \ - || layout_fail "/ is $root_src, not slot A" +if [ "$BOOTED" = B ]; then booted_uuid=21212121-2121-4121-8121-212121212121; booted_idx=1; else booted_uuid=20202020-2020-4020-8020-202020202020; booted_idx=0; fi +[ "$root_src" = "/dev/disk/by-partuuid/$booted_uuid" ] || [ "$(lsblk -no PARTUUID "$root_src")" = "$booted_uuid" ] \ + || layout_fail "/ is $root_src, not slot $BOOTED" # Grown: the state partition reaches the end of the disk, and its ext4 fills it. disk_bytes="$(lsblk -bdn -o SIZE "$root_disk")" state_dev="$(findmnt -no SOURCE /state 2>/dev/null)" @@ -239,10 +270,10 @@ esp_used="$(df -B1 --output=used /efi | tail -n1 | tr -d ' ')" state_used="$(df -B1 --output=used /state | tail -n1 | tr -d ' ')" echo "cloud-assertions: layout: measured: squashfs ${sq_bytes} bytes ($(( sq_bytes * 1000 / slot_a_bytes / 10 )).$(( sq_bytes * 1000 / slot_a_bytes % 10 ))% of the slot), ESP used ${esp_used} of ${esp_bytes} bytes, state partition used ${state_used} of ${state_fs_bytes} bytes" cmdline="$(cat /proc/cmdline)" -for w in rauc.slot=A panic=10 ro psi=1 BOOT_IMAGE=/usr/lib/moose/boot/vmlinuz; do +for w in "rauc.slot=$BOOTED" panic=10 ro psi=1 BOOT_IMAGE=/usr/lib/moose/boot/vmlinuz; do grep -qw -- "$w" <<<"$cmdline" || layout_fail "kernel command line lacks '$w': $cmdline" done -echo "cloud-assertions: layout: / is slot A read-only, booted by GRUB from the slot's own kernel ($cmdline)" +echo "cloud-assertions: layout: / is slot $BOOTED read-only, booted by GRUB from the slot's own kernel ($cmdline)" # The slot's initramfs can find a provider's disk. Hetzner Cloud presents the # boot disk as virtio-SCSI, which needs virtio_scsi and the SCSI disk driver # sd_mod; without sd_mod no /dev/sda appears and the boot hangs in the @@ -289,7 +320,7 @@ echo "cloud-assertions: layout: pinned daemon.json/subuid/subgid/login.defs, SSH # The bootloader side: RAUC reads the slot config and the grubenv, and the # box reboots rather than waits in emergency or rescue mode. rauc_out="$(rauc status 2>&1)" || layout_fail "rauc status failed: $(tail -n3 <<<"$rauc_out" | tr '\n' ' ')" -grep -qi 'booted from: *rootfs.0 (A)' <<<"$rauc_out" || layout_fail "rauc does not see slot A as booted: $(tr '\n' ' ' <<<"$rauc_out")" +grep -qi "booted from: *rootfs.${booted_idx} (${BOOTED})" <<<"$rauc_out" || layout_fail "rauc does not see slot $BOOTED as booted: $(tr '\n' ' ' <<<"$rauc_out")" # The keyring a bundle must chain to (#562) comes from the slot, never from the # state partition: a new image's keyring must reach the box, so the box never # writes it into the /etc upper layer. @@ -297,7 +328,7 @@ grep -qi 'booted from: *rootfs.0 (A)' <<<"$rauc_out" || layout_fail "rauc does n [ ! -e "$up/rauc/keyring.pem" ] || layout_fail "/etc/rauc/keyring.pem is in the /etc upper layer, so it no longer follows the image" grep -qx 'check-purpose=codesign' /etc/rauc/system.conf || layout_fail "/etc/rauc/system.conf does not ask for check-purpose=codesign" echo "cloud-assertions: layout: rauc keyring from the slot: $(openssl x509 -in /etc/rauc/keyring.pem -noout -subject 2>/dev/null || head -c 40 /etc/rauc/keyring.pem)" -grub-editenv /efi/grub/grubenv list | grep -qx 'A_OK=1' || layout_fail "grubenv does not mark slot A good: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ')" +grub-editenv /efi/grub/grubenv list | grep -qx "${BOOTED}_OK=1" || layout_fail "grubenv does not mark slot $BOOTED good: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ')" for u in emergency.service rescue.service; do systemctl cat "$u" 2>/dev/null | grep -q 'systemctl --no-block reboot' || layout_fail "$u has no reboot drop-in" done @@ -308,7 +339,7 @@ journalctl -k -b --no-pager 2>/dev/null | grep -q "moose-state: boot disk is $ro || dmesg 2>/dev/null | grep -q "moose-state: boot disk is $root_disk " \ || layout_fail "state-setup did not report $root_disk as the boot disk" [ "$(lsblk -no PKNAME "$state_dev")" = "$(basename "$root_disk")" ] || layout_fail "the state partition $state_dev is not on the boot disk $root_disk" -echo "cloud-assertions: layout: rauc sees slot A booted, grubenv has A_OK=1, emergency and rescue reboot" +echo "cloud-assertions: layout: rauc sees slot $BOOTED booted, grubenv has ${BOOTED}_OK=1, emergency and rescue reboot" # --- 1c. the baked host-agent carries a real build stamp (BUILD.md # Versioning: # "every build stamps two fields"). An unstamped build reports internal/version's @@ -631,6 +662,7 @@ access) DASH_HOST="$(json_str "$SEED" box_id).onmoose.io" ;; update) DASH_HOST="$(json_str "$SEED" box_id).onmoose.io" ;; ssh) DASH_HOST="$(json_str "$SEED" box_id).onmoose.io" ;; remap|remap-reboot) DASH_HOST="$(json_str "$SEED" box_id).onmoose.io" ;; +os-update|os-revert) DASH_HOST="$(json_str "$SEED" box_id).onmoose.io" ;; esac echo "cloud-assertions: probing control plane at Host=$DASH_HOST (mode=$MODE)" @@ -2297,6 +2329,219 @@ remap|remap-reboot) echo "cloud-assertions: remap-reboot: every tier checked again after a real reboot of the same disk; the caps-tier app kept its data (token $rs_token)" fi ;; +os-update|os-revert) + # THE A/B OS UPDATE (#563, UPDATES.md # 1). host-agent reads an OS target + # from its update-target source, downloads the bundle, checks its sha256 + # against the target, has RAUC install it into slot B, switches in the + # window, and on the next boot keeps or gives up the new slot. The bundle + # is the boot-proof slot repacked with a host-agent one patch release up + # (dev/cloud/test/build-os-test-bundle.sh), on a read-only second disk, and + # served to host-agent by a file server inside the guest. One stage per + # boot; the stage is on the state partition, so it survives the reboots. + # os-update: 1 (slot A) wrong digest refused, install ahead of the + # window, switch; 2 (slot B) marked good, everything intact. + # os-revert: 1 (slot A) install and switch, host-agent kept off slot B; + # 2 (slot B) handled at the top of this script; 3 (slot A) + # reverted on its own, everything intact. + [ -f "$SEED" ] || fail "$MODE mode but $SEED absent" + box_id="$(json_str "$SEED" box_id)" + apex="${box_id}.onmoose.io" + os_cred="$(tr -d '\r\n' < "${CREDENTIALS_DIRECTORY:-/nonexistent}/moose.os_test" 2>/dev/null || true)" + IFS=: read -r os_mode os_ver os_sum <<<"$os_cred" + [ -n "$os_ver" ] && [ -n "$os_sum" ] || fail "$MODE: moose.os_test credential missing or malformed ('$os_cred')" + base_ver="$(/usr/lib/moose/host-agent-real --version | awk '{print $2}')" + stage="$(os_stage)" + target_dir=/var/lib/moose/test-target + ha_log() { journalctl -u host-agent.service -b --no-pager -o short-unix 2>&1; } + wait_ha() { # PATTERN SECONDS WHAT + for _i in $(seq 1 "$2"); do ha_log | grep -q -- "$1" && return 0; sleep 1; done + fail "$MODE: $3 within $2 s: $(ha_log | grep -i 'os update\|update target' | tail -8 | cut -c1-400 | tr '\n' ' ')" + } + # The admin read of what host-agent decided (BRAIN_UI_PROTOCOL: admin-only). + os_read() { sed '1,/^\r*$/d' <<<"$(full_get /api/v1/system/update-target "$apex" "$session_cookie" 2>/dev/null || true)" | tr -d '\r'; } + os_field() { grep -o "\"os\":{.*" <<<"$1" | json_str_of "$(cat)" "$2"; } + rauc_slot_version() { rauc status --detailed --output-format=json 2>/dev/null | grep -o "\"bootname\":\"$1\"[^]]*" | grep -o '"version":"[^"]*"' | head -1 | cut -d'"' -f4; } + grubvar() { grub-editenv /efi/grub/grubenv list | sed -n "s/^$1=//p"; } + boot_gate() { + local fu ro + fu="$(systemctl list-units --state=failed --no-legend --plain 2>/dev/null | awk '{print $1}' | tr '\n' ' ')" + [ -z "${fu// /}" ] || fail "$MODE: failed units before the switch: $fu" + ro="$(journalctl -b --no-pager -o json 2>/dev/null | grep -v '"CONTAINER_NAME"' | grep -i 'read-only file system' | head -n 3 | cut -c1-400)" + [ -z "$ro" ] || fail "$MODE: something tried to write to the read-only slot: $ro" + } + write_os_target() { # SHA256 WINDOW + printf '{"version":"os-test","window":"%s","os":[{"version":"%s","bundle_url":"http://127.0.0.1:5001/bundles/os-test.raucb","bundle_sha256":"%s"}]}\n' \ + "$2" "$os_ver" "$1" > "$target_dir/target.json" + } + facts() { # the per-box state an OS update must not touch + echo "shadow=$(grep "^${owner}:" /etc/shadow | cut -d: -f1-2)" + echo "hostkey=$(ssh-keygen -lf /etc/ssh/ssh_host_ed25519_key.pub 2>/dev/null | awk '{print $2}')" + echo "machineid=$(cat /etc/machine-id)" + echo "data=$(cat "/home/${owner}/os-update-data.txt" 2>/dev/null)" + } + + if [ "$stage" = 1 ]; then + [ "$BOOTED" = A ] || fail "$MODE: the first boot is on slot '$BOOTED', want A" + sso_token="$(tr -d '\r\n' < "${CREDENTIALS_DIRECTORY:-/nonexistent}/moose.sso_token" 2>/dev/null || true)" + [ -n "$sso_token" ] || fail "$MODE: moose.sso_token credential missing" + sso_resp="$(full_get "/_moose/sso?token=${sso_token}" "$apex" 2>/dev/null || true)" + grep -q ' 303' <<<"$(status_of "$sso_resp")" || fail "$MODE: SSO landing did not 303: status='$(status_of "$sso_resp")'" + session_cookie="$(cookie_val "$sso_resp" moose_session)" + [ -n "$session_cookie" ] || fail "$MODE: no moose_session cookie from the SSO landing" + owner="$(json_str_of "$(full_get /api/v1/me "$apex" "$session_cookie" 2>/dev/null || true)" username)" + [ -n "$owner" ] && id -u "$owner" >/dev/null 2>&1 || fail "$MODE: the SSO owner '$owner' has no host account" + head -c 16 /dev/urandom | od -An -tx1 | tr -d ' \n' > "/home/${owner}/os-update-data.txt" + chown "$owner" "/home/${owner}/os-update-data.txt" + mkdir -p "$OS_STATE_DIR" && chmod 700 "$OS_STATE_DIR" + app_id="" + if [ "$MODE" = os-update ] && [ "$os_mode" = full ]; then + # An app, so "apps intact" means a container that comes back. + resp="$(full_send POST /api/v1/apps "$apex" "$session_cookie" '{"manifest_id":"whoami","scope":"personal"}' 2>/dev/null)" + job="$(json_str_of "$resp" job_id)" + [ -n "$job" ] || fail "$MODE: install whoami did not start: status='$(status_of "$resp")'" + st=""; jr="" + for _i in $(seq 1 300); do + jr="$(full_get "/api/v1/jobs/${job}" "$apex" "$session_cookie" 2>/dev/null || true)" + st="$(json_str_of "$jr" status)" + case "$st" in completed|failed|cancelled|stalled) break ;; esac + sleep 1 + done + [ "$st" = completed ] || fail "$MODE: install whoami ended '$st'" + app_id="$(json_str_of "$jr" instance_id)" + docker ps --filter "label=moose.instance_id=${app_id}" --filter status=running --format '{{.Names}}' | grep -q . \ + || fail "$MODE: whoami has no running container" + fi + printf 'session_cookie=%q\nowner=%q\napp_id=%q\nbase_ver=%q\n' "$session_cookie" "$owner" "$app_id" "$base_ver" > "$OS_STATE_DIR/before" + facts > "$OS_STATE_DIR/facts" + echo "cloud-assertions: $MODE: owner '$owner' signed in, data written${app_id:+, whoami installed ($app_id)}; running OS $base_ver on slot A" + + # The bundle disk, the in-guest source, and host-agent pointed at it. + mkdir -p "$target_dir/bundles" + bdev="$(blkid -L moose-os-test 2>/dev/null || true)" + [ -n "$bdev" ] || fail "$MODE: no disk labelled moose-os-test (the harness's bundle disk)" + mount -o ro "$bdev" "$target_dir/bundles" || fail "$MODE: cannot mount the bundle disk $bdev" + [ "$(sha256sum "$target_dir/bundles/os-test.raucb" | cut -d' ' -f1)" = "$os_sum" ] || fail "$MODE: the bundle on the disk is not the one the harness named" + mkdir -p /etc/systemd/system/host-agent.service.d + cat > /etc/systemd/system/host-agent.service.d/30-os-update-test.conf <<'UNIT' +[Service] +Environment=MOOSE_UPDATE_OS_URL_PREFIX=http://127.0.0.1:5001/ +UNIT + if [ "$MODE" = os-revert ]; then + # The broken slot: host-agent is kept from starting on slot B + # only, so the old slot still runs it after the revert. + cat > /etc/systemd/system/host-agent.service.d/90-os-revert-test.conf <<'UNIT' +[Service] +ExecStartPre=/bin/sh -c 'if grep -qw rauc.slot=B /proc/cmdline; then echo "os-revert test: host-agent is kept off slot B"; exit 1; fi' +UNIT + fi + systemctl daemon-reload + caddy_image="$(docker inspect -f '{{.Config.Image}}' moose-caddy 2>/dev/null || true)" + docker rm -f moose-test-target >/dev/null 2>&1 || true + # A shut window: twelve hours from now, one minute wide. + shut="$(printf '%02d:00-%02d:01' $(( (10#$(date +%H) + 12) % 24 )) $(( (10#$(date +%H) + 12) % 24 )))" + write_os_target "$(printf '%064d' 0)" "$shut" + docker run -d --name moose-test-target -p 127.0.0.1:5001:80 -v "$target_dir":/srv:ro "$caddy_image" \ + caddy file-server --root /srv --listen :80 >/dev/null 2>&1 || fail "$MODE: could not start the in-guest file server" + for _i in $(seq 1 60); do grep -q ' 200' <<<"$(http_status_addr 127.0.0.1 5001 /target.json 2>/dev/null)" && break; sleep 1; done + + if [ "$MODE" = os-update ]; then + # 1a. A WRONG DIGEST. The signature would pass; the digest the + # target names does not, so RAUC never sees the bundle. + systemctl restart host-agent.service || fail "$MODE: could not restart host-agent" + wait_ha "update target names; refusing it" 180 "host-agent did not refuse a bundle whose digest the target does not name" + [ -z "$(rauc_slot_version B)" ] || fail "$MODE: slot B holds '$(rauc_slot_version B)' after a refused download" + st="$(os_field "$(os_read)" state)" + [ "$st" = failed ] || fail "$MODE: the update-target read says os.state '$st' after a wrong digest, want failed" + echo "cloud-assertions: os-update: WRONG DIGEST OK (refused before RAUC saw it; slot B untouched; os.state=failed)" + + if [ "$os_mode" = refuse ]; then + # 1b-refuse. A run that publishes the OS: the image trusts only + # the release root, so the throwaway-signed bundle must be + # refused by RAUC even with the right digest. + write_os_target "$os_sum" "$shut" + systemctl restart host-agent.service + wait_ha 'install failed.*rauc install' 300 "RAUC did not refuse the throwaway-signed bundle" + [ -z "$(rauc_slot_version B)" ] || fail "$MODE: an image with the release keyring INSTALLED a throwaway-signed bundle" + echo "cloud-assertions: os-update: RELEASE KEYRING OK (the right digest, but RAUC refused the throwaway signature; slot B untouched)" + boot_gate + ok + fi + fi + + # 1b. INSTALL AHEAD OF THE WINDOW. The right digest and a shut + # window: the bundle goes into slot B, and the boot order stays. + write_os_target "$os_sum" "$shut" + systemctl restart host-agent.service + wait_ha "installed into the other slot" 300 "the bundle was not installed into slot B" + [ "$(rauc_slot_version B)" = "$os_ver" ] || fail "$MODE: slot B holds '$(rauc_slot_version B)', want $os_ver" + [ "$(grubvar ORDER)" = "A B" ] || fail "$MODE: the boot order moved outside the window: ORDER='$(grubvar ORDER)'" + for _i in $(seq 1 30); do [ "$(os_field "$(os_read)" state)" = installed ] && break; sleep 1; done + [ "$(os_field "$(os_read)" state)" = installed ] || fail "$MODE: os.state is '$(os_field "$(os_read)" state)', want installed" + t_dl="$(ha_log | grep 'downloading the bundle' | tail -1 | awk '{print $1}')" + t_got="$(ha_log | grep 'bundle downloaded and its digest matches' | tail -1 | awk '{print $1}')" + t_in="$(ha_log | grep 'installed into the other slot' | tail -1 | awk '{print $1}')" + echo "cloud-assertions: $MODE: INSTALL AHEAD OK (slot B holds $os_ver, ORDER still 'A B', os.state=installed); measured: download+digest $(awk -v a="$t_dl" -v b="$t_got" 'BEGIN{printf "%.1f", b-a}') s, rauc install $(awk -v a="$t_got" -v b="$t_in" 'BEGIN{printf "%.1f", b-a}') s, bundle $(stat -c %s "$target_dir/bundles/os-test.raucb") bytes" + + # 1c. THE SWITCH. An open window: host-agent puts slot B first and + # reboots. The next stage runs on slot B. + boot_gate + date +%s > "$OS_STATE_DIR/switch-at" + set_os_stage 2 + write_os_target "$os_sum" "00:00-23:59" + systemctl restart host-agent.service + echo "cloud-assertions: $MODE: window open; waiting for host-agent to switch to slot B and reboot" + sleep 900 + fail "$MODE: host-agent never switched slots and rebooted: $(ha_log | grep -i 'os update' | tail -6 | tr '\n' ' ')" + fi + + # The last stage: back on a slot that must be healthy and marked good. + # shellcheck disable=SC1090 + . "$OS_STATE_DIR/before" + if [ "$MODE" = os-update ]; then + [ "$stage" = 2 ] && [ "$BOOTED" = B ] || fail "os-update: stage $stage booted slot '$BOOTED', want stage 2 on slot B" + wait_ha "the new slot is healthy and marked good" 300 "host-agent did not mark the new slot good" + [ ! -e /var/lib/moose/os-update/trial-B ] || fail "os-update: the trial marker for slot B is still there" + [ "$(grubvar ORDER)" = "B A" ] && [ "$(grubvar B_OK)" = 1 ] && [ "$(grubvar B_TRY)" = 0 ] \ + || fail "os-update: grubenv after mark-good: $(grub-editenv /efi/grub/grubenv list | tr '\n' ' ')" + [ "$(/usr/lib/moose/host-agent-real --version | awk '{print $2}')" = "$os_ver" ] || fail "os-update: slot B's host-agent is not $os_ver" + want_outcome=good; want_state=current; want_ver="$os_ver" + t_good="$(ha_log | grep 'the new slot is healthy and marked good' | tail -1 | awk '{print $1}')" + echo "cloud-assertions: os-update: SWITCH OK (booted slot B, marked good, grubenv ORDER='B A' B_OK=1 B_TRY=0); measured: switch to marked good $(awk -v a="$(cat "$OS_STATE_DIR/switch-at")" -v b="$t_good" 'BEGIN{printf "%.0f", b-a}') s, the reboot included" + note_pat="updated its system to $os_ver" + else + [ "$stage" = 3 ] && [ "$BOOTED" = A ] || fail "os-revert: stage $stage booted slot '$BOOTED', want stage 3 back on slot A" + wait_ha "the box went back to the old slot" 120 "host-agent did not record the revert" + [ ! -e /var/lib/moose/os-update/trial-B ] || fail "os-revert: the trial marker for slot B is still there" + [ "$(grubvar B_OK)" = 0 ] && [ "$(grubvar A_OK)" = 1 ] || fail "os-revert: grubenv after the revert: $(grub-editenv /efi/grub/grubenv list | tr '\n' ' ')" + [ "$(/usr/lib/moose/host-agent-real --version | awk '{print $2}')" = "$base_ver" ] || fail "os-revert: back on slot A, host-agent is not $base_ver" + want_outcome=reverted; want_state=held; want_ver="$base_ver" + echo "cloud-assertions: os-revert: REVERT OK (slot B never came up; back on slot A, B marked bad, A good); measured: switch to back on slot A $(( $(date +%s) - $(cat "$OS_STATE_DIR/switch-at") )) s, two reboots and the safety net included" + note_pat="did not work, so moose went back" + fi + # What the brain reports, and what the box kept. + me="" + for _i in $(seq 1 90); do me="$(status_of "$(full_get /api/v1/me "$apex" "$session_cookie" 2>/dev/null || true)")"; grep -q ' 200' <<<"$me" && break; sleep 1; done + grep -q ' 200' <<<"$me" || fail "$MODE: the owner's session does not work after the OS move (status '$me')" + ver_body="$(full_get /api/v1/system/version "$apex" "$session_cookie" 2>/dev/null || true)" + [ "$(json_str_of "$ver_body" os_version)" = "$want_ver" ] && [ "$(json_str_of "$ver_body" os_slot)" = "$BOOTED" ] \ + || fail "$MODE: /api/v1/system/version reports os_version '$(json_str_of "$ver_body" os_version)' os_slot '$(json_str_of "$ver_body" os_slot)', want $want_ver $BOOTED" + rd=""; for _i in $(seq 1 60); do rd="$(os_read)"; [ "$(os_field "$rd" state)" = "$want_state" ] && break; sleep 2; done + [ "$(os_field "$rd" state)" = "$want_state" ] && [ "$(os_field "$rd" outcome)" = "$want_outcome" ] \ + || fail "$MODE: the update-target read says os.state '$(os_field "$rd" state)' outcome '$(os_field "$rd" outcome)', want $want_state $want_outcome: $(grep -o '"os":{.*' <<<"$rd" | cut -c1-400)" + now_facts="$(facts)" + [ "$now_facts" = "$(cat "$OS_STATE_DIR/facts")" ] \ + || fail "$MODE: per-box state changed across the OS move: before [$(tr '\n' ' ' < "$OS_STATE_DIR/facts")] now [$(tr '\n' ' ' <<<"$now_facts")]" + if [ -n "$app_id" ]; then + for _i in $(seq 1 180); do docker ps --filter "label=moose.instance_id=${app_id}" --filter status=running --format '{{.Names}}' | grep -q . && break; sleep 1; done + docker ps --filter "label=moose.instance_id=${app_id}" --filter status=running --format '{{.Names}}' | grep -q . \ + || fail "$MODE: the whoami app did not come back after the OS move" + fi + echo "cloud-assertions: $MODE: INTACT OK (owner '$owner', session, password hash, SSH host key, machine-id, data file${app_id:+ and the whoami app} unchanged; os_version $want_ver on slot $BOOTED; os.state=$want_state, last outcome $want_outcome)" + nb="" + for _i in $(seq 1 150); do nb="$(full_get /api/v1/notifications "$apex" "$session_cookie" 2>/dev/null || true)"; grep -q "$note_pat" <<<"$nb" && break; sleep 2; done + grep -q "$note_pat" <<<"$nb" || fail "$MODE: no admin notification '$note_pat': $(sed '1,/^\r*$/d' <<<"$nb" | cut -c1-400)" + echo "cloud-assertions: $MODE: NOTIFY OK (admins were told: '$note_pat')" + ;; *) fail "unknown assert mode '$MODE'" ;; diff --git a/dev/cloud/mkosi.extra/etc/rauc/system.conf b/dev/cloud/mkosi.extra/etc/rauc/system.conf index a4260ccb..ce1310ff 100644 --- a/dev/cloud/mkosi.extra/etc/rauc/system.conf +++ b/dev/cloud/mkosi.extra/etc/rauc/system.conf @@ -1,5 +1,6 @@ # RAUC slot config for the hosted A/B image (BUILD.md # 1b). The bundle and the -# keyring are built (#562); the install and mark-good are #563. +# keyring are built (#562); host-agent installs, switches and marks good (#563, +# internal/hostagent/osupdate). [system] compatible=moose-hosted-x86_64 bootloader=grub @@ -8,6 +9,11 @@ grubenv=/efi/grub/grubenv # survives a slot swap. data-directory=/var/lib/rauc bundle-formats=-plain +# An install writes the other slot and leaves the boot order alone. host-agent +# installs ahead of the update window and switches (`rauc status mark-active +# other`) only inside it (UPDATES.md # 1, #563), so a reboot for any other +# reason between the two never boots the new slot early. +activate-installed=false # The CA a bundle's signer must chain to (#562, BUILD.md # 1b # The bundle). # keyring.pem is not in this tree: dev/cloud/stage-control-plane.sh stages it diff --git a/dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.service b/dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.service new file mode 100644 index 00000000..c2a746c5 --- /dev/null +++ b/dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.service @@ -0,0 +1,10 @@ +[Unit] +Description=moose: reboot to the other slot if a new OS slot was never marked good +Documentation=https://github.com/onmoose/os/blob/main/docs/specs/UPDATES.md +# Only on a slot under trial. The script checks the marker itself too. +ConditionPathExistsGlob=/var/lib/moose/os-update/trial-* + +[Service] +Type=oneshot +ExecStart=/usr/lib/moose/os-trial-check +StandardOutput=journal+console diff --git a/dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.timer b/dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.timer new file mode 100644 index 00000000..39fc79ea --- /dev/null +++ b/dev/cloud/mkosi.extra/etc/systemd/system/moose-os-trial.timer @@ -0,0 +1,13 @@ +[Unit] +Description=moose: check that a new OS slot was marked good (UPDATES.md # 1) +Documentation=https://github.com/onmoose/os/blob/main/docs/specs/UPDATES.md + +[Timer] +# Later than host-agent's own trial deadline (MOOSE_OS_TRIAL_TIMEOUT, 10 min), +# so a host-agent that is alive always decides first. This only catches a +# host-agent that never started or hung. +OnBootSec=15min +AccuracySec=10s + +[Install] +WantedBy=timers.target diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check new file mode 100755 index 00000000..1fefdc31 --- /dev/null +++ b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check @@ -0,0 +1,20 @@ +#!/bin/sh +# The safety net for an OS update trial (UPDATES.md # 1, #563), run by +# moose-os-trial.timer some minutes after boot. +# +# host-agent writes /var/lib/moose/os-update/trial- before it switches to +# a new slot, and removes it once the new slot is healthy and marked good +# (internal/hostagent/osupdate). If the marker for the slot this box booted is +# still there now, the new slot never got that far: host-agent did not start, +# or it hung. host-agent cannot reboot the box then, so this does. GRUB skips a +# slot still on trial (TRY=1), so the next boot is the old slot. +# +# Nothing else: a boot that is not on trial has no marker, so this does nothing +# on every normal boot, and on a box that never updated its OS. +set -u +slot="$(sed -n 's/.*rauc\.slot=\([AB]\).*/\1/p' /proc/cmdline)" +[ -n "$slot" ] || exit 0 +[ -e "/var/lib/moose/os-update/trial-${slot}" ] || exit 0 +echo "moose-os-trial: slot ${slot} was not marked good in time; rebooting so the box goes back to the other slot" +rauc status mark-bad booted || echo "moose-os-trial: mark-bad failed; rebooting anyway, GRUB skips a slot still on trial" +systemctl reboot diff --git a/dev/cloud/mkosi.postinst.chroot b/dev/cloud/mkosi.postinst.chroot index 0a7272c5..c13bcff4 100755 --- a/dev/cloud/mkosi.postinst.chroot +++ b/dev/cloud/mkosi.postinst.chroot @@ -159,6 +159,13 @@ EOF # are committed static files under mkosi.extra/ (etc/moose/metadata-firewall.nft). enable_unit /etc/systemd/system/moose-metadata-firewall.service moose-metadata-firewall.service +# The OS update trial's safety net (UPDATES.md # 1, #563): reboots a box whose +# new slot was never marked good, when host-agent itself cannot. A timer, so +# it goes under timers.target. +mkdir -p /etc/systemd/system/timers.target.wants +ln -sf /etc/systemd/system/moose-os-trial.timer /etc/systemd/system/timers.target.wants/moose-os-trial.timer +chmod 0755 /usr/lib/moose/os-trial-check + # Docker's daemon-wide userns-remap (#530, BUILD.md # User-namespace remap). # daemon.json names moose-remap (made by /usr/lib/sysusers.d/moose.conf, with diff --git a/dev/cloud/run-cloud-tests.sh b/dev/cloud/run-cloud-tests.sh index 5aaea0fc..f3b214b2 100755 --- a/dev/cloud/run-cloud-tests.sh +++ b/dev/cloud/run-cloud-tests.sh @@ -94,6 +94,11 @@ BOX_ID_SSH=heron-birch # The remap boots (#531) share ONE overlay of their own and one box-id: the # second boot is a reboot of the disk the first one installed apps on. BOX_ID_REMAP=moss-lynx +# The OS update boots (#563) each provision their OWN box on a fresh overlay: +# they write slot B and reboot the box between slots, so no other boot may see +# that disk. +BOX_ID_OS_UPDATE=fern-stoat +BOX_ID_OS_REVERT=wren-maple # Which boots to run, space-separated (unseeded seeded frozen access update ssh remap). # Default: all. @@ -131,7 +136,16 @@ BOX_ID_REMAP=moss-lynx # checks them all again. One name runs both, because the second needs what # the first left on the disk. # All of these are in the gate: see ci-cloud-image.yml. -BOOTS="${MOOSE_CLOUD_BOOTS:-unseeded seeded frozen access update ssh remap}" +# - `os-update` and `os-revert` (#563) prove the A/B OS update on the booted +# image. `os-update`: a wrong digest is refused, the bundle installs into +# slot B ahead of the window, the box switches in the window, boots slot B +# and marks it good, with the owner, an app, the SSH host keys and the data +# intact. `os-revert`: the same bundle, but host-agent cannot start on +# slot B, so the image's safety-net timer reboots the box and it comes back +# on slot A on its own. Each reboots its box inside one QEMU run, so +# neither passes -no-reboot. Both need the test bundle from +# dev/cloud/test/build-os-test-bundle.sh (MOOSE_CLOUD_OS_BUNDLE_DIR). +BOOTS="${MOOSE_CLOUD_BOOTS:-unseeded seeded frozen access update ssh remap os-update os-revert}" # Which firmwares run the boots (#561): by default every boot runs under UEFI and # again under legacy BIOS. `bios` used to be a boot of its own (one un-seeded # boot under SeaBIOS); it is a firmware now. @@ -147,9 +161,9 @@ boot_count=0 for b in $BOOTS; do boot_count=$((boot_count + 1)) case "$b" in - unseeded|seeded|frozen|access|update|ssh|remap) ;; + unseeded|seeded|frozen|access|update|ssh|remap|os-update|os-revert) ;; bios) echo "'bios' is no longer a boot: every boot runs under both firmwares (MOOSE_CLOUD_FIRMWARES, default 'uefi bios')" >&2; exit 1 ;; - *) echo "unknown boot '$b' in MOOSE_CLOUD_BOOTS='$BOOTS' (known: unseeded seeded frozen access update ssh remap)" >&2; exit 1 ;; + *) echo "unknown boot '$b' in MOOSE_CLOUD_BOOTS='$BOOTS' (known: unseeded seeded frozen access update ssh remap os-update os-revert)" >&2; exit 1 ;; esac done [ "$boot_count" -gt 0 ] || { echo "MOOSE_CLOUD_BOOTS='$BOOTS' names no boot" >&2; exit 1; } @@ -354,8 +368,10 @@ run_boot() { -device "virtio-net-pci,netdev=n0,mac=52:54:00:c1:0d:01" -smbios "type=11,value=io.systemd.credential:moose.assert=${mode}" "$@" - -no-reboot ) + # A boot that reboots its own box (the OS update boots) keeps QEMU up + # across the reboot; every other boot ends QEMU on a reboot. + if [ -z "${KEEP_REBOOTS:-}" ]; then qemu_args+=( -no-reboot ); fi if [ "$firmware" = uefi ] && [ -n "$OVMF_VARS" ]; then qemu_args+=( -drive "if=pflash,format=raw,file=${OVMF_VARS}" ) fi @@ -680,6 +696,60 @@ if should_run remap; then echo "boot remap-reboot OK: every tier checked again after a real reboot of the same disk (box_id=${BOX_ID_REMAP})" fi +# --- 12. the OS update boots (#563). Each takes its own fresh overlay and +# box-id, a test-portal key and one owner assertion (the update-target read is +# admin-only), and a second, read-only disk holding the test bundle: the box is +# air-gapped, so the bundle reaches it as an ext4 image that the guest mounts +# and serves to host-agent from a file server inside the guest. The disk is +# made here, from MOOSE_CLOUD_OS_BUNDLE_DIR, with mke2fs -d (no root needed for +# that part). The guest reboots between slots inside one QEMU run. +# +# MOOSE_CLOUD_OS_REFUSE_ONLY=true runs only the refusal half of os-update: on +# a run that publishes the OS, the boot-proof image trusts only the release +# root, so the throwaway-signed test bundle must be refused (the maintainer's +# call for #563). +os_boot() { # NAME BOX_ID + local name="$1" box="$2" dir="${MOOSE_CLOUD_OS_BUNDLE_DIR:-}" mint key token disk sum ver mode + [ -n "$GO" ] && [ -x "$GO" ] || { echo "$name boot needs go to mint the owner assertion; none found (\$GO='${GO:-}')" >&2; exit 1; } + [ -n "$dir" ] && [ -f "$dir/os-test.raucb" ] && [ -f "$dir/os-test.sha256" ] && [ -f "$dir/os-test.version" ] || { + echo "$name boot needs the test bundle: set MOOSE_CLOUD_OS_BUNDLE_DIR to the output of dev/cloud/test/build-os-test-bundle.sh" >&2 + exit 1 + } + mapfile -t mint < <(mint_owner_assertion "$box") || true + key="${mint[0]:-}"; token="${mint[1]:-}" + [ -n "$key" ] && [ -n "$token" ] || { echo "$name boot: failed to mint the owner assertion" >&2; exit 1; } + disk="${RUN_DIR}/os-bundle-${FIRMWARE}-${name}.img" + local bytes; bytes="$(stat -c %s "$dir/os-test.raucb")" + mke2fs -q -t ext4 -L moose-os-test -d "$dir" "$disk" "$(( bytes / 1048576 + 64 ))M" + sum="$(tr -d '[:space:]' < "$dir/os-test.sha256")" + ver="$(tr -d '[:space:]' < "$dir/os-test.version")" + mode=full + [ "${MOOSE_CLOUD_OS_REFUSE_ONLY:-false}" = true ] && mode=refuse + + OVERLAY="${RUN_DIR}/overlay-${FIRMWARE}-${name}.qcow2" + new_overlay "$OVERLAY" + VERDICT_TIMEOUT=1500 + KEEP_REBOOTS=1 + if ! run_boot "${FIRMWARE}-${name}" "$name" \ + -drive "file=${disk},if=virtio,format=raw,readonly=on" \ + -smbios "type=11,value=$(seed_cred_keyed "$box" "$key" "http://127.0.0.1:5001/target.json")" \ + -smbios "type=11,value=io.systemd.credential.binary:moose.sso_token=$(printf '%s' "$token" | base64 -w0)" \ + -smbios "type=11,value=io.systemd.credential:moose.os_test=${mode}:${ver}:${sum}"; then + KEEP_REBOOTS="" + echo "cloud ${name} proof: ${VERDICT}" >&2 + exit 1 + fi + KEEP_REBOOTS="" +} +if should_run os-update; then + os_boot os-update "$BOX_ID_OS_UPDATE" + echo "boot os-update OK: the OS update installed into slot B, the box switched, booted slot B and marked it good, with its owner, app, host keys and data intact (box_id=${BOX_ID_OS_UPDATE})" +fi +if should_run os-revert; then + os_boot os-revert "$BOX_ID_OS_REVERT" + echo "boot os-revert OK: slot B never came up, the safety net rebooted the box and it went back to slot A on its own (box_id=${BOX_ID_OS_REVERT})" +fi + echo "firmware ${FIRMWARE}: every boot OK (${BOOTS})" done diff --git a/dev/cloud/test/bootstrap.sh b/dev/cloud/test/bootstrap.sh index 58a0d0ed..eced8b9d 100755 --- a/dev/cloud/test/bootstrap.sh +++ b/dev/cloud/test/bootstrap.sh @@ -33,7 +33,7 @@ WIRING="${CLOUD_DIR}/mkosi.extra.wiring" # shared production wiring (ExtraTree o PKGMNGR="${TEST_DIR}/mkosi.pkgmngr" CP_BUNDLE="${REPO_ROOT}/.dev/control-plane" CANARY="${WORK}/.cloud-boot-ready" -CANARY_VERSION="v27" # bump when staging/mkosi.conf/repart changes require a clean rebuild +CANARY_VERSION="v28" # bump when staging/mkosi.conf/repart changes require a clean rebuild # A change to the OS package lock (#560) must rebuild too, so all three lock # files are part of the canary. The resolved list is in it as well: a re-run # after only the list changed must not exit early and skip os_lock_check below. @@ -131,6 +131,17 @@ cp "${CLOUD_DIR}/cloud-assertions.sh" "$EXTRA/usr/local/bin/cloud-assertions.sh" chmod 0755 "$EXTRA/usr/local/bin/cloud-assertions.sh" cp "${TEST_DIR}/moose-cloud-assertions.service" "$EXTRA/etc/systemd/system/" +# The OS update trial's safety net fires after 3 minutes here instead of 15 +# (#563), so the os-revert boot does not sit out a quarter of an hour. Still +# longer than a healthy trial boot takes to mark its slot good, which the +# os-update boot proves under the same setting. The image that ships keeps 15. +mkdir -p "$EXTRA/etc/systemd/system/moose-os-trial.timer.d" +cat > "$EXTRA/etc/systemd/system/moose-os-trial.timer.d/10-cloud-test.conf" <<'EOF' +[Timer] +OnBootSec= +OnBootSec=180s +EOF + # --- 3b. app-install fixtures for the access-mode e2e (#308) — TEST-LANE ONLY. The # access boot installs whoami air-gapped and drives the per-app forward-auth access # modes through real Caddy. These land in the TEST ExtraTree (dev/cloud/test/ diff --git a/dev/cloud/test/build-os-test-bundle.sh b/dev/cloud/test/build-os-test-bundle.sh new file mode 100755 index 00000000..e1da104e --- /dev/null +++ b/dev/cloud/test/build-os-test-bundle.sh @@ -0,0 +1,98 @@ +#!/usr/bin/env bash +# Build the test-only OS bundle the os-update and os-revert boots install (#563). +# +# sudo -E dev/cloud/test/build-os-test-bundle.sh BOOT-PROOF.raw OUTDIR +# +# The bundle holds slot A of the boot-proof image, repacked with one change: a +# host-agent stamped one patch release above VERSION. So on slot B the box +# reports that version, and the update loop sees it as current. Nothing in it +# ever ships: it is signed with this checkout's throwaway key (dev/cloud/rauc.sh), +# which only a boot-proof image of the same build trusts. +# +# Kept cheap on purpose (the maintainer's call for #563): the slot is repacked +# with gzip at level 1, not xz, which GRUB also reads; the baked image tarballs +# under /var/lib/moose are left out, because a slot's /var/lib/moose is copied +# to the state partition only at a box's first boot and slot B never has one. +# +# Writes OUTDIR/os-test.raucb, OUTDIR/os-test.sha256 (the digest) and +# OUTDIR/os-test.version (the version in the bundle and in its host-agent). +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)" +# shellcheck source=dev/cloud/rauc.sh +. "${REPO_ROOT}/dev/cloud/rauc.sh" + +image="$(realpath "${1:?usage: build-os-test-bundle.sh BOOT-PROOF.raw OUTDIR}")" +out="${2:?usage: build-os-test-bundle.sh BOOT-PROOF.raw OUTDIR}" +GO="${GO:-$(command -v go || true)}" +[ -n "$GO" ] || { echo "build-os-test-bundle: go not found (set GO)" >&2; exit 1; } +CALLER="${SUDO_USER:-}" +t0=$(date +%s) + +rm -rf "$out" +mkdir -p "$out/work" +out="$(realpath "$out")" +[ -n "$CALLER" ] && chown -R "$CALLER" "$out" + +base="$(tr -d '[:space:]' < "${REPO_ROOT}/VERSION")" +IFS=. read -r maj min pat <<<"$base" +next="${maj}.${min}.$((pat + 1))" + +as_caller() { if [ -n "$CALLER" ]; then sudo -u "$CALLER" "$@"; else "$@"; fi; } + +# The host-agent for the new slot, built the way stage-control-plane.sh builds +# the baked one, with the next patch number. +commit="$(as_caller git -C "$REPO_ROOT" rev-parse --short HEAD 2>/dev/null || echo unknown)" +as_caller env CGO_ENABLED=1 CGO_CFLAGS=-D_GNU_SOURCE "$GO" build -C "$REPO_ROOT" -tags hosted \ + -ldflags "-X github.com/onmoose/os/internal/version.Version=${next} -X github.com/onmoose/os/internal/version.Commit=${commit}" \ + -o "$out/work/host-agent-real" ./cmd/host-agent-real/ +as_caller "$GO" build -C "$REPO_ROOT" -o "$out/work/slotbudget" ./dev/cloud/slotbudget +"$out/work/slotbudget" -extract "$out/work/slot.img" "$image" >/dev/null +rm -f "$out/work/slotbudget" + +rauc_throwaway_ca +cp "$RAUC_THROWAWAY_DIR/root-ca.pem" "$RAUC_THROWAWAY_DIR/signer.pem" "$RAUC_THROWAWAY_DIR/signer.key" "$out/work/" + +rauc_run "$out/work" ' + s=$(date +%s) + unsquashfs -q -n -d root slot.img + rm -f slot.img + echo "unsquashfs: $(( $(date +%s) - s )) s" + install -m 0755 host-agent-real root/usr/lib/moose/host-agent-real + # Left empty, not removed: state-setup binds onto these paths. + find root/var/lib/moose -mindepth 1 -delete + s=$(date +%s) + mkdir -p bundle + mksquashfs root bundle/rootfs.img -comp gzip -Xcompression-level 1 -noappend -quiet + echo "mksquashfs (gzip level 1): $(( $(date +%s) - s )) s, $(stat -c %s bundle/rootfs.img) bytes" + compatible="$(sed -n "s/^compatible=//p" root/etc/rauc/system.conf | head -n1)" + cp root/etc/rauc/system.conf system.conf + rm -rf root + cat > bundle/manifest.raucm </dev/null +' +mv "$out/work/os-test.raucb" "$out/os-test.raucb" +rm -rf "$out/work" +sha256sum "$out/os-test.raucb" | cut -d' ' -f1 > "$out/os-test.sha256" +echo "$next" > "$out/os-test.version" +[ -n "$CALLER" ] && chown -R "$CALLER":"$(id -gn "$CALLER")" "$out" 2>/dev/null || true +bytes="$(stat -c %s "$out/os-test.raucb")" +echo "os test bundle: ${next}, ${bytes} bytes, sha256 $(cat "$out/os-test.sha256"), built in $(( $(date +%s) - t0 )) s" +{ + echo "### OS test bundle (#563, boot-proof only)" + echo "" + echo "Version ${next}, $(awk -v a="$bytes" 'BEGIN { printf "%.1f MB", a / 1e6 }'), built in $(( $(date +%s) - t0 )) s. Slot repacked with gzip level 1; never shipped." +} >> "${GITHUB_STEP_SUMMARY:-/dev/null}" diff --git a/dev/cloud/test/moose-cloud-assertions.service b/dev/cloud/test/moose-cloud-assertions.service index 7c271f18..ee44a6ad 100644 --- a/dev/cloud/test/moose-cloud-assertions.service +++ b/dev/cloud/test/moose-cloud-assertions.service @@ -35,6 +35,7 @@ ImportCredential=moose.sso_token # other boot, like the first. ImportCredential=moose.sso_token2 ImportCredential=moose.sso_token3 +ImportCredential=moose.os_test # The script polls the control plane up (docker load + brain bootstrap + compose # up race this unit) and can run for minutes; the default 90s start timeout would # kill it mid-poll before it writes a verdict. No timeout — the QEMU harness owns diff --git a/internal/api/system.go b/internal/api/system.go index fbe3f571..41a54c9c 100644 --- a/internal/api/system.go +++ b/internal/api/system.go @@ -23,7 +23,7 @@ func (s *Server) registerSystem(api huma.API) { }, s.systemStorage) huma.Register(api, huma.Operation{ OperationID: "get-system-version", Method: "GET", Path: "/api/v1/system/version", - Summary: "What this box is running: brain version and commit, host-agent version, UI image", + Summary: "What this box is running: brain version and commit, host-agent version, UI image, OS version and slot", }, s.systemVersion) } @@ -52,6 +52,11 @@ type SystemVersionDTO struct { Commit string `json:"commit"` HostAgentVersion string `json:"host_agent_version,omitempty"` UIImage string `json:"ui_image,omitempty"` + // OSVersion and OSSlot are the OS release the box runs and the A/B slot + // it booted ("A" or "B", #563). Absent when host-agent cannot tell (the + // appliance until #564) or could not be reached. + OSVersion string `json:"os_version,omitempty"` + OSSlot string `json:"os_slot,omitempty"` } // hostVersionReadTimeout bounds the host-agent leg of the version read. A var, @@ -101,6 +106,7 @@ func (s *Server) systemVersion(ctx context.Context, _ *struct{}) (*struct{ Body slog.Warn("system-version: host status read failed", "err", err) } else { out.HostAgentVersion = status.AgentVersion + out.OSVersion, out.OSSlot = status.OSVersion, status.OSSlot } if img, err := lifecycle.ControlPlaneUIImage(s.controlPlaneDir); err != nil { diff --git a/internal/api/systemupdate.go b/internal/api/systemupdate.go index 919d0b47..27e45cdc 100644 --- a/internal/api/systemupdate.go +++ b/internal/api/systemupdate.go @@ -111,6 +111,55 @@ type UpdateTargetDTO struct { WindowFrom string `json:"window_from,omitempty"` AutoApply bool `json:"auto_apply"` Profile string `json:"profile,omitempty"` + // OS is stream A's last decision, beside stream B's above (UPDATES.md # + // 1, #563). Absent when host-agent reports nothing about the OS. + OS *OSUpdateDTO `json:"os,omitempty"` +} + +// OSUpdateDTO is stream A on the update-target read: the OS this box runs, +// the release its update target names, and what the box last did about it. +// State is the word the dashboard writes its sentence from; Detail is a +// diagnostic underneath it, never UI copy. +type OSUpdateDTO struct { + State string `json:"state" enum:"unsupported,none,refused,current,installing,installed,waiting,rebooting,held,failed"` + Running string `json:"running,omitempty"` + Slot string `json:"slot,omitempty"` + Target *OSReleaseDTO `json:"target,omitempty"` + Detail string `json:"detail,omitempty"` + Last *OSOutcomeDTO `json:"last,omitempty"` +} + +// OSReleaseDTO is one OS release: the version and its bundle, pinned by +// sha256. +type OSReleaseDTO struct { + Version string `json:"version"` + BundleURL string `json:"bundle_url"` + BundleSHA256 string `json:"bundle_sha256"` +} + +// OSOutcomeDTO is the outcome of the last OS switch: "good" (the new slot came +// up healthy) or "reverted" (it did not, and the box went back). +type OSOutcomeDTO struct { + ID string `json:"id"` + Outcome string `json:"outcome" enum:"good,reverted"` + Version string `json:"version"` + From string `json:"from,omitempty"` + At string `json:"at"` +} + +// osUpdateDTO converts host-agent's stream A report. +func osUpdateDTO(o *protocol.OSUpdate) *OSUpdateDTO { + if o == nil { + return nil + } + out := &OSUpdateDTO{State: o.State, Running: o.Running, Slot: o.Slot, Detail: o.Detail} + if o.Target != nil { + out.Target = &OSReleaseDTO{Version: o.Target.Version, BundleURL: o.Target.BundleURL, BundleSHA256: o.Target.BundleSHA256} + } + if o.Last != nil { + out.Last = &OSOutcomeDTO{ID: o.Last.ID, Outcome: o.Last.Outcome, Version: o.Last.Version, From: o.Last.From, At: o.Last.At} + } + return out } // getSystemUpdateTarget reports what the box could be running. A pure read, so @@ -143,6 +192,7 @@ func (s *Server) getSystemUpdateTarget(ctx context.Context, _ *struct{}) (*struc WindowFrom: t.WindowFrom, AutoApply: t.AutoApply, Profile: t.Profile, + OS: osUpdateDTO(t.OS), } if t.Target != nil { out.Target = &UpdateTargetOfferDTO{ diff --git a/internal/hostagent/agent.go b/internal/hostagent/agent.go index 71092963..cf5f7baa 100644 --- a/internal/hostagent/agent.go +++ b/internal/hostagent/agent.go @@ -437,6 +437,11 @@ type Agent struct { // the endpoint reports state "unknown". UpdateTarget UpdateTargetReporter + // OS, when non-nil, backs os_version and os_slot of GET /v1/system/status: + // the OS release this box runs and the slot it booted (#563). Wired only + // on a box in the A/B layout (the hosted build); nil leaves both empty. + OS OSReporter + // Net, when non-nil, backs the interfaces field of GET /v1/discovery/state // with the LAN set. Swapped per binary: netstate.NMProvider (NetworkManager // over DBus) vs FakeNetState. When nil, interfaces reports empty — "not @@ -444,6 +449,12 @@ type Agent struct { Net NetState } +// OSReporter is a consumer-side interface for the running OS release and +// booted slot. Provider: osupdate.Applier. +type OSReporter interface { + Running() (version, slot string) +} + // SystemSampler is a consumer-side interface for the raw system-resources // sample behind GET /v1/system/resources. One call = one snapshot of raw // cumulative counters (never rates — the brain derives those by diffing @@ -582,7 +593,13 @@ func (a *Agent) systemStatus(w http.ResponseWriter, r *http.Request) { if a.DiskSpace != nil { disks = a.DiskSpace.Disks() } + var osVersion, osSlot string + if a.OS != nil { + osVersion, osSlot = a.OS.Running() + } writeJSON(w, http.StatusOK, protocol.SystemStatus{ + OSVersion: osVersion, + OSSlot: osSlot, Hostname: "moose-dev", UptimeS: int64(time.Since(a.startedAt).Seconds()), DiskPressure: false, diff --git a/internal/hostagent/jobs.go b/internal/hostagent/jobs.go index 9eb928f9..eba1effc 100644 --- a/internal/hostagent/jobs.go +++ b/internal/hostagent/jobs.go @@ -241,3 +241,57 @@ func (a *Agent) jobStatus(w http.ResponseWriter, r *http.Request) { } writeJSON(w, http.StatusOK, job) } + +// IsJobRunning reports whether err is the job lock's refusal: another job is +// already in flight. Callers outside this package (the OS update applier, via +// cmd/host-agent-real) treat it as "wait", not as a failure. +func IsJobRunning(err error) bool { return errors.Is(err, errJobRunning) } + +// StartJob runs fn as a job of the given kind under the one global job lock +// (the same lock system-update takes), bounded by maxDuration. It is how the +// OS update's install and switch run (#563): they are dangerous in the same way +// a control-plane update is, so they must never overlap one, or each other. +// +// The record has no Result: that field is the control-plane update's. A +// failure ends the job `failed` with the error text, and a run past +// maxDuration with the `job-timeout` code. +func (a *Agent) StartJob(kind string, maxDuration time.Duration, fn func(ctx context.Context) error) (protocol.Job, error) { + job, err := a.jobs.start(kind) + if err != nil { + return protocol.Job{}, err + } + go func() { + ctx, cancel := context.WithTimeout(context.Background(), maxDuration) + defer cancel() + err := fn(ctx) + code, msg := "", "" + if err != nil { + code, msg = protocol.JobErrorFailed, err.Error() + if errors.Is(ctx.Err(), context.DeadlineExceeded) { + code = protocol.JobErrorTimeout + } + } + a.jobs.finishPlain(job.ID, code, msg) + }() + return job, nil +} + +// finishPlain closes a job that carries no Result and releases the lock. +func (r *jobRegistry) finishPlain(id, code, message string) { + r.mu.Lock() + defer r.mu.Unlock() + job, ok := r.jobs[id] + if !ok { + return + } + job.FinishedAt = r.now().UTC().Format(time.RFC3339) + if code == "" { + job.Status = protocol.JobStatusCompleted + } else { + job.Status = protocol.JobStatusFailed + job.Error = &protocol.Error{Code: code, Message: message} + } + if r.running == id { + r.running = "" + } +} diff --git a/internal/hostagent/osupdate/osupdate.go b/internal/hostagent/osupdate/osupdate.go new file mode 100644 index 00000000..824d7722 --- /dev/null +++ b/internal/hostagent/osupdate/osupdate.go @@ -0,0 +1,638 @@ +// Package osupdate is stream A's transaction (UPDATES.md # 1, #563): it puts +// a new OS release into the other A/B slot, switches to it inside the update +// window, and decides on the next boot whether the new slot stays. +// +// The update-target loop (internal/hostagent/updatetarget) decides WHICH +// release and WHEN the window is open. This package does the work, in three +// parts: +// +// 1. Install, ahead of the window (an os-install job). Download the bundle to +// the state partition, check its sha256 against the one the answer names, +// and hand it to RAUC, which checks its signature against the image's +// keyring and writes it into the other slot. system.conf has +// activate-installed=false, so the boot order does not change yet. A +// failure here changes nothing the box runs. +// 2. Switch, inside the window (an os-switch job). `rauc status mark-active +// other` puts the new slot first with OK=1 and TRY=0 (RAUC's GRUB backend +// resets the slot's try flag there; this package checks it did), writes a +// trial marker for the new slot, and reboots. +// 3. Decide, at the next start of host-agent (Boot). +// - On the new slot, with its trial marker: the boot is on trial. Once the +// brain answers its health check the slot is marked good and the marker +// goes. If it does not answer in time, the slot is marked bad and the box +// reboots, and GRUB boots the old slot. If host-agent itself never gets +// this far, moose-os-trial.timer, a unit in the image, reboots the box. +// - On the old slot with the new slot's marker still there: the new slot +// failed. The box records the revert, marks the new slot bad and stays. +// - Any other boot is not on trial and is marked good at once, so a problem +// that has nothing to do with the OS never moves a box back to its older +// OS (the maintainer's call for #563). +// +// What survives a reboot lives on the state partition, in Dir: state.json (the +// record below) and one trial- marker file, which the image's timer reads +// without host-agent. +package osupdate + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "log/slog" + "net/http" + "os" + "path/filepath" + "strings" + "sync" + "time" + + "github.com/onmoose/os/internal/hostagent/updatetarget" + "github.com/onmoose/os/internal/protocol" +) + +// DefaultDir is where the record, the trial marker and the downloaded bundle +// live: under /var/lib/moose, a bind mount from the state partition, so it is +// the same directory on both slots. +const DefaultDir = "/var/lib/moose/os-update" + +// DefaultTrialTimeout is how long a trial boot waits for the brain before it +// gives the slot up. The image's moose-os-trial.timer fires later (15 min), so +// a live host-agent always decides first. +const DefaultTrialTimeout = 10 * time.Minute + +// Job bounds. A 438 MB download and install takes well under a minute on a +// hosted box; the bound is there for a stalled transfer. +const ( + installMaxDuration = 30 * time.Minute + switchMaxDuration = 5 * time.Minute +) + +// RAUC is the slice of RAUC this package drives. Consumer-side interface; the +// provider is CLIRAUC. +type RAUC interface { + Status(ctx context.Context) (Status, error) + Install(ctx context.Context, path string) error + Mark(ctx context.Context, state, which string) error + GrubEnvVars(ctx context.Context) (map[string]string, error) +} + +// Jobs runs work under host-agent's one job lock. Consumer-side; the provider +// is an adapter over hostagent.Agent.StartJob in cmd/host-agent-real. A +// refusal because a job is running must wrap ErrBusy. +type Jobs interface { + StartJob(kind string, maxDuration time.Duration, fn func(ctx context.Context) error) (jobID string, err error) +} + +// ErrBusy is the job lock's refusal, as Jobs reports it. +var ErrBusy = errors.New("another job is running") + +// Doer is the HTTP surface the download needs. +type Doer interface { + Do(*http.Request) (*http.Response, error) +} + +// Applier is stream A's transaction. Build it with every field set except the +// optional ones, call Boot once at start, then hand it to the update loop. +type Applier struct { + RAUC RAUC + Jobs Jobs + // Version is the OS release this slot carries: host-agent's own version, + // stamped from VERSION, the same number the slot's bundle was made with. + Version string + // Dir holds the record, the trial markers and the download. Empty means + // DefaultDir. + Dir string + // FloorFile is where the running brain writes its minimum_host_agent. A + // missing file means no floor check. + FloorFile string + // Healthy waits until the brain answers its health check or ctx ends. + Healthy func(ctx context.Context) error + // Reboot reboots the box. Nil means `systemctl reboot`. + Reboot func() error + // HTTP downloads the bundle. Nil means a plain client. + HTTP Doer + // TrialTimeout bounds the trial; zero means DefaultTrialTimeout. + TrialTimeout time.Duration + // Now is the clock; nil means time.Now. + Now func() time.Time + + mu sync.Mutex + slot string // the booted slot, read at Boot + running string // the kind of OS job in flight, "" for none +} + +// record is state.json. It is written by the slot that switches and read by +// the slot that boots next, which can be either release, so its fields only +// ever grow. +type record struct { + // Installed is what this package last wrote into a slot that is not the + // booted one. Nil when the other slot holds nothing of ours. + Installed *installed `json:"installed,omitempty"` + // Attempt is the last install attempt, for one attempt per target per + // night. + Attempt *attempt `json:"attempt,omitempty"` + // Switch is the last switch, for one switch per target per night and for + // the outcome the next boot records. + Switch *switched `json:"switch,omitempty"` + // Last is the last outcome, for the report and the admin notification. + Last *protocol.OSOutcome `json:"last,omitempty"` +} + +type installed struct { + Version string `json:"version"` + Digest string `json:"digest"` + Slot string `json:"slot"` + At time.Time `json:"at"` +} + +type attempt struct { + Version string `json:"version"` + Digest string `json:"digest"` + Night time.Time `json:"night"` + Error string `json:"error,omitempty"` +} + +type switched struct { + Version string `json:"version"` + FromVersion string `json:"from_version"` + From string `json:"from"` + To string `json:"to"` + Night time.Time `json:"night"` + At time.Time `json:"at"` +} + +func (a *Applier) dir() string { + if a.Dir != "" { + return a.Dir + } + return DefaultDir +} + +func (a *Applier) now() time.Time { + if a.Now != nil { + return a.Now() + } + return time.Now() +} + +func (a *Applier) recordPath() string { return filepath.Join(a.dir(), "state.json") } + +// TrialMarker is the file that says a slot is on trial. moose-os-trial.service +// reads the same path, so its name is part of the image. +func TrialMarker(dir, slot string) string { return filepath.Join(dir, "trial-"+slot) } + +func (a *Applier) load() (record, error) { + var r record + b, err := os.ReadFile(a.recordPath()) + if errors.Is(err, os.ErrNotExist) { + return r, nil + } + if err != nil { + return r, err + } + if err := json.Unmarshal(b, &r); err != nil { + return record{}, fmt.Errorf("read %s: %w", a.recordPath(), err) + } + return r, nil +} + +// save writes the record atomically and syncs it, because the next reader may +// be the other slot after a reboot. +func (a *Applier) save(r record) error { + if err := os.MkdirAll(a.dir(), 0o700); err != nil { + return err + } + b, err := json.MarshalIndent(r, "", " ") + if err != nil { + return err + } + return writeSynced(a.recordPath(), b) +} + +func writeSynced(path string, b []byte) error { + tmp := path + ".tmp" + f, err := os.OpenFile(tmp, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, 0o600) + if err != nil { + return err + } + if _, err := f.Write(b); err != nil { + f.Close() + return err + } + if err := f.Sync(); err != nil { + f.Close() + return err + } + if err := f.Close(); err != nil { + return err + } + if err := os.Rename(tmp, path); err != nil { + return err + } + d, err := os.Open(filepath.Dir(path)) + if err != nil { + return err + } + defer d.Close() + return d.Sync() +} + +func other(slot string) string { + switch slot { + case "A": + return "B" + case "B": + return "A" + } + return "" +} + +// Running reports the OS release and booted slot. +func (a *Applier) Running() (string, string) { + a.mu.Lock() + defer a.mu.Unlock() + return a.Version, a.slot +} + +// Floor reads the running brain's minimum_host_agent. +func (a *Applier) Floor() string { + if a.FloorFile == "" { + return "" + } + b, err := os.ReadFile(a.FloorFile) + if err != nil { + return "" + } + return strings.TrimSpace(string(b)) +} + +// Last is the last recorded outcome, or nil. +func (a *Applier) Last() *protocol.OSOutcome { + r, err := a.load() + if err != nil { + return nil + } + return r.Last +} + +// Boot reads the booted slot and decides what this boot is (see the package +// comment). It returns at once; a trial runs in the background. +func (a *Applier) Boot(ctx context.Context) { + st, err := a.RAUC.Status(ctx) + if err != nil { + slog.Error("os update: cannot read the slots; this box will not update its OS until it can", "err", err) + return + } + a.mu.Lock() + a.slot = st.Booted + a.mu.Unlock() + booted, oth := st.Booted, other(st.Booted) + + if _, err := os.Stat(TrialMarker(a.dir(), booted)); err == nil { + slog.Info("os update: this boot is on trial; waiting for the brain before marking the slot good", + "os", a.Version, "slot", booted) + go a.trial(booted) + return + } + + // Not on trial: the slot is good. Marked at once, so an unrelated problem + // later in this boot never sends the box back to the other slot. + if err := a.RAUC.Mark(ctx, "good", "booted"); err != nil { + slog.Error("os update: could not mark the booted slot good", "err", err, "slot", booted) + } + + if oth == "" { + return + } + if _, err := os.Stat(TrialMarker(a.dir(), oth)); err != nil { + return + } + // The other slot was on trial and the box is back here: it reverted. + r, err := a.load() + if err != nil { + slog.Error("os update: cannot read the record", "err", err) + } + out := &protocol.OSOutcome{Outcome: protocol.OSOutcomeReverted, From: a.Version, At: a.now().UTC().Format(time.RFC3339)} + if r.Switch != nil { + out.Version = r.Switch.Version + out.From = r.Switch.FromVersion + out.ID = outcomeID(r.Switch) + } else { + out.ID = "os-reverted-" + a.now().UTC().Format("20060102T150405Z") + } + if err := a.RAUC.Mark(ctx, "bad", "other"); err != nil { + slog.Error("os update: could not mark the failed slot bad", "err", err, "slot", oth) + } + r.Last = out + r.Installed = nil + if err := a.save(r); err != nil { + slog.Error("os update: could not write the record", "err", err) + } + if err := os.Remove(TrialMarker(a.dir(), oth)); err != nil { + slog.Error("os update: could not remove the trial marker", "err", err, "slot", oth) + } + slog.Warn("os update: the new slot did not come up healthy; the box went back to the old slot", + "os", out.Version, "slot", booted) +} + +func outcomeID(s *switched) string { + return fmt.Sprintf("os-%s-%d", s.Version, s.At.Unix()) +} + +// trial waits for the brain, then keeps or gives up the slot. +func (a *Applier) trial(slot string) { + timeout := a.TrialTimeout + if timeout <= 0 { + timeout = DefaultTrialTimeout + } + ctx, cancel := context.WithTimeout(context.Background(), timeout) + err := a.Healthy(ctx) + cancel() + bg := context.Background() + if err != nil { + slog.Error("os update: the new slot is not healthy; marking it bad and rebooting to the old slot", + "err", err, "os", a.Version, "slot", slot) + if err := a.RAUC.Mark(bg, "bad", "booted"); err != nil { + slog.Error("os update: could not mark the slot bad; rebooting anyway, GRUB skips a slot still on trial", "err", err) + } + a.reboot() + return + } + if err := a.RAUC.Mark(bg, "good", "booted"); err != nil { + // Without the mark GRUB skips this slot on the next boot. Reboot now, + // inside the window, rather than leave a slot that will revert at + // some random later reboot. + slog.Error("os update: could not mark the new slot good; rebooting to the old slot", "err", err, "slot", slot) + a.reboot() + return + } + // The marker goes first: once the slot is good, nothing may reboot it + // away, and the image's timer reboots any slot that still has one. + if err := os.Remove(TrialMarker(a.dir(), slot)); err != nil { + slog.Error("os update: could not remove the trial marker", "err", err, "slot", slot) + } + r, err := a.load() + if err != nil { + slog.Error("os update: cannot read the record", "err", err) + } + out := &protocol.OSOutcome{Outcome: protocol.OSOutcomeGood, Version: a.Version, At: a.now().UTC().Format(time.RFC3339)} + if r.Switch != nil { + out.From = r.Switch.FromVersion + out.ID = outcomeID(r.Switch) + } else { + out.ID = "os-" + a.Version + "-" + a.now().UTC().Format("20060102T150405Z") + } + r.Last = out + r.Installed = nil + if err := a.save(r); err != nil { + slog.Error("os update: could not write the record", "err", err) + } + slog.Info("os update: the new slot is healthy and marked good", "os", a.Version, "slot", slot) +} + +func (a *Applier) reboot() { + var err error + if a.Reboot != nil { + err = a.Reboot() + } else { + _, err = run(context.Background(), "systemctl", "reboot") + } + if err != nil { + slog.Error("os update: reboot failed", "err", err) + } +} + +// Apply is the loop's call (updatetarget.OSApplier). +func (a *Applier) Apply(rel updatetarget.OSRelease, open bool, night time.Time) updatetarget.OSDecision { + a.mu.Lock() + slot, running := a.slot, a.running + a.mu.Unlock() + switch running { + case protocol.JobKindOSInstall: + return updatetarget.OSDecision{State: protocol.OSUpdateInstalling} + case protocol.JobKindOSSwitch: + return updatetarget.OSDecision{State: protocol.OSUpdateRebooting} + } + if slot == "" { + return updatetarget.OSDecision{State: protocol.OSUpdateUnsupported, Detail: "the booted slot is not known"} + } + r, err := a.load() + if err != nil { + return updatetarget.OSDecision{State: protocol.OSUpdateFailed, Detail: err.Error()} + } + if s := r.Switch; s != nil && s.Version == rel.Version && s.Night.Equal(night) { + return updatetarget.OSDecision{State: protocol.OSUpdateHeld, + Detail: "this release was already tried tonight; the next window tries again"} + } + if in := r.Installed; in != nil && in.Version == rel.Version && in.Digest == rel.BundleSHA256 && in.Slot == other(slot) { + if !open { + return updatetarget.OSDecision{State: protocol.OSUpdateInstalled} + } + return a.start(protocol.JobKindOSSwitch, switchMaxDuration, protocol.OSUpdateRebooting, func(ctx context.Context) error { + return a.doSwitch(ctx, rel, night) + }) + } + if at := r.Attempt; at != nil && at.Version == rel.Version && at.Digest == rel.BundleSHA256 && at.Night.Equal(night) && at.Error != "" { + return updatetarget.OSDecision{State: protocol.OSUpdateFailed, Detail: at.Error} + } + return a.start(protocol.JobKindOSInstall, installMaxDuration, protocol.OSUpdateInstalling, func(ctx context.Context) error { + return a.doInstall(ctx, rel, night) + }) +} + +func (a *Applier) start(kind string, d time.Duration, state string, fn func(ctx context.Context) error) updatetarget.OSDecision { + a.mu.Lock() + a.running = kind + a.mu.Unlock() + id, err := a.Jobs.StartJob(kind, d, func(ctx context.Context) error { + defer func() { + a.mu.Lock() + a.running = "" + a.mu.Unlock() + }() + return fn(ctx) + }) + if err != nil { + a.mu.Lock() + a.running = "" + a.mu.Unlock() + if errors.Is(err, ErrBusy) { + return updatetarget.OSDecision{State: protocol.OSUpdateWaiting, Detail: "another job is running; the next check tries again"} + } + return updatetarget.OSDecision{State: protocol.OSUpdateFailed, Detail: err.Error()} + } + return updatetarget.OSDecision{State: state, JobID: id} +} + +// doInstall downloads, checks and installs rel into the other slot. +func (a *Applier) doInstall(ctx context.Context, rel updatetarget.OSRelease, night time.Time) (err error) { + _, slot := a.Running() + target := other(slot) + defer func() { + r, lerr := a.load() + if lerr != nil { + slog.Error("os update: cannot read the record", "err", lerr) + } + r.Attempt = &attempt{Version: rel.Version, Digest: rel.BundleSHA256, Night: night} + if err != nil { + r.Attempt.Error = err.Error() + slog.Error("os update: install failed; the running slot is untouched", "err", err, "os", rel.Version, "slot", target) + } else { + r.Installed = &installed{Version: rel.Version, Digest: rel.BundleSHA256, Slot: target, At: a.now().UTC()} + } + if serr := a.save(r); serr != nil { + slog.Error("os update: could not write the record", "err", serr) + if err == nil { + err = serr + } + } + }() + + if err := os.MkdirAll(a.dir(), 0o700); err != nil { + return err + } + path := filepath.Join(a.dir(), "bundle.raucb") + defer os.Remove(path) + slog.Info("os update: downloading the bundle", "os", rel.Version, "slot", target, "url", updatetarget.RedactURL(rel.BundleURL)) + if err := a.download(ctx, rel, path); err != nil { + return err + } + slog.Info("os update: bundle downloaded and its digest matches", "os", rel.Version, "digest", rel.BundleSHA256, + "url", updatetarget.RedactURL(rel.BundleURL)) + // The other slot is marked bad while RAUC writes it, so a reboot in the + // middle never boots half a slot. + if err := a.RAUC.Install(ctx, path); err != nil { + return fmt.Errorf("rauc install: %w", err) + } + st, err := a.RAUC.Status(ctx) + if err != nil { + return err + } + if got := st.Slots[target].Version; got != rel.Version { + return fmt.Errorf("after the install slot %s holds version %q, not %s", target, got, rel.Version) + } + slog.Info("os update: installed into the other slot; it waits for the update window", + "os", rel.Version, "slot", target) + return nil +} + +// download fetches the bundle to path and checks its sha256. The file is +// written as .part and renamed only once its digest matches, so RAUC never sees +// a bundle the answer did not name. +func (a *Applier) download(ctx context.Context, rel updatetarget.OSRelease, path string) error { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, rel.BundleURL, nil) + if err != nil { + return fmt.Errorf("download %s: %w", updatetarget.RedactURL(rel.BundleURL), updatetarget.CauseOf(err)) + } + client := a.HTTP + if client == nil { + client = &http.Client{} + } + resp, err := client.Do(req) + if err != nil { + return fmt.Errorf("download %s: %w", updatetarget.RedactURL(rel.BundleURL), updatetarget.CauseOf(err)) + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + return fmt.Errorf("download %s: HTTP %d", updatetarget.RedactURL(rel.BundleURL), resp.StatusCode) + } + part := path + ".part" + defer os.Remove(part) + f, err := os.OpenFile(part, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, 0o600) + if err != nil { + return err + } + h := sha256.New() + _, err = io.Copy(io.MultiWriter(f, h), resp.Body) + if cerr := f.Close(); err == nil { + err = cerr + } + if err != nil { + return fmt.Errorf("download %s: %w", updatetarget.RedactURL(rel.BundleURL), err) + } + if got := hex.EncodeToString(h.Sum(nil)); got != rel.BundleSHA256 { + return fmt.Errorf("the bundle's sha256 is %s, not the %s the update target names; refusing it", got, rel.BundleSHA256) + } + return os.Rename(part, path) +} + +// doSwitch makes the installed slot the next boot and reboots. +func (a *Applier) doSwitch(ctx context.Context, rel updatetarget.OSRelease, night time.Time) error { + _, slot := a.Running() + target := other(slot) + st, err := a.RAUC.Status(ctx) + if err != nil { + return err + } + if got := st.Slots[target].Version; got != rel.Version { + return fmt.Errorf("slot %s holds %q, not %s; not switching", target, got, rel.Version) + } + r, err := a.load() + if err != nil { + return err + } + // Recorded before anything moves, so a failure below still counts as + // tonight's one attempt. + r.Switch = &switched{Version: rel.Version, FromVersion: a.Version, From: slot, To: target, Night: night, At: a.now().UTC()} + if err := a.save(r); err != nil { + return err + } + undo := func(cause error) error { + if err := a.RAUC.Mark(context.Background(), "active", "booted"); err != nil { + slog.Error("os update: could not put the booted slot first again", "err", err) + } + os.Remove(TrialMarker(a.dir(), target)) + return cause + } + if err := writeSynced(TrialMarker(a.dir(), target), []byte(rel.Version+"\n")); err != nil { + return err + } + if err := a.RAUC.Mark(ctx, "active", "other"); err != nil { + return undo(fmt.Errorf("rauc mark-active: %w", err)) + } + // RAUC's GRUB backend puts the slot first with OK=1 and TRY=0. A stale + // TRY=1 would make GRUB skip the new slot and the update would look like + // a failed boot (#570). Check it rather than trust it. + env, err := a.RAUC.GrubEnvVars(ctx) + if err != nil { + return undo(err) + } + if order := strings.Fields(env["ORDER"]); len(order) == 0 || order[0] != target || env[target+"_OK"] != "1" || env[target+"_TRY"] != "0" { + return undo(fmt.Errorf("after mark-active the grubenv is ORDER=%q %s_OK=%q %s_TRY=%q; not rebooting", + env["ORDER"], target, env[target+"_OK"], target, env[target+"_TRY"])) + } + slog.Info("os update: switching slots and rebooting", "os", rel.Version, "slot", target) + a.reboot() + return nil +} + +// Peek reports what the box is doing about rel right now, from the job in +// flight and the record, without starting anything. The update loop decides +// once per tick (every 15 minutes); a reader in between would otherwise see +// "installing" long after the install ended. ok is false when Peek has nothing +// newer than the loop's last decision. +func (a *Applier) Peek(rel updatetarget.OSRelease) (state, detail string, ok bool) { + a.mu.Lock() + slot, running := a.slot, a.running + a.mu.Unlock() + switch running { + case protocol.JobKindOSInstall: + return protocol.OSUpdateInstalling, "", true + case protocol.JobKindOSSwitch: + return protocol.OSUpdateRebooting, "", true + } + r, err := a.load() + if err != nil { + return "", "", false + } + if in := r.Installed; in != nil && in.Version == rel.Version && in.Digest == rel.BundleSHA256 && in.Slot == other(slot) { + return protocol.OSUpdateInstalled, "", true + } + if at := r.Attempt; at != nil && at.Version == rel.Version && at.Digest == rel.BundleSHA256 && at.Error != "" { + return protocol.OSUpdateFailed, at.Error, true + } + return "", "", false +} diff --git a/internal/hostagent/osupdate/osupdate_test.go b/internal/hostagent/osupdate/osupdate_test.go new file mode 100644 index 00000000..f4e9ed95 --- /dev/null +++ b/internal/hostagent/osupdate/osupdate_test.go @@ -0,0 +1,351 @@ +package osupdate + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "errors" + "net/http" + "net/http/httptest" + "os" + "strings" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/onmoose/os/internal/hostagent/updatetarget" + "github.com/onmoose/os/internal/protocol" +) + +// fakeRAUC is a two-slot box with a grubenv, in memory. +type fakeRAUC struct { + mu sync.Mutex + booted string + versions map[string]string + env map[string]string + marks []string + installErr error + installed []byte + // staleTry makes mark-active leave the new slot's TRY at 1, the #570 case. + staleTry bool +} + +func newFakeRAUC(booted string) *fakeRAUC { + return &fakeRAUC{booted: booted, versions: map[string]string{}, + env: map[string]string{"ORDER": "A B", "A_OK": "1", "A_TRY": "0", "B_OK": "0", "B_TRY": "0"}} +} + +func (f *fakeRAUC) Status(context.Context) (Status, error) { + f.mu.Lock() + defer f.mu.Unlock() + st := Status{Booted: f.booted, Slots: map[string]SlotInfo{}} + for _, s := range []string{"A", "B"} { + st.Slots[s] = SlotInfo{Version: f.versions[s], Good: f.env[s+"_OK"] == "1" && f.env[s+"_TRY"] == "0"} + } + return st, nil +} + +func (f *fakeRAUC) Install(_ context.Context, path string) error { + f.mu.Lock() + defer f.mu.Unlock() + if f.installErr != nil { + return f.installErr + } + b, err := os.ReadFile(path) + if err != nil { + return err + } + f.installed = b + o := other(f.booted) + f.versions[o] = strings.TrimSpace(string(b)) + f.env[o+"_OK"], f.env[o+"_TRY"] = "0", "0" + return nil +} + +func (f *fakeRAUC) Mark(_ context.Context, state, which string) error { + f.mu.Lock() + defer f.mu.Unlock() + slot := f.booted + if which == "other" { + slot = other(f.booted) + } + f.marks = append(f.marks, state+":"+slot) + switch state { + case "good": + f.env[slot+"_OK"], f.env[slot+"_TRY"] = "1", "0" + case "bad": + f.env[slot+"_OK"], f.env[slot+"_TRY"] = "0", "0" + case "active": + f.env[slot+"_OK"], f.env[slot+"_TRY"] = "1", "0" + if f.staleTry { + f.env[slot+"_TRY"] = "1" + } + f.env["ORDER"] = slot + " " + other(slot) + } + return nil +} + +func (f *fakeRAUC) GrubEnvVars(context.Context) (map[string]string, error) { + f.mu.Lock() + defer f.mu.Unlock() + m := map[string]string{} + for k, v := range f.env { + m[k] = v + } + return m, nil +} + +// syncJobs runs a job to the end before it returns, so a test reads the +// outcome right after Apply. +type syncJobs struct { + busy bool + errs []error +} + +func (j *syncJobs) StartJob(_ string, d time.Duration, fn func(ctx context.Context) error) (string, error) { + if j.busy { + return "", ErrBusy + } + ctx, cancel := context.WithTimeout(context.Background(), d) + defer cancel() + j.errs = append(j.errs, fn(ctx)) + return "j_1", nil +} + +type harness struct { + a *Applier + rauc *fakeRAUC + jobs *syncJobs + reboots atomic.Int32 + healthy error + rel updatetarget.OSRelease + night time.Time +} + +func newHarness(t *testing.T, booted string) *harness { + t.Helper() + body := "0.15.1\n" + sum := sha256.Sum256([]byte(body)) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + _, _ = w.Write([]byte(body)) + })) + t.Cleanup(srv.Close) + h := &harness{rauc: newFakeRAUC(booted), jobs: &syncJobs{}, night: time.Date(2026, 10, 3, 0, 0, 0, 0, time.UTC)} + h.rel = updatetarget.OSRelease{Version: "0.15.1", BundleURL: srv.URL + "/b.raucb", BundleSHA256: hex.EncodeToString(sum[:])} + h.a = &Applier{ + RAUC: h.rauc, Jobs: h.jobs, Version: "0.15.0", Dir: t.TempDir(), + Healthy: func(context.Context) error { return h.healthy }, + Reboot: func() error { h.reboots.Add(1); return nil }, + } + h.a.Boot(context.Background()) + return h +} + +func TestNormalBootIsMarkedGoodAtOnce(t *testing.T) { + h := newHarness(t, "A") + if len(h.rauc.marks) != 1 || h.rauc.marks[0] != "good:A" { + t.Fatalf("a boot not on trial must be marked good at once, marks %v", h.rauc.marks) + } + if v, s := h.a.Running(); v != "0.15.0" || s != "A" { + t.Fatalf("Running() = %s %s", v, s) + } +} + +func TestInstallThenSwitch(t *testing.T) { + h := newHarness(t, "A") + d := h.a.Apply(h.rel, false, h.night) + if d.State != protocol.OSUpdateInstalling || h.jobs.errs[0] != nil { + t.Fatalf("install: %+v %v", d, h.jobs.errs) + } + if h.rauc.versions["B"] != "0.15.1" { + t.Fatalf("slot B holds %q", h.rauc.versions["B"]) + } + if _, err := os.Stat(h.a.dir() + "/bundle.raucb"); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("the downloaded bundle was not removed: %v", err) + } + if d := h.a.Apply(h.rel, false, h.night); d.State != protocol.OSUpdateInstalled { + t.Fatalf("outside the window an installed slot must wait, got %+v", d) + } + if h.reboots.Load() != 0 { + t.Fatal("rebooted outside the window") + } + d = h.a.Apply(h.rel, true, h.night) + if d.State != protocol.OSUpdateRebooting || h.jobs.errs[1] != nil || h.reboots.Load() != 1 { + t.Fatalf("switch: %+v %v reboots=%d", d, h.jobs.errs, h.reboots.Load()) + } + if h.rauc.env["ORDER"] != "B A" || h.rauc.env["B_TRY"] != "0" { + t.Fatalf("grubenv after the switch: %v", h.rauc.env) + } + if _, err := os.Stat(TrialMarker(h.a.dir(), "B")); err != nil { + t.Fatalf("no trial marker for B: %v", err) + } + if d := h.a.Apply(h.rel, true, h.night); d.State != protocol.OSUpdateHeld { + t.Fatalf("a second switch the same night must be held, got %+v", d) + } +} + +func TestWrongDigestIsRefusedAndNothingInstalled(t *testing.T) { + h := newHarness(t, "A") + h.rel.BundleSHA256 = strings.Repeat("0", 64) + h.a.Apply(h.rel, true, h.night) + if h.jobs.errs[0] == nil || !strings.Contains(h.jobs.errs[0].Error(), "sha256") { + t.Fatalf("want a digest refusal, got %v", h.jobs.errs[0]) + } + if h.rauc.installed != nil { + t.Fatal("RAUC saw a bundle the answer did not name") + } + if d := h.a.Apply(h.rel, true, h.night); d.State != protocol.OSUpdateFailed || len(h.jobs.errs) != 1 { + t.Fatalf("a failed install must not be retried the same night: %+v, %d jobs", d, len(h.jobs.errs)) + } + if d := h.a.Apply(h.rel, true, h.night.Add(24*time.Hour)); d.State != protocol.OSUpdateInstalling { + t.Fatalf("the next night tries again, got %+v", d) + } +} + +func TestStaleTryStopsTheSwitch(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.rauc.staleTry = true + h.a.Apply(h.rel, true, h.night) + if h.jobs.errs[1] == nil || h.reboots.Load() != 0 { + t.Fatalf("a TRY=1 after mark-active must stop the reboot: %v, reboots %d", h.jobs.errs[1], h.reboots.Load()) + } + if h.rauc.env["ORDER"] != "A B" { + t.Fatalf("the booted slot must be put first again, ORDER=%q", h.rauc.env["ORDER"]) + } + if _, err := os.Stat(TrialMarker(h.a.dir(), "B")); !errors.Is(err, os.ErrNotExist) { + t.Fatal("the trial marker was left behind") + } +} + +func TestBusyLockWaits(t *testing.T) { + h := newHarness(t, "A") + h.jobs.busy = true + if d := h.a.Apply(h.rel, true, h.night); d.State != protocol.OSUpdateWaiting { + t.Fatalf("got %+v", d) + } +} + +// reboot simulates the box coming back on slot `booted`, with the record and +// markers on the shared state partition. +func (h *harness) reboot(t *testing.T, booted, version string) *Applier { + t.Helper() + h.rauc.mu.Lock() + h.rauc.booted = booted + h.rauc.marks = nil + h.rauc.mu.Unlock() + a := &Applier{RAUC: h.rauc, Jobs: h.jobs, Version: version, Dir: h.a.dir(), + Healthy: func(context.Context) error { return h.healthy }, + Reboot: func() error { h.reboots.Add(1); return nil }} + return a +} + +func waitFor(t *testing.T, cond func() bool) { + t.Helper() + for i := 0; i < 200; i++ { + if cond() { + return + } + time.Sleep(5 * time.Millisecond) + } + t.Fatal("timed out") +} + +func TestTrialGood(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + b := h.reboot(t, "B", "0.15.1") + b.Boot(context.Background()) + waitFor(t, func() bool { return b.Last() != nil }) + last := b.Last() + if last.Outcome != protocol.OSOutcomeGood || last.Version != "0.15.1" || last.From != "0.15.0" || last.ID == "" { + t.Fatalf("outcome %+v", last) + } + if _, err := os.Stat(TrialMarker(b.dir(), "B")); !errors.Is(err, os.ErrNotExist) { + t.Fatal("the trial marker is still there") + } + if h.rauc.env["B_OK"] != "1" || h.rauc.env["B_TRY"] != "0" { + t.Fatalf("B not marked good: %v", h.rauc.env) + } +} + +func TestTrialUnhealthyRebootsAndOldSlotRecordsRevert(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + h.healthy = errors.New("brain never answered") + b := h.reboot(t, "B", "0.15.1") + before := h.reboots.Load() + b.Boot(context.Background()) + waitFor(t, func() bool { return h.reboots.Load() > before }) + h.rauc.mu.Lock() + bOK := h.rauc.env["B_OK"] + h.rauc.mu.Unlock() + if bOK != "0" { + t.Fatalf("B must be marked bad: %v", h.rauc.env) + } + // GRUB boots A again. + a := h.reboot(t, "A", "0.15.0") + a.Boot(context.Background()) + last := a.Last() + if last == nil || last.Outcome != protocol.OSOutcomeReverted || last.Version != "0.15.1" { + t.Fatalf("outcome %+v", last) + } + if _, err := os.Stat(TrialMarker(a.dir(), "B")); !errors.Is(err, os.ErrNotExist) { + t.Fatal("the trial marker is still there") + } + if d := a.Apply(h.rel, true, h.night); d.State != protocol.OSUpdateHeld { + t.Fatalf("after a revert the same release waits for the next night, got %+v", d) + } +} + +func TestFloorFile(t *testing.T) { + dir := t.TempDir() + a := &Applier{FloorFile: dir + "/minimum-host-agent"} + if a.Floor() != "" { + t.Fatal("a missing floor file must read as no floor") + } + if err := os.WriteFile(a.FloorFile, []byte("0.4.0\n"), 0o644); err != nil { + t.Fatal(err) + } + if a.Floor() != "0.4.0" { + t.Fatalf("got %q", a.Floor()) + } +} + +func TestParseStatusAndGrubEnv(t *testing.T) { + js := `{"compatible":"moose-hosted-x86_64","booted":"B","boot_primary":"rootfs.1","slots":[ + {"rootfs.0":{"class":"rootfs","bootname":"A","state":"inactive","boot_status":"good"}}, + {"rootfs.1":{"class":"rootfs","bootname":"B","state":"booted","boot_status":"bad", + "slot_status":{"bundle":{"compatible":"moose-hosted-x86_64","version":"0.15.1"},"status":"ok"}}}]}` + st, err := parseStatus([]byte(js)) + if err != nil { + t.Fatal(err) + } + if st.Booted != "B" || st.Slots["B"].Version != "0.15.1" || !st.Slots["A"].Good || st.Slots["B"].Good { + t.Fatalf("got %+v", st) + } + env := parseGrubEnv([]byte("# GRUB Environment Block\nORDER=B A\nB_TRY=0\n")) + if env["ORDER"] != "B A" || env["B_TRY"] != "0" { + t.Fatalf("got %v", env) + } +} + +func TestPeek(t *testing.T) { + h := newHarness(t, "A") + if _, _, ok := h.a.Peek(h.rel); ok { + t.Fatal("nothing done yet, Peek must have nothing to say") + } + bad := h.rel + bad.BundleSHA256 = strings.Repeat("0", 64) + h.a.Apply(bad, false, h.night) + if st, d, ok := h.a.Peek(bad); !ok || st != protocol.OSUpdateFailed || !strings.Contains(d, "sha256") { + t.Fatalf("after a wrong digest: %s %q %v", st, d, ok) + } + h.a.Apply(h.rel, false, h.night) + if st, _, ok := h.a.Peek(h.rel); !ok || st != protocol.OSUpdateInstalled { + t.Fatalf("after the install: %s %v", st, ok) + } +} diff --git a/internal/hostagent/osupdate/rauc.go b/internal/hostagent/osupdate/rauc.go new file mode 100644 index 00000000..355f727a --- /dev/null +++ b/internal/hostagent/osupdate/rauc.go @@ -0,0 +1,142 @@ +package osupdate + +import ( + "bufio" + "bytes" + "context" + "encoding/json" + "fmt" + "os/exec" + "strings" +) + +// Status is what `rauc status` reports that this package uses. +type Status struct { + // Booted is the bootname of the slot the box booted ("A" or "B"). + Booted string + // Slots maps a bootname to what RAUC knows about that slot. + Slots map[string]SlotInfo +} + +// SlotInfo is one slot's RAUC status. +type SlotInfo struct { + // Version is the bundle version RAUC installed into the slot, or "" for a + // slot RAUC never wrote (slot A as the image shipped it). + Version string + // Good is RAUC's boot status for the slot: in the bootloader's order, + // _OK=1 and _TRY=0. + Good bool +} + +// CLIRAUC drives RAUC through its command line, which talks to the +// rauc-service D-Bus daemon. GrubEnv is the grubenv RAUC's GRUB backend writes +// (/etc/rauc/system.conf); empty means DefaultGrubEnv. +type CLIRAUC struct { + GrubEnv string +} + +// DefaultGrubEnv is where the hosted image keeps the grubenv (BUILD.md # 1b). +const DefaultGrubEnv = "/efi/grub/grubenv" + +func run(ctx context.Context, name string, args ...string) ([]byte, error) { + cmd := exec.CommandContext(ctx, name, args...) + var stderr bytes.Buffer + cmd.Stderr = &stderr + out, err := cmd.Output() + if err != nil { + return out, fmt.Errorf("%s %s: %w: %s", name, strings.Join(args, " "), err, strings.TrimSpace(stderr.String())) + } + return out, nil +} + +// raucStatusJSON is the part of `rauc status --detailed --output-format=json` +// read here. RAUC writes slots as a list of one-key objects, keyed by the slot +// name (rootfs.0). +type raucStatusJSON struct { + Booted string `json:"booted"` + Slots []map[string]raucSlotJSON `json:"slots"` +} + +type raucSlotJSON struct { + Bootname string `json:"bootname"` + BootStatus string `json:"boot_status"` + SlotStatus *struct { + Bundle *struct { + Version string `json:"version"` + } `json:"bundle"` + } `json:"slot_status"` +} + +// parseStatus reads RAUC's JSON status. +func parseStatus(b []byte) (Status, error) { + var j raucStatusJSON + if err := json.Unmarshal(b, &j); err != nil { + return Status{}, fmt.Errorf("read rauc status: %w", err) + } + st := Status{Booted: j.Booted, Slots: map[string]SlotInfo{}} + for _, m := range j.Slots { + for _, s := range m { + if s.Bootname == "" { + continue + } + info := SlotInfo{Good: s.BootStatus == "good"} + if s.SlotStatus != nil && s.SlotStatus.Bundle != nil { + info.Version = s.SlotStatus.Bundle.Version + } + st.Slots[s.Bootname] = info + } + } + if st.Booted == "" { + return Status{}, fmt.Errorf("rauc status names no booted slot") + } + return st, nil +} + +// Status reads RAUC's view of the slots. +func (r CLIRAUC) Status(ctx context.Context) (Status, error) { + out, err := run(ctx, "rauc", "status", "--detailed", "--output-format=json") + if err != nil { + return Status{}, err + } + return parseStatus(out) +} + +// Install writes the bundle into the other slot. With activate-installed=false +// in system.conf it does not change the boot order: the switch is a separate +// step, inside the window. +func (r CLIRAUC) Install(ctx context.Context, path string) error { + _, err := run(ctx, "rauc", "install", path) + return err +} + +// Mark runs `rauc status mark- `: state is good, bad or active; +// which is booted or other. +func (r CLIRAUC) Mark(ctx context.Context, state, which string) error { + _, err := run(ctx, "rauc", "status", "mark-"+state, which) + return err +} + +// GrubEnvVars reads the grubenv as key=value pairs. +func (r CLIRAUC) GrubEnvVars(ctx context.Context) (map[string]string, error) { + path := r.GrubEnv + if path == "" { + path = DefaultGrubEnv + } + out, err := run(ctx, "grub-editenv", path, "list") + if err != nil { + return nil, err + } + return parseGrubEnv(out), nil +} + +func parseGrubEnv(b []byte) map[string]string { + m := map[string]string{} + sc := bufio.NewScanner(bytes.NewReader(b)) + for sc.Scan() { + k, v, ok := strings.Cut(sc.Text(), "=") + if ok { + m[k] = v + } + } + return m +} diff --git a/internal/hostagent/updatetarget/http.go b/internal/hostagent/updatetarget/http.go index 46f0247b..765bf51a 100644 --- a/internal/hostagent/updatetarget/http.go +++ b/internal/hostagent/updatetarget/http.go @@ -147,6 +147,17 @@ type wireTarget struct { // Window is optional. An answer that leaves it out has no opinion about // when this box may update, and the box then keeps its own setting. Window string `json:"window"` + // OS is optional too: the OS releases on the way to this box's OS target, + // oldest first (os.go, UPDATES.md # 1). Left out, the answer has no + // opinion about the OS. + OS []wireOS `json:"os"` +} + +// wireOS is one OS release in the answer. Part of the same contract. +type wireOS struct { + Version string `json:"version"` + BundleURL string `json:"bundle_url"` + BundleSHA256 string `json:"bundle_sha256"` } // Target reads the update-target URL. @@ -198,7 +209,12 @@ func (s HTTPSource) Target(ctx context.Context) (Target, error) { if err := json.Unmarshal(b, &w); err != nil { return Target{}, fmt.Errorf("updatetarget: parse the answer from %s: %w", RedactURL(url), err) } + var osList []OSRelease + for _, o := range w.OS { + osList = append(osList, OSRelease{Version: o.Version, BundleURL: o.BundleURL, BundleSHA256: o.BundleSHA256}) + } return Target{ + OS: osList, Version: w.Version, BrainImage: w.BrainImage, BrainDigest: w.BrainDigest, diff --git a/internal/hostagent/updatetarget/loop.go b/internal/hostagent/updatetarget/loop.go index 92447ba5..65bab56a 100644 --- a/internal/hostagent/updatetarget/loop.go +++ b/internal/hostagent/updatetarget/loop.go @@ -10,6 +10,7 @@ import ( "time" "github.com/onmoose/os/internal/hostagent/controlplane" + "github.com/onmoose/os/internal/protocol" ) // PollInterval is how often the box asks its source what it should be running. @@ -42,6 +43,29 @@ type Applier interface { StartUpdate(brainRef, uiRef string) (jobID string, err error) } +// OSApplier is stream A's transaction (internal/hostagent/osupdate). Consumer- +// side interface. Nil on a box that cannot update its OS. +type OSApplier interface { + // Running is the OS release this box runs and the slot it booted. + Running() (version, slot string) + // Floor is the running control plane's minimum_host_agent, or "" when the + // box cannot read it. + Floor() string + // Apply moves the box towards rel: it installs into the other slot when + // that slot does not hold rel yet, and switches and reboots only when + // open is true. night names tonight's window, for the one attempt per + // target per night. It never blocks on the work: the work runs as a job. + Apply(rel OSRelease, open bool, night time.Time) OSDecision +} + +// OSDecision is what the applier did or is doing. State is one of the +// protocol.OSUpdate* values. +type OSDecision struct { + State string + Detail string + JobID string +} + // RunningPair reports the control-plane images the box is running right now, so // the loop can tell "the target moved" from "nothing to do". type RunningPair interface { @@ -100,6 +124,13 @@ type Loop struct { // Profile is the environment profile this loop runs on (ENVIRONMENT.md), // carried only so the logs say which one made a decision. Profile string + // OS applies stream A. Nil means this box cannot update its OS (the + // appliance until #564, a test), and the OS part of an answer is then + // reported as unsupported and never acted on. + OS OSApplier + // OSURLPrefix is the expected start of every bundle URL; empty means + // DefaultOSURLPrefix. + OSURLPrefix string // Interval between polls; zero means PollInterval. Interval time.Duration // Now and After exist so a test can drive a week of polling in @@ -118,6 +149,9 @@ type Loop struct { // that moves every night must not keep re-announcing a window that never // moved. lastWindow string + // lastOS dedupes stream A's lines, apart from lastQuiet for the same + // reason as lastWindow. + lastOS string // attempted is the target version this loop last started an update for, and // the night it did it in. One attempt per night per version: a failed // update reverts the box, so the next tick sees the same difference again, @@ -301,6 +335,14 @@ func (l *Loop) Tick(ctx context.Context) { return } + // The window is resolved first, before either stream is judged. An + // answer's window outranks the box's setting the moment the answer is read + // (UPDATES.md # 8.4), so a box that is already current still reports the + // hour it would update in, and stream A uses the same window whatever + // stream B's part says. The log line is deduped, so resolving it every tick + // costs nothing. + w, windowFrom := l.windowFor(t) + // Untrusted input, validated before anything is pulled. A refusal is loud // and repeated: an operator needs to see that a box has been declining its // target since Tuesday, and a source stuck on a bad answer is a fleet-wide @@ -313,20 +355,17 @@ func (l *Loop) Tick(ctx context.Context) { // // The refused answer is kept in the snapshot. Nothing acts on it — the // version and refs are what an operator needs to see to fix the source. - l.record(OutcomeRefused, t, l.window(), l.windowFrom(), err) + l.record(OutcomeRefused, t, w, windowFrom, err) l.quiet(slog.LevelError, "refused:"+t.Version+":"+t.BrainImage+":"+t.UIImage, "update target: refusing the answer; nothing pulled, box unchanged", "err", err, "brain", t.Version, "image", t.BrainImage) + // Stream A is judged on its own (os.go): a refused control-plane part + // holds back only stream B, the same way a bad OS part holds back + // only stream A. + l.tickOS(t, w, false) return } - // The window is resolved here, before the compare, not further down on the - // apply path. An answer's window outranks the box's setting the moment the - // answer is read (UPDATES.md # 8.4), so a box that is already current still - // reports the hour it would update in. The log line is deduped, so resolving - // it every tick costs nothing. - w, windowFrom := l.windowFor(t) - brain, ui, err := l.Current.Running() if err != nil { // The box could not read its own declaration, so it cannot tell whether @@ -339,14 +378,26 @@ func (l *Loop) Tick(ctx context.Context) { l.record(OutcomeUnreachable, t, w, windowFrom, err) l.quiet(slog.LevelWarn, "running-err:"+err.Error(), "update target: cannot read what this box is running", "err", err) + // Stream B cannot tell whether it has work; stream A can, and goes on. + l.tickOS(t, w, false) return } l.record(OutcomeOK, t, w, windowFrom, nil) + cpBusy := l.applyControlPlane(t, w, brain, ui) + l.tickOS(t, w, cpBusy) +} + +// applyControlPlane is stream B's half of a tick: compare the running pair +// with the target and start the update in the window. It reports whether +// stream B holds the window right now (it started an update this tick, or +// could not start one because a job runs), so stream A waits: stream B goes +// first in the window and the OS reboot goes last (UPDATES.md # 7). +func (l *Loop) applyControlPlane(t Target, w Window, brain, ui string) (busy bool) { if brain == t.BrainImage && ui == t.UIImage { // The overwhelmingly common case. No pull, no work, and one line the // first time it is true. l.quiet(slog.LevelInfo, "current:"+t.Version, "update target: already on the target version", "brain", t.Version) - return + return false } if !l.AutoApply || l.Applier == nil { @@ -355,19 +406,20 @@ func (l *Loop) Tick(ctx context.Context) { // slice; this is where the fact enters the box. l.quiet(slog.LevelInfo, "available:"+t.Version, "update target: a different control plane is available", "brain", t.Version, "image", t.BrainImage) - return + return false } now := l.now() if !w.Contains(now) { l.quiet(slog.LevelInfo, "holding:"+t.Version, "update target: holding a new version for the update window", "brain", t.Version, "step", "waiting", "zone", now.Location().String()) - return + return false } if night := occurrenceNight(w, now); l.attempted.version == t.Version && !night.After(l.attempted.night) { // Already tried this version tonight. If it failed it has already put - // the box back, and the next window is soon enough to try again. - return + // the box back, and the next window is soon enough to try again. It + // no longer holds the window, so stream A may go. + return false } l.attempted.version, l.attempted.night = t.Version, occurrenceNight(w, now) @@ -377,10 +429,66 @@ func (l *Loop) Tick(ctx context.Context) { // Includes the ordinary "a job is already running" refusal, which is why // this is a Warn and not an Error. slog.Warn("update target: could not start the update", "err", err, "brain", t.Version) - return + return true } slog.Info("update target: applying", "brain", t.Version, "image", t.BrainImage, "step", "accepted", "job_id", jobID) + return true +} + +// tickOS is stream A's half of a tick (UPDATES.md # 1): check the OS part, +// pick the release to install, and hand it to the OS applier with whether the +// window is open to it. The applier installs ahead of the window and switches +// and reboots only inside it. +func (l *Loop) tickOS(t Target, w Window, cpBusy bool) { + if l.OS == nil { + l.recordOS(OSSnapshot{State: protocol.OSUpdateUnsupported, Detail: "this host-agent cannot update the OS"}) + return + } + if len(t.OS) == 0 { + l.recordOS(OSSnapshot{State: protocol.OSUpdateNone}) + l.quietOS(slog.LevelInfo, "none", "os update: the answer names no OS release; staying on this OS") + return + } + prefix := l.OSURLPrefix + if prefix == "" { + prefix = DefaultOSURLPrefix + } + running, slot := l.OS.Running() + if err := ValidateOS(t.OS, prefix); err != nil { + l.recordOS(OSSnapshot{State: protocol.OSUpdateRefused, Detail: err.Error()}) + l.quietOS(slog.LevelError, "refused:"+err.Error(), "os update: refusing the OS part of the answer; nothing downloaded", + "err", err, "os", running) + return + } + rel, current, err := PickOS(running, l.OS.Floor(), t.OS) + if err != nil { + l.recordOS(OSSnapshot{State: protocol.OSUpdateRefused, Detail: err.Error()}) + l.quietOS(slog.LevelError, "refused:"+err.Error(), "os update: refusing the OS target; nothing downloaded", + "err", err, "os", running) + return + } + if current { + l.recordOS(OSSnapshot{State: protocol.OSUpdateCurrent, Target: &rel}) + l.quietOS(slog.LevelInfo, "current:"+rel.Version, "os update: already on the target OS", "os", running, "slot", slot) + return + } + now := l.now() + open := w.Contains(now) && !cpBusy + d := l.OS.Apply(rel, open, occurrenceNight(w, now)) + l.recordOS(OSSnapshot{State: d.State, Detail: d.Detail, Target: &rel}) + l.quietOS(slog.LevelInfo, "decision:"+rel.Version+":"+d.State+":"+d.Detail, "os update: "+d.State, + "os", rel.Version, "digest", rel.BundleSHA256, "slot", slot, "job_id", d.JobID) +} + +// quietOS is quiet for stream A, with its own memory: the two streams change +// independently. +func (l *Loop) quietOS(level slog.Level, key, msg string, args ...any) { + if l.lastOS == key { + return + } + l.lastOS = key + slog.Log(context.Background(), level, msg, args...) } // Run ticks until ctx is done: once at the start, then on the jittered interval. diff --git a/internal/hostagent/updatetarget/os.go b/internal/hostagent/updatetarget/os.go new file mode 100644 index 00000000..3d10dd95 --- /dev/null +++ b/internal/hostagent/updatetarget/os.go @@ -0,0 +1,170 @@ +package updatetarget + +import ( + "errors" + "fmt" + "net/url" + "regexp" + "strings" + + "golang.org/x/mod/semver" +) + +// This file is stream A's half of the answer (UPDATES.md # 1, #563): the OS +// releases the box may install, and the rules for which one it picks. +// +// The answer names OS releases as a list, not one release: the newest patch of +// each minor, from the oldest one still supported up to the target, which is +// the last entry. A box never skips a minor, because only a minor release may +// change an on-disk format and the release in the other slot must always be +// able to read what this one wrote. +// +// **The OS part is checked on its own.** A bad OS part refuses stream A only; +// the control-plane pair in the same answer still applies. The two streams are +// separate transactions with separate rollbacks, so a mistake in one must not +// hold the other back. + +// OSRelease is one OS release in the answer. +type OSRelease struct { + // Version is the moose (OS) release, X.Y.Z. It is what the box compares + // with its own version, so unlike the control-plane version it is used. + Version string + // BundleURL is where the box downloads the RAUC bundle. It must start with + // the expected prefix (DefaultOSURLPrefix, or MOOSE_UPDATE_OS_URL_PREFIX). + BundleURL string + // BundleSHA256 pins the bundle's bytes. The box installs only a bundle + // whose sha256 is this one, on top of RAUC's signature check, so a leaked + // signer alone cannot make a box install anything (DECISIONS.md 2026-10-02). + BundleSHA256 string +} + +// DefaultOSURLPrefix is where published OS bundles live: the GitHub Release +// of each OS release (BUILD.md # 6). The digest pins the bytes; the prefix is +// the check that stops a well-formed answer from making the box fetch from +// anywhere at all, the same job the expected repositories do for the brain. +const DefaultOSURLPrefix = "https://github.com/onmoose/os/releases/download/" + +// ErrOSRefused wraps every reason the OS part of an answer is refused. +var ErrOSRefused = errors.New("updatetarget: OS part refused") + +var sha256Hex = regexp.MustCompile(`^[0-9a-f]{64}$`) + +// canonical turns "1.2.3" or "v1.2.3" into "v1.2.3", or "" when it is not a +// plain X.Y.Z. Pre-release and build suffixes are refused: an OS release is +// always a plain release (BUILD.md # Versioning). +func canonical(v string) string { + v = strings.TrimSpace(v) + if !strings.HasPrefix(v, "v") { + v = "v" + v + } + if !semver.IsValid(v) || semver.Canonical(v) != v || semver.Prerelease(v) != "" || semver.Build(v) != "" { + return "" + } + return v +} + +// ValidateOS is the boundary check on the OS part. It runs before anything is +// downloaded. prefix is the expected start of every bundle URL; empty means +// any http or https URL. +func ValidateOS(list []OSRelease, prefix string) error { + prev := "" + for i, r := range list { + v := canonical(r.Version) + if v == "" { + return fmt.Errorf("%w: entry %d: %q is not an X.Y.Z version", ErrOSRefused, i, r.Version) + } + if prev != "" && semver.Compare(prev, v) >= 0 { + return fmt.Errorf("%w: the releases are not in ascending order (%s after %s)", ErrOSRefused, r.Version, strings.TrimPrefix(prev, "v")) + } + prev = v + if !sha256Hex.MatchString(r.BundleSHA256) { + return fmt.Errorf("%w: %s: the bundle is not pinned to a sha256 digest", ErrOSRefused, r.Version) + } + u, err := url.Parse(r.BundleURL) + if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" { + return fmt.Errorf("%w: %s: the bundle URL is not an absolute http or https URL", ErrOSRefused, r.Version) + } + if prefix != "" && !strings.HasPrefix(r.BundleURL, prefix) { + return fmt.Errorf("%w: %s: the bundle URL %s does not start with %s", ErrOSRefused, r.Version, RedactURL(r.BundleURL), prefix) + } + } + return nil +} + +// line is a version's MAJOR.MINOR, as two numbers. +type line struct{ major, minor int } + +func lineOf(v string) line { + var l line + var patch int + _, _ = fmt.Sscanf(strings.TrimPrefix(v, "v"), "%d.%d.%d", &l.major, &l.minor, &patch) + return l +} + +func (a line) less(b line) bool { + return a.major < b.major || (a.major == b.major && a.minor < b.minor) +} + +// PickOS chooses what to install from a validated list (UPDATES.md # 1): +// +// - the target is the last entry; +// - target on a later minor: the first entry on a minor above the box's own +// (the next step; the rest waits for a later window); +// - target on the box's own minor, or one minor back: the target itself; +// - target further back, or below the running control plane's floor: refused. +// +// running is this box's OS version and floor the running control plane's +// minimum_host_agent ("" when the box cannot read it, and then there is no +// floor check). current is true when the box already runs the target. +func PickOS(running, floor string, list []OSRelease) (rel OSRelease, current bool, err error) { + if len(list) == 0 { + return OSRelease{}, false, fmt.Errorf("%w: the answer names no OS release", ErrOSRefused) + } + run := canonical(running) + if run == "" { + return OSRelease{}, false, fmt.Errorf("%w: this box's own version %q is not an X.Y.Z version", ErrOSRefused, running) + } + target := list[len(list)-1] + tv := canonical(target.Version) + if semver.Compare(tv, run) == 0 { + return target, true, nil + } + rl, tl := lineOf(run), lineOf(tv) + switch { + case rl.less(tl): + for _, r := range list { + if rl.less(lineOf(canonical(r.Version))) { + rel = r + break + } + } + case tl == rl: + rel = target + case oneLineBack(tl, rl, list): + rel = target + default: + return OSRelease{}, false, fmt.Errorf("%w: the target %s is more than one minor back from %s", ErrOSRefused, target.Version, running) + } + if f := canonical(floor); f != "" && semver.Compare(canonical(rel.Version), f) < 0 { + return OSRelease{}, false, fmt.Errorf("%w: %s is below the running control plane's floor %s", ErrOSRefused, rel.Version, floor) + } + return rel, false, nil +} + +// oneLineBack reports whether t is the line right before r. Within a major +// that is minor-1. Across a major (r is X.0) it is a line of major X-1, and the +// box can only check that the list names no later line of that major. +func oneLineBack(t, r line, list []OSRelease) bool { + if t.major == r.major { + return t.minor == r.minor-1 + } + if r.minor != 0 || t.major != r.major-1 { + return false + } + for _, e := range list { + if l := lineOf(canonical(e.Version)); l.major == t.major && l.minor > t.minor { + return false + } + } + return true +} diff --git a/internal/hostagent/updatetarget/os_test.go b/internal/hostagent/updatetarget/os_test.go new file mode 100644 index 00000000..a361931e --- /dev/null +++ b/internal/hostagent/updatetarget/os_test.go @@ -0,0 +1,214 @@ +package updatetarget + +import ( + "context" + "errors" + "strings" + "testing" + "time" + + "github.com/onmoose/os/internal/protocol" +) + +func rel(v string) OSRelease { + return OSRelease{ + Version: v, + BundleURL: DefaultOSURLPrefix + "v" + v + "/moose-v" + v + "-amd64.raucb", + BundleSHA256: strings.Repeat("a", 64), + } +} + +func TestValidateOS(t *testing.T) { + ok := []OSRelease{rel("1.3.4"), rel("1.4.2")} + if err := ValidateOS(ok, DefaultOSURLPrefix); err != nil { + t.Fatalf("a good list was refused: %v", err) + } + cases := map[string][]OSRelease{ + "no digest": {{Version: "1.4.2", BundleURL: rel("1.4.2").BundleURL}}, + "short digest": {{Version: "1.4.2", BundleURL: rel("1.4.2").BundleURL, BundleSHA256: "abc"}}, + "upper digest": {{Version: "1.4.2", BundleURL: rel("1.4.2").BundleURL, BundleSHA256: strings.Repeat("A", 64)}}, + "not a version": {{Version: "latest", BundleURL: rel("1.4.2").BundleURL, BundleSHA256: strings.Repeat("a", 64)}}, + "pre-release": {{Version: "1.4.2-rc1", BundleURL: rel("1.4.2").BundleURL, BundleSHA256: strings.Repeat("a", 64)}}, + "other host": {{Version: "1.4.2", BundleURL: "https://evil.example/b.raucb", BundleSHA256: strings.Repeat("a", 64)}}, + "not a URL": {{Version: "1.4.2", BundleURL: "/b.raucb", BundleSHA256: strings.Repeat("a", 64)}}, + "out of order": {rel("1.4.2"), rel("1.3.4")}, + "duplicate entry": {rel("1.4.2"), rel("1.4.2")}, + } + for name, list := range cases { + if err := ValidateOS(list, DefaultOSURLPrefix); !errors.Is(err, ErrOSRefused) { + t.Errorf("%s: want a refusal, got %v", name, err) + } + } + // An empty prefix accepts any http(s) URL: the in-guest source of the boot lane. + if err := ValidateOS([]OSRelease{{Version: "1.4.2", BundleURL: "http://127.0.0.1:5001/b.raucb", BundleSHA256: strings.Repeat("a", 64)}}, ""); err != nil { + t.Errorf("an empty prefix refused a plain URL: %v", err) + } +} + +func TestPickOS(t *testing.T) { + cases := []struct { + name, running, floor string + list []string + want string + current, refused bool + }{ + {name: "current", running: "1.4.2", list: []string{"1.3.9", "1.4.2"}, want: "1.4.2", current: true}, + {name: "patch up", running: "1.4.0", list: []string{"1.4.2"}, want: "1.4.2"}, + {name: "patch down", running: "1.4.2", list: []string{"1.4.0"}, want: "1.4.0"}, + {name: "next minor", running: "1.3.1", list: []string{"1.3.9", "1.4.2"}, want: "1.4.2"}, + {name: "never skips a minor", running: "1.2.0", list: []string{"1.2.5", "1.3.9", "1.4.2"}, want: "1.3.9"}, + {name: "across a major", running: "1.9.3", list: []string{"1.9.4", "2.0.1"}, want: "2.0.1"}, + {name: "one minor back", running: "1.4.2", list: []string{"1.3.9"}, want: "1.3.9"}, + {name: "two minors back", running: "1.4.2", list: []string{"1.2.9"}, refused: true}, + {name: "one minor back across a major", running: "2.0.1", list: []string{"1.9.4"}, want: "1.9.4"}, + {name: "not the last line of the major", running: "2.0.1", list: []string{"1.8.4", "1.9.4"}, refused: false, want: "1.9.4"}, + {name: "a later line of the old major exists", running: "2.0.1", list: []string{"1.8.4", "1.9.0", "1.8.9"}, refused: true}, + {name: "below the floor", running: "1.4.2", floor: "1.4.0", list: []string{"1.3.9"}, refused: true}, + {name: "above the floor", running: "1.4.2", floor: "1.3.0", list: []string{"1.3.9"}, want: "1.3.9"}, + {name: "running is not a version", running: "dev", list: []string{"1.4.2"}, refused: true}, + } + for _, c := range cases { + var list []OSRelease + for _, v := range c.list { + list = append(list, rel(v)) + } + got, current, err := PickOS(c.running, c.floor, list) + if c.refused { + if err == nil { + t.Errorf("%s: want a refusal, got %s", c.name, got.Version) + } + continue + } + if err != nil { + t.Errorf("%s: %v", c.name, err) + continue + } + if got.Version != c.want || current != c.current { + t.Errorf("%s: got %s current=%v, want %s current=%v", c.name, got.Version, current, c.want, c.current) + } + } +} + +// fakeOS records what the loop asked of stream A. +type fakeOS struct { + running string + calls []bool // the open flag of each Apply + rel OSRelease +} + +func (f *fakeOS) Running() (string, string) { return f.running, "A" } +func (f *fakeOS) Floor() string { return "" } +func (f *fakeOS) Apply(r OSRelease, open bool, _ time.Time) OSDecision { + f.calls = append(f.calls, open) + f.rel = r + if open { + return OSDecision{State: protocol.OSUpdateRebooting} + } + return OSDecision{State: protocol.OSUpdateInstalled} +} + +func osLoop(t Target, running string, w Window, now time.Time) (*Loop, *fakeOS, *fakeApplier) { + f := &fakeOS{running: running} + ap := &fakeApplier{} + l := &Loop{ + Source: &fakeSource{target: t}, + Current: fakeRunning{brain: t.BrainImage, ui: t.UIImage}, + Applier: ap, + AutoApply: true, + Window: w, + OS: f, + Now: func() time.Time { return now }, + } + return l, f, ap +} + +func TestTickOS(t *testing.T) { + w, _ := ParseWindow("03:00-04:00") + in := time.Date(2026, 10, 3, 3, 10, 0, 0, time.Local) + out := time.Date(2026, 10, 3, 12, 0, 0, 0, time.Local) + base := Target{Version: "v0.16.0", BrainImage: brainRef, UIImage: uiRef} + base.OS = []OSRelease{rel("0.15.1")} + + t.Run("installs ahead of the window, switches inside it", func(t *testing.T) { + l, f, _ := osLoop(base, "0.15.0", w, out) + l.Tick(context.Background()) + if len(f.calls) != 1 || f.calls[0] { + t.Fatalf("outside the window: want one Apply with open=false, got %v", f.calls) + } + if s := l.Snapshot().OS; s.State != protocol.OSUpdateInstalled || s.Target == nil || s.Target.Version != "0.15.1" { + t.Fatalf("snapshot: %+v", s) + } + l.Now = func() time.Time { return in } + l.Tick(context.Background()) + if len(f.calls) != 2 || !f.calls[1] { + t.Fatalf("inside the window: want open=true, got %v", f.calls) + } + }) + + t.Run("stream B goes first in the window", func(t *testing.T) { + tgt := base + l, f, ap := osLoop(tgt, "0.15.0", w, in) + l.Current = fakeRunning{brain: oldBrain, ui: tgt.UIImage} + l.Tick(context.Background()) + if len(ap.calls) != 1 { + t.Fatalf("want the control-plane update started, got %d calls", len(ap.calls)) + } + if len(f.calls) != 1 || f.calls[0] { + t.Fatalf("the OS must not switch while stream B holds the window, got %v", f.calls) + } + }) + + t.Run("a bad OS part refuses stream A only", func(t *testing.T) { + tgt := base + tgt.OS = []OSRelease{{Version: "0.15.1", BundleURL: rel("0.15.1").BundleURL}} + l, f, _ := osLoop(tgt, "0.15.0", w, in) + l.Current = fakeRunning{brain: oldBrain, ui: tgt.UIImage} + ap := l.Applier.(*fakeApplier) + l.Tick(context.Background()) + if len(f.calls) != 0 { + t.Fatalf("a refused OS part reached the applier") + } + if s := l.Snapshot().OS; s.State != protocol.OSUpdateRefused { + t.Fatalf("want refused, got %+v", s) + } + if len(ap.calls) != 1 { + t.Fatalf("the control plane must still update, got %d calls", len(ap.calls)) + } + }) + + t.Run("an answer with only an OS part still moves the OS", func(t *testing.T) { + tgt := Target{OS: []OSRelease{rel("0.15.1")}, Window: "00:00-23:59"} + l, f, ap := osLoop(tgt, "0.15.0", w, out) + l.Tick(context.Background()) + if l.Snapshot().Outcome != OutcomeRefused || len(ap.calls) != 0 { + t.Fatalf("the control-plane part must be refused: %+v, %d calls", l.Snapshot(), len(ap.calls)) + } + if len(f.calls) != 1 || !f.calls[0] { + t.Fatalf("stream A must still be judged, in the answer's window: %v", f.calls) + } + }) + + t.Run("current and none", func(t *testing.T) { + l, f, _ := osLoop(base, "0.15.1", w, in) + l.Tick(context.Background()) + if len(f.calls) != 0 || l.Snapshot().OS.State != protocol.OSUpdateCurrent { + t.Fatalf("current: calls %v, snapshot %+v", f.calls, l.Snapshot().OS) + } + tgt := base + tgt.OS = nil + l, _, _ = osLoop(tgt, "0.15.0", w, in) + l.Tick(context.Background()) + if l.Snapshot().OS.State != protocol.OSUpdateNone { + t.Fatalf("none: %+v", l.Snapshot().OS) + } + }) + + t.Run("no applier is unsupported", func(t *testing.T) { + l, _, _ := osLoop(base, "0.15.0", w, in) + l.OS = nil + l.Tick(context.Background()) + if l.Snapshot().OS.State != protocol.OSUpdateUnsupported { + t.Fatalf("got %+v", l.Snapshot().OS) + } + }) +} diff --git a/internal/hostagent/updatetarget/state.go b/internal/hostagent/updatetarget/state.go index 7ae16471..62c46724 100644 --- a/internal/hostagent/updatetarget/state.go +++ b/internal/hostagent/updatetarget/state.go @@ -56,6 +56,27 @@ type Snapshot struct { // so this is the resolved value, not the configured one. Window Window WindowFrom string + // OS is stream A's last decision. It is kept across a tick that ends + // before the OS part is read (an unreachable source), so it is the last + // decision, not this tick's. + OS OSSnapshot +} + +// OSSnapshot is stream A's last decision. State is one of the +// protocol.OSUpdate* values; the zero value means stream A has not been +// decided yet. +type OSSnapshot struct { + State string + Detail string + // Target is the release the box picked, when it picked one. + Target *OSRelease +} + +// recordOS stores stream A's decision for this tick. +func (l *Loop) recordOS(o OSSnapshot) { + l.snapMu.Lock() + defer l.snapMu.Unlock() + l.snap.OS = o } // Snapshot returns the last tick's decision. Safe to call from another @@ -74,5 +95,6 @@ func (l *Loop) record(o Outcome, t Target, w Window, from string, err error) { } l.snapMu.Lock() defer l.snapMu.Unlock() + s.OS = l.snap.OS l.snap = s } diff --git a/internal/hostagent/updatetarget/target.go b/internal/hostagent/updatetarget/target.go index 6f534e94..49fea064 100644 --- a/internal/hostagent/updatetarget/target.go +++ b/internal/hostagent/updatetarget/target.go @@ -80,6 +80,11 @@ type Target struct { // window it cannot state properly must not stop the box updating. The loop // parses it, warns, and falls back. Window string + // OS is stream A's part of the answer: OS releases in ascending order, the + // last one the target (os.go). Empty means the source has no opinion about + // the OS, and the box stays on the OS it runs. It is checked on its own + // (ValidateOS), so a bad OS part never blocks the control-plane pair. + OS []OSRelease } // Source is the seam: one call, one answer. diff --git a/internal/notify/notify.go b/internal/notify/notify.go index b18824ab..d3ef215d 100644 --- a/internal/notify/notify.go +++ b/internal/notify/notify.go @@ -384,3 +384,51 @@ func healthDedupKey(id, instanceKey string) string { } return "health:" + id + ":" + instanceKey } + +// SourceUpdate is the source_kind of an update-outcome notification +// (NOTIFICATIONS.md # Updates). +const SourceUpdate = "update" + +// OSUpdateDedupKey is the dedup key of one OS update outcome. The outcome id +// comes from host-agent and names one switch, so a re-read of the same outcome +// maps to the same row. +func OSUpdateDedupKey(outcomeID string) string { return "os-update:" + outcomeID } + +// OSUpdateOutcome tells admins what the last OS update did (UPDATES.md # 1 +// step 6, NOTIFICATIONS.md # Updates): info when the box moved to the new +// release, warning when the new release did not come up and the box went back +// on its own. The caller makes sure each outcome is raised once. +func (n *Notifier) OSUpdateOutcome(outcomeID, outcome, version, from string) { + note := Notification{ + TS: n.now().UnixMilli(), + Category: CategoryUpdates, + SourceKind: SourceUpdate, + SourceID: outcomeID, + DedupKey: OSUpdateDedupKey(outcomeID), + Audience: AudienceAdmins, + Variant: VariantTransparency, + ActionLabel: "Open About", + ActionRoute: "/settings/about", + } + switch outcome { + case "good": + note.Severity = SeverityInfo + note.Summary = "moose updated its system to " + version + note.Body = "Your moose installed system version " + version + " overnight and restarted. Everything came back as it was." + case "reverted": + note.Severity = SeverityWarning + note.Summary = "A system update did not work, so moose went back" + note.Body = "Your moose tried to install system version " + version + " overnight. It did not start correctly, so moose went back to version " + from + " on its own. Nothing was lost. It tries again in a later night." + default: + return + } + if err := n.store.RaiseNotification(note); err != nil { + slog.Error("notify: raise failed", "source_id", outcomeID, "err", err) + return + } + n.publish(events.NotificationCreated, map[string]any{ + "dedup_key": note.DedupKey, + "category": string(note.Category), + "severity": string(note.Severity), + }) +} diff --git a/internal/notify/notify_test.go b/internal/notify/notify_test.go index 6253f210..8e9da90f 100644 --- a/internal/notify/notify_test.go +++ b/internal/notify/notify_test.go @@ -418,3 +418,21 @@ func TestValidCategory(t *testing.T) { } } } + +func TestOSUpdateOutcome(t *testing.T) { + st := &fakeStore{} + n := New(st, nil) + n.OSUpdateOutcome("os-0.15.1-1", "good", "0.15.1", "0.15.0") + n.OSUpdateOutcome("os-0.15.2-2", "reverted", "0.15.2", "0.15.1") + n.OSUpdateOutcome("os-x", "something else", "1", "0") + if len(st.raised) != 2 { + t.Fatalf("want 2 raises, got %d", len(st.raised)) + } + good, rev := st.raised[0], st.raised[1] + if good.Severity != SeverityInfo || good.Audience != AudienceAdmins || good.Category != CategoryUpdates || good.DedupKey != "os-update:os-0.15.1-1" { + t.Fatalf("good: %+v", good) + } + if rev.Severity != SeverityWarning || rev.SourceKind != SourceUpdate { + t.Fatalf("reverted: %+v", rev) + } +} diff --git a/internal/protocol/host.go b/internal/protocol/host.go index e2cb6802..a04f0e6e 100644 --- a/internal/protocol/host.go +++ b/internal/protocol/host.go @@ -74,10 +74,15 @@ type UnpublishRequest struct { // for the install-plan footprint (DECISIONS.md 2026-06-13). Same Bavail/Blocks // statfs semantics per entry; absent volumes are omitted, never zero-filled. type SystemStatus struct { - Hostname string `json:"hostname"` - UptimeS int64 `json:"uptime_s"` - DiskPressure bool `json:"disk_pressure"` - AgentVersion string `json:"agent_version"` + Hostname string `json:"hostname"` + UptimeS int64 `json:"uptime_s"` + DiskPressure bool `json:"disk_pressure"` + AgentVersion string `json:"agent_version"` + // OSVersion and OSSlot are the OS release this box runs and the slot it + // booted ("A" or "B"). Empty when this host-agent cannot tell (the + // appliance until #564, the fake). + OSVersion string `json:"os_version,omitempty"` + OSSlot string `json:"os_slot,omitempty"` DataDiskFreeBytes int64 `json:"data_disk_free_bytes"` DataDiskTotalBytes int64 `json:"data_disk_total_bytes"` Disks []DiskSpace `json:"disks"` @@ -482,6 +487,14 @@ type Error struct { // JobKindSystemUpdate is the kind name of the control-plane update job. const JobKindSystemUpdate = "system-update" +// JobKindOSInstall and JobKindOSSwitch are the two OS update jobs (#563): the +// download and install into the other slot, and the switch and reboot. They +// take the same lock as system-update, so no two of them ever run at once. +const ( + JobKindOSInstall = "os-install" + JobKindOSSwitch = "os-switch" +) + // Job status values. `cancelled`, `cancelling`, and `stalled` from the spec are // not produced by this host-agent: there is no cancel route, and a run that // passes its MaxDuration is cancelled and rolled back, so it lands on a known @@ -599,8 +612,92 @@ type UpdateTarget struct { // Profile is the environment profile that made the decision, "appliance" or // "hosted" (ENVIRONMENT.md). Profile string `json:"profile,omitempty"` + // OS is stream A's last decision, beside stream B's above (UPDATES.md # 1, + // #563). Nil when this host-agent reports nothing about the OS. + OS *OSUpdate `json:"os,omitempty"` } +// OSUpdate is stream A on GET /v1/system/update-target: the OS image this box +// runs, what its update target names, and what the box last did about it. +type OSUpdate struct { + // State is one of the OSUpdate* constants. + State string `json:"state"` + // Running is the OS release this box runs: host-agent's own version, which + // is the version of the slot it ships in. Slot is "A" or "B", empty when + // the box could not read it. + Running string `json:"running,omitempty"` + Slot string `json:"slot,omitempty"` + // Target is the OS release the box picked from the answer: the next step + // on the way to the answer's target, or the target itself (UPDATES.md # 1). + Target *OSRelease `json:"target,omitempty"` + // Detail is a diagnostic for refused, failed, held and unsupported. Not UI + // copy, like UpdateTarget.Detail. + Detail string `json:"detail,omitempty"` + // Last is the outcome of the last OS switch, kept across reboots. + Last *OSOutcome `json:"last,omitempty"` +} + +// OSRelease is one OS release an update target names: the version and the +// bundle the box downloads, pinned by its sha256. +type OSRelease struct { + Version string `json:"version"` + BundleURL string `json:"bundle_url"` + BundleSHA256 string `json:"bundle_sha256"` +} + +// OSOutcome is what happened the last time the box switched to a new OS slot. +type OSOutcome struct { + // ID names this outcome once, so the brain sends one notification for it + // and not one per read. + ID string `json:"id"` + // Outcome is "good" (the new slot came up healthy and was marked good) or + // "reverted" (it did not, and the box went back to the old slot). + Outcome string `json:"outcome"` + // Version is the release the box switched to; From the one it came from. + Version string `json:"version"` + From string `json:"from,omitempty"` + // At is when the outcome was recorded, RFC3339. + At string `json:"at"` +} + +// The OSUpdate states. +const ( + // OSUpdateUnsupported: this box cannot update its OS (the appliance until + // #564, the fake host-agent, a box not in the A/B layout). + OSUpdateUnsupported = "unsupported" + // OSUpdateNone: the answer names no OS release. The box stays on its OS. + OSUpdateNone = "none" + // OSUpdateRefused: the answer's OS part was refused (no digest, a URL + // outside the expected prefix, a step the box may not take). Nothing was + // downloaded. + OSUpdateRefused = "refused" + // OSUpdateCurrent: the box runs the target. + OSUpdateCurrent = "current" + // OSUpdateInstalling: the bundle is being downloaded and written into the + // other slot. + OSUpdateInstalling = "installing" + // OSUpdateInstalled: the target is in the other slot and waits for the + // update window. + OSUpdateInstalled = "installed" + // OSUpdateRebooting: the box switched slots and is rebooting. + OSUpdateRebooting = "rebooting" + // OSUpdateWaiting: the box would act now, but another job holds the job + // lock. The next check tries again. + OSUpdateWaiting = "waiting" + // OSUpdateHeld: the box already tried this target tonight. It tries again + // in the next window. + OSUpdateHeld = "held" + // OSUpdateFailed: the last download or install failed. Detail says why. It + // is tried again the next night. + OSUpdateFailed = "failed" +) + +// The OSOutcome values. +const ( + OSOutcomeGood = "good" + OSOutcomeReverted = "reverted" +) + // ControlPlanePair is a brain + UI image reference pair. Empty fields mean the // box could not read that half of its own declaration. type ControlPlanePair struct { diff --git a/internal/store/notify.go b/internal/store/notify.go index 1b8a2a56..ea7f9e97 100644 --- a/internal/store/notify.go +++ b/internal/store/notify.go @@ -400,3 +400,16 @@ func (s *Store) MarkAllNotificationsRead(userID string, isAdmin bool, at time.Ti userID, at.UnixMilli(), userID, userID, userID) return err } + +// HasNotification reports whether any notification row, in any state, carries +// dedupKey. A source that must raise one notification per event for good (an +// OS update outcome, #563) uses it, because RaiseNotification coalesces into a +// row that is not dismissed and resets its read state: re-raising an outcome +// the admin already read would make it unread again. +func (s *Store) HasNotification(dedupKey string) (bool, error) { + var n int + if err := s.db.QueryRow(`SELECT COUNT(*) FROM notifications WHERE dedup_key = ?`, dedupKey).Scan(&n); err != nil { + return false, err + } + return n > 0, nil +} diff --git a/internal/store/notify_test.go b/internal/store/notify_test.go index 4deca800..79a80c2d 100644 --- a/internal/store/notify_test.go +++ b/internal/store/notify_test.go @@ -963,3 +963,16 @@ func dedupSet(ns []notify.Notification, want ...string) bool { } return true } + +func TestHasNotification(t *testing.T) { + s := open(t) + if ok, err := s.HasNotification("os-update:x"); err != nil || ok { + t.Fatalf("empty table: %v %v", ok, err) + } + if err := s.RaiseNotification(newNotification("os-update:x")); err != nil { + t.Fatal(err) + } + if ok, err := s.HasNotification("os-update:x"); err != nil || !ok { + t.Fatalf("after a raise: %v %v", ok, err) + } +} diff --git a/web-ui/src/generated/openapi.ts b/web-ui/src/generated/openapi.ts index 49962ba6..25573177 100644 --- a/web-ui/src/generated/openapi.ts +++ b/web-ui/src/generated/openapi.ts @@ -955,7 +955,7 @@ export interface paths { path?: never; cookie?: never; }; - /** What this box is running: brain version and commit, host-agent version, UI image */ + /** What this box is running: brain version and commit, host-agent version, UI image, OS version and slot */ get: operations["get-system-version"]; put?: never; post?: never; @@ -1920,6 +1920,28 @@ export interface components { /** Format: int64 */ ts: number; }; + OSOutcomeDTO: { + at: string; + from?: string; + id: string; + /** @enum {string} */ + outcome: "good" | "reverted"; + version: string; + }; + OSReleaseDTO: { + bundle_sha256: string; + bundle_url: string; + version: string; + }; + OSUpdateDTO: { + detail?: string; + last?: components["schemas"]["OSOutcomeDTO"]; + running?: string; + slot?: string; + /** @enum {string} */ + state: "unsupported" | "none" | "refused" | "current" | "installing" | "installed" | "waiting" | "rebooting" | "held" | "failed"; + target?: components["schemas"]["OSReleaseDTO"]; + }; "Parse-custom-overlayRequest": { /** * Format: uri @@ -2168,6 +2190,8 @@ export interface components { readonly $schema?: string; commit: string; host_agent_version?: string; + os_slot?: string; + os_version?: string; ui_image?: string; version: string; }; @@ -2210,6 +2234,7 @@ export interface components { checked_at?: string; detail?: string; from?: string; + os?: components["schemas"]["OSUpdateDTO"]; profile?: string; running: components["schemas"]["ControlPlanePairDTO"]; /** @enum {string} */ From c5dc4df65385b245833d1566105599a1f76f18e4 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 16:48:27 +0100 Subject: [PATCH 02/18] ci: fix the OS test bundle script's repo root (#563) --- CLAUDE.md | 2 +- dev/cloud/test/build-os-test-bundle.sh | 2 +- docs/architecture.md | 17 +++++++++++++--- docs/dev/contributing.md | 2 +- docs/dev/hosted-boot-proof.md | 28 +++++++++++++++++++++++--- docs/specs/BRAIN_HOST_PROTOCOL.md | 23 ++++++++++++++++++++- docs/specs/BUILD.md | 13 ++++++------ docs/specs/ENVIRONMENT.md | 2 +- docs/specs/NOTIFICATIONS.md | 2 +- docs/specs/TESTING.md | 2 ++ docs/specs/UPDATES.md | 23 ++++++++++++++++----- 11 files changed, 93 insertions(+), 23 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index ee280bc2..faa77f4b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -92,7 +92,7 @@ Small set of rules. Codified now so we don't have to back them out later. - **Consumer-side interfaces.** Interfaces live in the package that *uses* them, not the package that implements them. `lifecycle.DockerDriver` lives in `internal/lifecycle/`, not in a hypothetical `internal/docker/`. Provider packages export concrete types only. Exception: a single interface shared by three or more consumers can move to the provider, but default to consumer-side until that's true. - **Layer boundaries.** `internal/lifecycle` is the transaction owner; only `cmd/brain` and `internal/api` may import it. `internal/store` is the persistence boundary; only `internal/lifecycle`, `internal/api`, `internal/auth`, `internal/audit`, and `cmd/brain` may import it. Anything else reaching in is breaking the model — push the call through the right seam instead. - **`log/slog` is the only logger.** No `"log"` imports, no `fmt.Println` for diagnostics. Structured fields, not interpolated strings: `slog.Info("app installed", "instance_id", id)`, not `slog.Info(fmt.Sprintf("installed %s", id))`. The default handler is set in `cmd/brain/main.go`; use `slog.Default()` (the package-level functions) — don't thread `*slog.Logger` through constructors. -- **Standard structured fields.** Use these key names so journalctl/jq filters stay stable: `instance_id`, `manifest_id`, `slug`, `service`, `image`, `host`, `upstream`, `step`, `err`, `output`, `user_id`, `username`, `role`, `action`, `actor_user_id`, `target_kind`, `target_id`, `retry_after`, `retry_after_s`, `name`, `uid`, `iface`, `interfaces`, `src`, `dir`, `profile`, `box_id`, `zone`, `exposure`, `trusted_proxies`, `brain`, `ui`, `minimum_host_agent`, `keys`, `state_dir`, `job_id`, `window`, `url`, `from`, `count`, `dropped`, `tier`, `remap_base`, `gid`, `image_user`. `host` is a machine or upstream hostname only — a single network interface name is `iface`, a list of them is `interfaces` (never overload `host` for either). `src` is a source filesystem path (bind-source, folder-source); `dir` is a relative bind dir path. `profile` is the resolved environment profile (`appliance`|`hosted`, `ENVIRONMENT.md`). `box_id` is the hosted box's provisioned identity (`ENVIRONMENT.md` # Provisioning). `zone` is an IANA time-zone name (`TIME.md`, host-agent set-timezone). `exposure` is an app's per-instance access mode (`restricted`|`public`, `ENVIRONMENT.md` #306). `trusted_proxies` is the configured set of proxies whose `X-Forwarded-For` the brain reads when deriving a client IP (`BRAIN_UI_PROTOCOL.md` # Rate limiting & abuse). `brain` and `ui` are the two control-plane versions a release names, and `minimum_host_agent` the host-agent version it requires (`RELEASE_MANIFEST.md`); use them for versions, not for image refs — an image ref is `image`. `keys` is how many signing keys a build accepts, and `state_dir` a state directory path (`src` stays for a source path being read or bound). `job_id` is a host-agent job id (`internal/hostagent/jobs.go`), and `window` the configured update window (`UPDATES.md` # 8.4). `url` is an HTTP endpoint the box reads, and `from` says which of several configured sources a setting came from, one of `answer` (the control plane's update-target answer), `seed`, `env`, `default` (`UPDATES.md` # 8.4); where there are only two sources, the boolean `from_` form is used instead (`from_ledger`). `dropped` lists what the box read but could not use, from catalog data it reads leniently (AI provider entries, manifest `role` and `requires` keys), and `count` is how many there were. `tier` is an instance's user-namespace tier (`default`|`caps`|`image`|`host`, `APP_ISOLATION.md` # User-namespace tiers), and `remap_base` the first host id of the Docker remap range (0 for none). `image_user` is the user an image sets (its `Config.User`, a name or a number, as the image wrote it), which the image tier resolves to ids. `name` is an app instance's display name (it rides alongside `instance_id`, never instead of it), and `uid` a numeric Unix user id — the allocated app-service identity, a resolved home owner, or an image user's id. `gid` is its numeric group id. `retry_after` and `retry_after_s` come from the two throttles that `AUTH.md` # Rate limiting keeps apart on purpose. Both are correct. Do not merge them. `retry_after` is the login backoff's wait, written as a duration string (`internal/api/auth.go`); that path sends no `Retry-After` header, by design. `retry_after_s` is the general request limiter's wait, written as a whole number of seconds; it matches that limiter's `retry_after_s` JSON field and the `Retry-After` header it sets (`internal/api/ratelimit.go`, `BRAIN_UI_PROTOCOL.md` # 429 contract). To find every throttled request you must search for both keys. That is the price of the split, not a bug. Adding a new recurring field? Add it here. +- **Standard structured fields.** Use these key names so journalctl/jq filters stay stable: `instance_id`, `manifest_id`, `slug`, `service`, `image`, `host`, `upstream`, `step`, `err`, `output`, `user_id`, `username`, `role`, `action`, `actor_user_id`, `target_kind`, `target_id`, `retry_after`, `retry_after_s`, `name`, `uid`, `iface`, `interfaces`, `src`, `dir`, `profile`, `box_id`, `zone`, `exposure`, `trusted_proxies`, `brain`, `ui`, `minimum_host_agent`, `keys`, `state_dir`, `job_id`, `window`, `url`, `from`, `count`, `dropped`, `tier`, `remap_base`, `gid`, `image_user`, `os`, `slot`, `digest`. `host` is a machine or upstream hostname only — a single network interface name is `iface`, a list of them is `interfaces` (never overload `host` for either). `src` is a source filesystem path (bind-source, folder-source); `dir` is a relative bind dir path. `profile` is the resolved environment profile (`appliance`|`hosted`, `ENVIRONMENT.md`). `box_id` is the hosted box's provisioned identity (`ENVIRONMENT.md` # Provisioning). `zone` is an IANA time-zone name (`TIME.md`, host-agent set-timezone). `exposure` is an app's per-instance access mode (`restricted`|`public`, `ENVIRONMENT.md` #306). `trusted_proxies` is the configured set of proxies whose `X-Forwarded-For` the brain reads when deriving a client IP (`BRAIN_UI_PROTOCOL.md` # Rate limiting & abuse). `brain` and `ui` are the two control-plane versions a release names, and `minimum_host_agent` the host-agent version it requires (`RELEASE_MANIFEST.md`); use them for versions, not for image refs — an image ref is `image`. `keys` is how many signing keys a build accepts, and `state_dir` a state directory path (`src` stays for a source path being read or bound). `job_id` is a host-agent job id (`internal/hostagent/jobs.go`), and `window` the configured update window (`UPDATES.md` # 8.4). `url` is an HTTP endpoint the box reads, and `from` says which of several configured sources a setting came from, one of `answer` (the control plane's update-target answer), `seed`, `env`, `default` (`UPDATES.md` # 8.4); where there are only two sources, the boolean `from_` form is used instead (`from_ledger`). `dropped` lists what the box read but could not use, from catalog data it reads leniently (AI provider entries, manifest `role` and `requires` keys), and `count` is how many there were. `tier` is an instance's user-namespace tier (`default`|`caps`|`image`|`host`, `APP_ISOLATION.md` # User-namespace tiers), and `remap_base` the first host id of the Docker remap range (0 for none). `image_user` is the user an image sets (its `Config.User`, a name or a number, as the image wrote it), which the image tier resolves to ids. `os` is an OS (moose) release version, as `brain` and `ui` are control-plane versions; `slot` is an A/B OS slot (`A`|`B`, `BUILD.md` # 1b); `digest` is an OS bundle's sha256 (`UPDATES.md` # 1). `name` is an app instance's display name (it rides alongside `instance_id`, never instead of it), and `uid` a numeric Unix user id — the allocated app-service identity, a resolved home owner, or an image user's id. `gid` is its numeric group id. `retry_after` and `retry_after_s` come from the two throttles that `AUTH.md` # Rate limiting keeps apart on purpose. Both are correct. Do not merge them. `retry_after` is the login backoff's wait, written as a duration string (`internal/api/auth.go`); that path sends no `Retry-After` header, by design. `retry_after_s` is the general request limiter's wait, written as a whole number of seconds; it matches that limiter's `retry_after_s` JSON field and the `Retry-After` header it sets (`internal/api/ratelimit.go`, `BRAIN_UI_PROTOCOL.md` # 429 contract). To find every throttled request you must search for both keys. That is the price of the split, not a bug. Adding a new recurring field? Add it here. - **Typed errors at boundaries, not everywhere.** Define a sentinel/typed error only when a *consumer* needs to discriminate (HTTP status, retry decision, UI text). `store.ErrNotFound` exists because the API maps it to 404. Don't pre-declare error types speculatively. - **No premature abstraction.** Don't introduce an interface, factory, or DI container until at least two concrete consumers exist. It bites hardest in Go where every extra interface is import-graph weight. - **`internal/` for everything except `cmd/`.** No `pkg/`. Anything inside `internal/` is private to this module by Go's own rules — no public API surface to maintain. diff --git a/dev/cloud/test/build-os-test-bundle.sh b/dev/cloud/test/build-os-test-bundle.sh index e1da104e..bfa02945 100755 --- a/dev/cloud/test/build-os-test-bundle.sh +++ b/dev/cloud/test/build-os-test-bundle.sh @@ -18,7 +18,7 @@ # OUTDIR/os-test.version (the version in the bundle and in its host-agent). set -euo pipefail -REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)" +REPO_ROOT="$(cd "$(dirname "$0")/../../.." && pwd)" # shellcheck source=dev/cloud/rauc.sh . "${REPO_ROOT}/dev/cloud/rauc.sh" diff --git a/docs/architecture.md b/docs/architecture.md index 9f4ffa8e..e2e7a4b0 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -14,7 +14,7 @@ one is JavaScript, one is a container we don't write. |---|---|---|---| | **`moose-brain`** | `cmd/brain/`, `internal/` | The control-plane daemon. Owns SQLite state, the REST+SSE API, the app lifecycle, and the Caddy config. One Go binary. | Real | | **`host-agent` (fake)** | `cmd/host-agent/` | Privileged side used in the inner dev loop. Speaks the real `BRAIN_HOST_PROTOCOL.md` wire format over a UNIX socket; the host operations themselves (Avahi, LUKS, PAM, apt) are stubbed in memory. | **Fake** (real wire, canned ops) | -| **`host-agent-real`** | `cmd/host-agent-real/`, `internal/hostagent/` | The real privileged binary. Seam-injected reporters: PAM password verify (`pamverifier`), `/proc` system sampling (`procsource`), disk usage, RAM pressure, journal streaming, service health, reboot-required flag, user manager, system time-zone setter (`timezone`, `timedatectl set-timezone` — the first-run wizard's Step 3, wired in both build profiles), per-account SSH access (`sshaccess`, #463 — renders `sshd_config.d/moose-allowed.conf` whole from the enabled set, validates with `sshd -t`, and starts or stops sshd so :22 is open only while someone has SSH on; wired in both build profiles). Discovery is real: per-LAN-interface Avahi announcements (`avahipublisher`) driven by the NetworkManager LAN set (`netstate`), with an avahi-daemon.conf allowlist sync and IP-change replay. Seeds the brain's Docker transport then launches the brain container on startup (`brainlaunch`: `EnsureTransport` creates `moose-ingress` + runs the `docker-socket-proxy`; `Launch` docker-loads the bundled image if absent, lockstep `moose.protocol.major` OCI-label check, `docker run --restart unless-stopped` on the ingress net with `DOCKER_HOST` at the proxy). Before the launch it makes sure the household shared tree `/srv/moose/shared` exists as `root:moose-shared` `02770` (`brainlaunch.EnsureSharedTree`) and mounts it into the brain at the same path, so the brain can prepare shared folder sources (#519). Personal folder sources are prepared by host-agent itself, over `POST /v1/users/{username}/prepare-folder` (`usermgr`, a walk that never follows a symlink), so the brain never mounts `/home`. Host ops not yet wired: LUKS/TPM, apt, NM configuration (WiFi setup, `/v1/network/*`). A build-tagged slim **`hosted`** profile (`go build -tags hosted`, #204/C1c) compiles the discovery/NetworkManager stack out for the cloud image — `avahipublisher`/`netstate` unwired, no-op publisher, nil `Net` — keeping the same PAM/user-mgmt/health-system/brain-launch seams (`cmd/host-agent-real/wiring_appliance.go` vs `wiring_hosted.go`). | Partial — see "What is not built yet" | +| **`host-agent-real`** | `cmd/host-agent-real/`, `internal/hostagent/` | The real privileged binary. Seam-injected reporters: PAM password verify (`pamverifier`), `/proc` system sampling (`procsource`), disk usage, RAM pressure, journal streaming, service health, reboot-required flag, user manager, system time-zone setter (`timezone`, `timedatectl set-timezone` — the first-run wizard's Step 3, wired in both build profiles), per-account SSH access (`sshaccess`, #463 — renders `sshd_config.d/moose-allowed.conf` whole from the enabled set, validates with `sshd -t`, and starts or stops sshd so :22 is open only while someone has SSH on; wired in both build profiles). Discovery is real: per-LAN-interface Avahi announcements (`avahipublisher`) driven by the NetworkManager LAN set (`netstate`), with an avahi-daemon.conf allowlist sync and IP-change replay. Seeds the brain's Docker transport then launches the brain container on startup (`brainlaunch`: `EnsureTransport` creates `moose-ingress` + runs the `docker-socket-proxy`; `Launch` docker-loads the bundled image if absent, lockstep `moose.protocol.major` OCI-label check, `docker run --restart unless-stopped` on the ingress net with `DOCKER_HOST` at the proxy). Before the launch it makes sure the household shared tree `/srv/moose/shared` exists as `root:moose-shared` `02770` (`brainlaunch.EnsureSharedTree`) and mounts it into the brain at the same path, so the brain can prepare shared folder sources (#519). Personal folder sources are prepared by host-agent itself, over `POST /v1/users/{username}/prepare-folder` (`usermgr`, a walk that never follows a symlink), so the brain never mounts `/home`. The A/B OS update on the hosted build (`osupdate`, #563: RAUC install into the other slot, switch in the window, trial boot, revert). Host ops not yet wired: LUKS/TPM, NM configuration (WiFi setup, `/v1/network/*`). A build-tagged slim **`hosted`** profile (`go build -tags hosted`, #204/C1c) compiles the discovery/NetworkManager stack out for the cloud image — `avahipublisher`/`netstate` unwired, no-op publisher, nil `Net` — keeping the same PAM/user-mgmt/health-system/brain-launch seams (`cmd/host-agent-real/wiring_appliance.go` vs `wiring_hosted.go`). | Partial — see "What is not built yet" | | **Caddy** | `dev/caddy.json`, `dev/docker-compose.yml` | Reverse proxy. Terminates `*.local` (appliance) or `*..onmoose.io` over real Let's Encrypt HTTPS (hosted, via a custom acme-dns build) and routes to app containers + the brain. Configured live by the brain via Caddy's admin API. | Real (container) | | **`web-ui`** | `web-ui/` | Vue 3 + Vite + TanStack Query dashboard. Talks only to the brain. Tailwind 4 with the Oatmeal `@theme` tokens; `reka-ui` + `cn()` are present as shadcn-vue scaffolding, but the owned components in `components/ui/` (`Button`, `Heading`) are hand-written from the Oatmeal patterns, not pulled through the shadcn CLI (#261). The catalog install is a set of steps, one need per step with the saved accounts listed and one picked, and the last step installs; an app with no steps installs from the App page (`INSTALL_STEPS.md`, `DASHBOARD.md` # Install authorization). Internal code architecture: [`dev/web-ui.md`](dev/web-ui.md). | Real | | **SQLite** | `$STATE_DIR/moose.db` | The brain's only persistent store. Schema + queries in `internal/store/`. | Real | @@ -191,8 +191,19 @@ So this doc isn't read as a claim about the finished product: `os-release` environment, `dev/release/sign-bundle.sh`; an offline root CA, `docs/dev/rauc-signing.md`) and attached beside the image. host-agent's hosted build reports the state partition as its "System" volume - (`diskusage.NewHosted`). Not built: the update itself - (#563) and the appliance layout (#564), so a box never updates its OS today. **The OS + (`diskusage.NewHosted`). **A hosted box updates its OS (#563):** + `internal/hostagent/osupdate` downloads the bundle the update target names, + checks its sha256 against the target, has RAUC install it into the other + slot ahead of the window, switches and reboots inside it (`os-install` and + `os-switch` jobs under the one job lock, after stream B), and on the next + boot marks the slot good once the brain answers, or reboots back. An image + timer (`moose-os-trial.timer`) reboots a slot whose host-agent never + started. Only the first boot after a switch is on trial. The brain reports + `os_version`/`os_slot` and stream A's decision, and raises one admin + notification per outcome. Proven by the `os-update` and `os-revert` boots + under both firmwares. Not built: the OS part of the private control plane's + answer (described in `docs/progress/host-agent-os-update.md`), so no + production box moves its OS yet, and the appliance layout (#564). **The OS package lock is built (#560)** for the hosted image: `dev/os-lock/` holds a snapshot.debian.org timestamp, exact Docker pins and the resolved package list, both cloud builds fail when they resolve to a different list, and diff --git a/docs/dev/contributing.md b/docs/dev/contributing.md index 0fe7bfec..a32771c2 100644 --- a/docs/dev/contributing.md +++ b/docs/dev/contributing.md @@ -240,7 +240,7 @@ What `handover` does next depends on one secret: ### OS patch releases from a lock bump -Merging a lock bump puts the change on `dev`, not on any box. It ships with an OS release (`VERSION`). **Until the A/B applier lands (#561 to #564), an OS release reaches only new boxes**; a running box keeps the OS it was built with. +Merging a lock bump puts the change on `dev`, not on any box. It ships with an OS release (`VERSION`). **A hosted box installs an OS release once its update target names it** (#563, `../specs/UPDATES.md` # 1). Until the control plane sends the OS part of that answer, and on the appliance until #564, an OS release reaches only new boxes; a running box keeps the OS it was built with. - **A bump that changes any package from `trixie-security` is released within 7 days.** The bump PR's body says so when it applies. - **Any other bump ships with the next normal release.** diff --git a/docs/dev/hosted-boot-proof.md b/docs/dev/hosted-boot-proof.md index 65b137e3..09424c51 100644 --- a/docs/dev/hosted-boot-proof.md +++ b/docs/dev/hosted-boot-proof.md @@ -29,6 +29,9 @@ Net: a provisioned box logs both milestones, binds `:443`, and serves every ` -f publish=false`. Builds the image, then runs the `unseeded seeded access update ssh remap` boots under QEMU (`remap` is two boots), under UEFI and again under legacy BIOS. Add `-f boots="remap"` (any space-separated list of boot names) to run only those; the script refuses a name it does not know, so a typo fails instead of running nothing. `publish=true` (the default) additionally **attaches the compressed image, the signed OS update bundle and their checksums to the `v` GitHub Release (#562; it needs the release root CA and the `os-release` signer, `rauc-signing.md`) and pushes the brain + UI images to ghcr as `v`** (#352 removed the old provider-snapshot upload, so the lane holds no hosting-provider credential), so only do that deliberately. To publish one release line only, pass `-f publish=false` with `-f publish_os=true` or `-f publish_control_plane=true` (#559). Either way the publish steps only add what is missing: they never replace an asset already on the Release or an image tag already on ghcr. A Release with only one of the image and its checksum refuses until a person deletes the stray asset. A full run takes about 12 min. +- **CI (preferred: no local root/KVM, no image push):** `gh workflow run "CI / Cloud image" --ref -f publish=false`. Builds the image, then runs the `unseeded seeded access update ssh remap os-update os-revert` boots under QEMU (`remap` is two boots; the OS boots reboot their box inside one run), under UEFI and again under legacy BIOS. Add `-f boots="remap"` (any space-separated list of boot names) to run only those; the script refuses a name it does not know, so a typo fails instead of running nothing. `publish=true` (the default) additionally **attaches the compressed image, the signed OS update bundle and their checksums to the `v` GitHub Release (#562; it needs the release root CA and the `os-release` signer, `rauc-signing.md`) and pushes the brain + UI images to ghcr as `v`** (#352 removed the old provider-snapshot upload, so the lane holds no hosting-provider credential), so only do that deliberately. To publish one release line only, pass `-f publish=false` with `-f publish_os=true` or `-f publish_control_plane=true` (#559). Either way the publish steps only add what is missing: they never replace an asset already on the Release or an image tag already on ghcr. A Release with only one of the image and its checksum refuses until a person deletes the stray asset. A full run takes about 12 min. - **How a run is laid out** (`../progress/ci-cloud-image-speedup.md`; `sign` from #562). Four jobs: - `build` runs every assert, picks the boots (a typo in `boots` fails here, before the build), builds the production image, then the boot-proof image, and uploads the boot-proof image as a qcow2 artifact (`cloud-image-boot-qcow2`). On a run that publishes, it also uploads what the publish job needs. It also downloads the QEMU packages once, on a fresh runner, and hands them to the boot jobs (`cloud-image-qemu-debs`), so one slow Ubuntu mirror cannot hold up ten jobs. This job takes about 8 min, and the resolved package list (`cloud-packages-lock`) is uploaded here. - - `boot` is a matrix: one job per boot group and firmware, named like `Boot update (bios)`, all running at once. `unseeded seeded` is one group because those boots share one disk; `frozen` joins it when you ask for it. Every other boot is its own group. So the full list is 10 jobs, and `-f boots="remap"` is 2. The longest is `update`, about 2.6 min of boot. A red boot does not cancel the others, so one run shows every failing boot. **To debug a red boot, open that one job's log**: it holds the serial dump for that boot only. + - `boot` is a matrix: one job per boot group and firmware, named like `Boot update (bios)`, all running at once. `unseeded seeded` is one group because those boots share one disk; `frozen` joins it when you ask for it. Every other boot is its own group. So the full list is 14 jobs, and `-f boots="remap"` is 2. The longest is `update`, about 2.6 min of boot. A red boot does not cancel the others, so one run shows every failing boot. **To debug a red boot, open that one job's log**: it holds the serial dump for that boot only. - `publish` runs only when a line publishes, and only after `build` and every `boot` job passed. When the OS line publishes it first waits for `sign`, and attaches the image, the bundle `sign` re-signed and their checksums as one set (#562; `dev/release/attach-image.sh` detects and refuses a mixed set). For the control plane it loads the brain and UI tarballs the image baked and pushes those, so it pushes the images the boots ran, not a rebuild. - `sign` runs only when the OS line publishes, after every `boot` job passed. It is the only job that enters the `os-release` environment, and it does nothing else: it checks the bundle's sha256 against the one the build job reported, re-signs it with the release signer (`dev/release/sign-bundle.sh`) and hands it to `publish`. A red `sign` with "no release signer" means the environment's secrets are missing; on a branch other than `main` or a `v*` tag, the environment refuses the job before any step runs. - **Layer cache.** A run that publishes nothing builds the hosted Caddy image with the GitHub Actions layer cache (`MOOSE_BUILD_CACHE=gha`, set in the workflow from the same expression as the publish switches). Its inputs are all pinned, so the cache hits on every run after a branch's first. A run that publishes builds it fresh, with no cache read. The brain and UI always build plain: they rebuild on every commit, so a cache cost more to upload than it saved. -- **Local:** `sudo make test-cloud-qemu` (needs root + `/dev/kvm`). Scope boots with `MOOSE_CLOUD_BOOTS="seeded"` to reproduce the wildcard path alone, `"update"` for the updater proof alone, `"remap"` for the two remap boots alone, or the default `"unseeded seeded frozen access update ssh remap"` for the full run. `MOOSE_CLOUD_FIRMWARES="bios"` (or `"uefi"`) runs them under one firmware only; `bios` is no longer a boot name, and the script refuses it. `MOOSE_CLOUD_QCOW2=` boots a qcow2 that is already built and builds nothing (this is how each CI boot job runs). +- **Local:** `sudo make test-cloud-qemu` (needs root + `/dev/kvm`). Scope boots with `MOOSE_CLOUD_BOOTS="seeded"` to reproduce the wildcard path alone, `"update"` for the updater proof alone, `"remap"` for the two remap boots alone, `"os-update"` or `"os-revert"` for the OS update boots (they need `MOOSE_CLOUD_OS_BUNDLE_DIR`, the output of `dev/cloud/test/build-os-test-bundle.sh`), or the default `"unseeded seeded frozen access update ssh remap os-update os-revert"` for the full run. `MOOSE_CLOUD_FIRMWARES="bios"` (or `"uefi"`) runs them under one firmware only; `bios` is no longer a boot name, and the script refuses it. `MOOSE_CLOUD_QCOW2=` boots a qcow2 that is already built and builds nothing (this is how each CI boot job runs). ## Related history (frozen snapshots — background, not the current view) diff --git a/docs/specs/BRAIN_HOST_PROTOCOL.md b/docs/specs/BRAIN_HOST_PROTOCOL.md index 0940ef09..0f65eaa5 100644 --- a/docs/specs/BRAIN_HOST_PROTOCOL.md +++ b/docs/specs/BRAIN_HOST_PROTOCOL.md @@ -87,6 +87,10 @@ GET /v1/system/status # omitted, not zero-filled). data_disk_* are kept for the install-plan # footprint (DECISIONS.md 2026-06-13). The brain re-serves disks to the UI at # GET /api/v1/system/storage (BRAIN_UI_PROTOCOL.md). + # os_version and os_slot (#563, optional): the OS release this box runs and + # the A/B slot it booted ("A" or "B"). Set on a box in the A/B layout (the + # hosted build); left out elsewhere. The brain re-serves them on + # GET /api/v1/system/version. GET /v1/system/resources → 200 OK @@ -135,6 +139,21 @@ Three things about the payload: **200 always**, like `/v1/health/system`. Every way this can go wrong is a state in the payload, so an HTTP error would tell the brain "ask again later" about facts that are not going to change on their own. The brain re-serves it at `GET /api/v1/system/update-target`, admin-only. +**Stream A on the same read (`os`, as built #563).** Beside stream B's fields, the payload carries an `os` object, so one read says what both streams decided: + +``` + "os": { "state": "installed", "running": "0.15.0", "slot": "A", + "target": { "version": "0.15.1", "bundle_url": "https://github.com/onmoose/os/releases/download/v0.15.1/moose-v0.15.1-amd64.raucb", + "bundle_sha256": "…64 hex…" }, + "detail": "…", + "last": { "id": "os-0.15.1-1790000000", "outcome": "good", "version": "0.15.1", "from": "0.15.0", "at": "2026-10-03T03:12:00Z" } } +``` + +- **`state`**: `unsupported` (this box cannot update its OS: the appliance until #564, the fake), `none` (the answer names no OS release), `refused` (its OS part was refused, nothing downloaded), `current`, `installing`, `installed` (in the other slot, waiting for the window), `waiting` (another job holds the lock), `rebooting`, `held` (already tried tonight), `failed` (the last download or install failed; `detail` says why). +- **`running`** is `host-agent`'s own version, the version of the slot it ships in. **`target`** is the release the box picked: the next step on the way to the answer's target, or the target itself (`UPDATES.md` # 1). +- **`last`** is the last switch's outcome, `good` or `reverted`, kept on the state partition across reboots. **`id`** names it once: the brain raises one admin notification per id (`NOTIFICATIONS.md` # Updates). +- The loop decides once per tick; between ticks host-agent refreshes `installing`, `installed`, `waiting` and `failed` from the job and its record, so a finished install is reported at once. + **Health findings report (`GET /v1/health/system`).** The brain can't read host hardware directly (it's containerized behind the socket-proxy), so all *physical* health detection — SMART, `statfs`, mount flags, `systemctl is-active`, memory pressure, the pending-reboot flag (`/var/run/reboot-required`) — is host-agent's job. host-agent samples on its own cadence and the brain polls this one report on the 60s heartbeat, reconciling findings into typed health issues (`HEALTH.md` # Detector catalog, locus B). It returns findings across domains (storage, drives, services, resources, time, system) in one payload — **not** a proliferation of per-domain endpoints — so the brain's `ApplyFindings(category, …)` reconcile can clear-absent / raise-present per category atomically. This supersedes the slice-1 single-purpose storage report (`/run/moose/health/storage.json` boot reporter stays; the polled endpoint generalizes). See `DECISIONS.md` 2026-05-29. ``` @@ -395,7 +414,7 @@ When `completed`, the response carries a `result` field with the operation's out **Why not "everything is a job":** read-only / fast routes don't need the cognitive overhead of "is this done? where's the result?" and the extra JSON nesting. The dividing line is explicit per route and documented in the API contract. -**As built (#381): one kind, and a subset of the machinery.** `system-update` is the only job kind that exists today (`internal/hostagent/jobs.go`). The framework above — a kind registry with typed attributes, resource-class serialization with queue positions, cancel, and the SSE log stream of Pattern C — is **not built**, because it would be an abstraction with a single consumer (`CLAUDE.md` # Go code discipline). What is built is the part one dangerous job needs: +**As built (#381): one kind, and a subset of the machinery.** (Since #563 there are three kinds under the same lock: `os-install` and `os-switch` run the OS update, started by host-agent's own update loop, never over the socket. They take the one lock, so no OS job ever overlaps a control-plane update or another OS job. They carry no `result`; a failure ends `failed` with the error text. The rest of this paragraph is as #381 built it.) `system-update` was the only job kind (`internal/hostagent/jobs.go`). The framework above — a kind registry with typed attributes, resource-class serialization with queue positions, cancel, and the SSE log stream of Pattern C — is **not built**, because it would be an abstraction with a single consumer (`CLAUDE.md` # Go code discipline). What is built is the part one dangerous job needs: - `POST /v1/jobs/system-update` with `{ "brain_image"?, "ui_image"? }` → `202` `{job_id, status, kind, started_at}`. An empty ref means "leave that component alone"; both empty is a `400`. `501` when this host-agent has no updater wired (the fake binary), the same degrade as `journal_follow`. - `GET /v1/jobs/{id}` → the record: `status`, `started_at`, `finished_at`, plus `error` `{code, message}` and a `result` `{brain_changed, ui_changed, reverted, failure_mode, revert_error}` once it ends. `404` for an unknown id. @@ -539,6 +558,8 @@ Protocol-shaped rules about *when and how* the protocol is exercised. Not new pr 1. Brain stops accepting new jobs. 2. Brain waits for running jobs to drain. Hard cap (5 minutes): if a job is still running, tonight's attempt ends **not applied**, with "an operation is still running". The box stays on its current slot, nothing reverts, and the next window tries again. It counts as tonight's one attempt (`UPDATES.md` # 1). 3. The box switches slots and reboots; the new slot's `host-agent` starts at boot. + +**As built (#563):** steps 1 and 2 are host-agent's job lock, not a brain drain. The switch is a job under the one lock, so it never starts while a control-plane update runs, and stream B goes first in the window. The brain has no long-running job of its own that must finish first: app auto-update is not built. Steps 3 and 4 are as written. 4. Brain reconnects with backoff; resumes. Brain treats "host-agent unreachable" during this window as expected, not as an error. diff --git a/docs/specs/BUILD.md b/docs/specs/BUILD.md index 5c5ad542..7bcf7010 100644 --- a/docs/specs/BUILD.md +++ b/docs/specs/BUILD.md @@ -119,7 +119,7 @@ Both images, hosted and appliance, configure Docker with a **daemon-wide `userns The box's OS is one image written into one of two slots (`UPDATES.md` # 1, `DECISIONS.md` 2026-10-01). This section is the image's shape: what is on the disk, how it boots, and where per-box state lives. The spike in #485 proved the shape on both firmwares (`../progress/ab-update-engine-spike.md`). -> **Status: the hosted image is built in this layout (#561), with no update yet.** # As built (#561), for the hosted image below says what it does and lists the state inventory. The signed bundle is built too (#562, # The bundle). The update itself (install, mark good, revert) is #563, and the appliance image (`dev/test-qemu/`) still builds a single root until #564. The OS package lock for the hosted image is built too (# The OS package lock, #560). +> **Status: the hosted image is built in this layout (#561) and updates itself (#563).** # As built (#561), for the hosted image below says what it does and lists the state inventory. The signed bundle is built too (#562, # The bundle), and `host-agent` installs it, switches, marks good and reverts (#563, `UPDATES.md` # 1 # As built). The appliance image (`dev/test-qemu/`) still builds a single root until #564. The OS package lock for the hosted image is built too (# The OS package lock, #560). ### Partition layout @@ -161,7 +161,7 @@ Every byte the OS reserves is taken from every box for life, so the budget is me GRUB boots the box under UEFI (`x86_64-efi`) and under legacy BIOS (`i386-pc`), from **one** `grub.cfg` and **one** `grubenv` on the ESP. This replaces the hosted image's split of systemd-boot for UEFI and GRUB for BIOS. - The slot choice is RAUC's GRUB backend: `ORDER`, `_OK` and `_TRY` in `grubenv`. GRUB boots the first slot in `ORDER` that is good and not yet tried, and sets its `TRY` before booting it. `rauc status mark-good` clears it. A slot that never gets there is skipped on the next boot. **One attempt per update.** -- When no slot is left to try, GRUB falls back to the **known-good slot** rather than stopping at a prompt nobody will see: the last good slot in `ORDER`, because an install puts the new slot first and the slot it came from last. **Only the fallback slot's `TRY` is cleared.** A slot that failed keeps `TRY=1` and is never booted again until a new install into it resets its flag (#563). +- When no slot is left to try, GRUB falls back to the **known-good slot** rather than stopping at a prompt nobody will see: the last good slot in `ORDER`, because an install puts the new slot first and the slot it came from last. **Only the fallback slot's `TRY` is cleared.** A slot that failed keeps `TRY=1` and is never booted again until a new install into it resets its flag. As built (#563), the old slot's `host-agent` also marks a failed slot bad (`OK=0`), and the switch to a newly installed slot (`rauc status mark-active other`) sets its `OK=1 TRY=0`. - The kernel and initramfs live **inside** each slot, where GRUB reads them from the slot's squashfs (`squash4` and `xzio`). A slot is one self-contained unit: the kernel always matches the root it boots. - A slot that hangs instead of rebooting would never revert, so the image reboots on emergency and rescue, sets `panic=`, and runs a watchdog where the machine has one (`UPDATES.md` # 1). - **Appliance, Secure Boot:** shim plus Debian's signed GRUB, which limits the modules and files GRUB may load. That this `grub.cfg` works under it is not proved yet (`NEXT.md` # A/B OS image). @@ -186,10 +186,10 @@ Three rules follow from the overlay, and every image change must respect them: ### As built (#561), for the hosted image -`dev/cloud/` builds this layout. No box updates its OS yet: nothing installs a bundle (#562, #563) and nothing marks a slot good. +`dev/cloud/` builds this layout. Since #563 `host-agent` installs a bundle into the other slot, switches in the window and marks a slot good (`UPDATES.md` # 1 # As built); the rest of this list is as #561 built it. - **Partitions.** The image holds the ESP (128M, label `esp`), the BIOS boot partition (1M) and slot A (1G, label `moose-slot-a`, PARTUUID `20202020-2020-4020-8020-202020202020`, the UUID the single root had before) (`dev/cloud/mkosi.repart/`), about 1.13 GiB in all. Slot A is `Format=squashfs` with `Compression=xz`. The image is built with 512-byte sectors (`SectorSize=512` in `mkosi.conf`) so the 128 MiB ESP can be FAT32: systemd-repart always formats an ESP as FAT32, and with its default 4096-byte filesystem sectors FAT32 needs about 260 MB, so at 128 MiB UEFI firmware saw no filesystem on it. With 512-byte sectors it needs about 36 MB. Hosted VM disks use 512-byte logical sectors; a disk with 4096-byte logical sectors could not read this ESP, which matters for the appliance (#564), not for hosted. The runtime definitions in `/usr/lib/repart.d/` add slot B (1G, empty, label `moose-slot-b`, PARTUUID `21212121-2121-4121-8121-212121212121`) and the state partition (label `moose-state`, ext4, the rest of the disk) at first boot. The stock `systemd-repart.service` is masked, so the initramfs is the one place repart runs. A test in `dev/cloud/slotbudget` checks that the build-time and runtime definitions agree on every size. -- **Boot.** GRUB for both firmwares (`Bootloader=grub`, `BiosBootloader=grub`), from `/grub/grub.cfg` and `/grub/grubenv` on the ESP. The first grubenv is `ORDER="A B" A_OK=1 A_TRY=0 B_OK=0 B_TRY=0`. Each menu entry loads `squash4` and `xzio` and boots `/usr/lib/moose/boot/vmlinuz` and `initrd.img` from its slot with `ro console=tty0 console=ttyS0 psi=1 panic=10 fsck.mode=skip rauc.slot= root=PARTUUID=`. `fsck.mode=skip` is there because the slot has nothing to check, and state-setup checks the state partition itself. Until #563 marks a slot good, GRUB sets `A_TRY=1` on one boot and its fallback resets it on the next, so the box always boots slot A. **The ESP holds no kernel.** The postinst moves the kernel from `/usr/lib/modules//` into `/usr/lib/moose/boot/`, so mkosi (with `Bootable=auto`) finds no kernel to copy to the ESP and adds no menu entries; `grub.cfg` is the whole menu. The ESP holds GRUB's EFI binary, its modules for both firmwares, `grub.cfg` and `grubenv`. +- **Boot.** GRUB for both firmwares (`Bootloader=grub`, `BiosBootloader=grub`), from `/grub/grub.cfg` and `/grub/grubenv` on the ESP. The first grubenv is `ORDER="A B" A_OK=1 A_TRY=0 B_OK=0 B_TRY=0`. Each menu entry loads `squash4` and `xzio` and boots `/usr/lib/moose/boot/vmlinuz` and `initrd.img` from its slot with `ro console=tty0 console=ttyS0 psi=1 panic=10 fsck.mode=skip rauc.slot= root=PARTUUID=`. `fsck.mode=skip` is there because the slot has nothing to check, and state-setup checks the state partition itself. Since #563 `host-agent` marks the booted slot good at every start (a trial boot only once the brain is healthy), so `TRY` is back to 0 on every boot that came up. **The ESP holds no kernel.** The postinst moves the kernel from `/usr/lib/modules//` into `/usr/lib/moose/boot/`, so mkosi (with `Bootable=auto`) finds no kernel to copy to the ESP and adds no menu entries; `grub.cfg` is the whole menu. The ESP holds GRUB's EFI binary, its modules for both firmwares, `grub.cfg` and `grubenv`. - **Early setup, in the initramfs.** The slot's initramfs is Debian's initramfs-tools (`linux-image-amd64` already depends on it), built by `mkinitramfs` in `mkosi.postinst.chroot`. `squashfs`, `overlay` and `ext4` are added to its module list. Its `local-bottom` hook (`/etc/initramfs-tools/scripts/local-bottom/moose-state`) runs after the slot is mounted read-only and before systemd: it mounts `/proc`, `/sys`, `/dev` and a `/run` tmpfs into the slot and runs `/usr/lib/moose/state-setup` chrooted into it, so the tools are the slot's own. The script finds the boot disk (the parent disk of the device behind `/`, through sysfs), runs `systemd-repart` on that disk, and looks for the `moose-state` partition on that disk only: a partition with the label on any other disk is never touched, and no match on the boot disk is fatal. It then checks the state partition with `e2fsck -p`, mounts it at `/state`, grows its ext4 with `systemd-growfs`, mounts the `/etc` overlay (`upperdir=/state/etc/upper`), pins the four files on the first boot (marker `/state/etc/.moose-pinned`), and makes the bind mounts (`/home` from `srv/moose/home`). Any failure panics, and `panic=10` reboots, so a slot never comes up without its state. - **Copy once.** One rule for every bind mount: when a state directory does not exist yet, the slot's directory is copied into it first (to a temporary name, then renamed), and then bound. So what the image baked under a bound path reaches a new box: the control-plane image tarballs, `control-plane/compose.yml` and `caddy.json` under `/var/lib/moose`, and the systemd state under `/var/lib/systemd`. The copy costs disk: the tarballs are 383 MB on the state partition (# Disk budget). **After the first boot the state copy is the one in use, and a newer slot's baked copy stays hidden.** That is right for the control plane: stream B owns it after first boot, and it updates through the control-plane update, not the OS. The work that bakes the last released control plane into the image (#566) must keep this rule in mind: a baked control plane only reaches boxes made from that image. - **The slot is read-only** (a squashfs, `ro` on the command line, and no root line in `/etc/fstab`). The boot lane fails a boot that ends with a failed unit, or whose journal shows a host write that hit the read-only slot (`cloud-assertions.sh` # ok), so a write the inventory missed is found in CI, not on a box. @@ -233,7 +233,8 @@ One **RAUC bundle** per OS release, in the `verity` format: the slot image in a - **The keyring.** `/etc/rauc/keyring.pem` is staged into the generated wiring tree (`dev/cloud/rauc.sh`, `MOOSE_RAUC_KEYRING`). An image built to publish the OS line carries only the **release root CA**, `dev/release/rauc/release-ca.pem`. Every other build carries only a **throwaway root** made once per checkout under `.dev/rauc/throwaway/`. The key is never a file in the repo, and no image that ships trusts the throwaway. The boot lane checks that the keyring is there, that it comes from the slot and not the `/etc` upper layer (so a new image's keyring reaches the box), and that `system.conf` asks for `check-purpose=codesign`. Without that line RAUC wants the S/MIME purpose and refuses a code-signing cert. - **Signing and its checks.** The build job always signs with the throwaway signer, and then checks the bundle against the `system.conf` and keyring read back out of the slot with `unsquashfs`, not against the copies in the repo: the slot carries the keyring the build staged; the bundle verifies against the throwaway root; with the throwaway keyring the image accepts it, and with the release keyring it must **refuse** it; and an unrelated CA (a wrong key) is refused. It also rehearses the `sign` job's re-signing with a throwaway "release" CA, so that script runs on every build. A **`sign` job** of its own, after every boot passed, alone re-signs the bundle with the release signer (`dev/release/sign-bundle.sh`, `rauc resign`; the payload stays the same bytes), but only a bundle whose sha256 matches the one the build job reported, and checks it against the image's own config and keyring. The publish job then attaches it, after checking the digest `sign` reported. So the signature vouches for what the build job produced; the box's check of the digest its update target names (#563) is the control for a bad build. It refuses when the signer secrets are empty, when the image keyring is not the committed release root, or when the image would accept the throwaway-signed bundle. - **Key custody** (`DECISIONS.md` 2026-10-02): an offline root CA with the maintainer, and a signer it issued as the secrets `RAUC_SIGNING_CERT` and `RAUC_SIGNING_KEY` of the GitHub Environment `os-release` (only `main` and `v*` tags; only the `sign` job enters it). No CRLs. Making the root, issuing and rotating a signer and replacing the root are in `docs/dev/rauc-signing.md`. **Until `release-ca.pem` is committed, `release.yml` tags nothing for a merge that bumps `VERSION`** (on either line, even when the merge bumps `CONTROL_PLANE_VERSION` too), and a dispatch that publishes the OS fails at its first step. Until the secrets exist, the `sign` job refuses and nothing is published. The exact rule is in `docs/dev/contributing.md` # Release model. -- **Not proved here:** `rauc install` into slot B. RAUC's `raw` handler taking `rootfs.img` is first exercised by the installer's boot (#563). +- **Installed by `host-agent` since #563.** The `os-update` boot proves `rauc install` into slot B (RAUC's `raw` handler takes `rootfs.img`), the switch and the trial boot, under both firmwares; `os-revert` proves the revert (`docs/dev/hosted-boot-proof.md`). RAUC's install leaves the boot order alone (`activate-installed=false` in `system.conf`): `host-agent` switches with `rauc status mark-active other` only inside the window, and checks the grubenv reads `OK=1 TRY=0` for the new slot before it reboots (the stale try flag of #570). +- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`), its `host-agent` never marked it good, so it marks the slot bad and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 3 minutes (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15. ### The OS package lock @@ -457,7 +458,7 @@ The case that matters is an **upstream Caddy security release**: bump `CADDY_IMA - `moose-vX.Y.Z-amd64.qcow2` — the **cloud VM image** (priority target; the hosted product provisions tenants from it — `ENVIRONMENT.md` # Provisioning). Emitted by mkosi `Format=disk`. - `moose-vX.Y.Z-amd64.raw` — the **bare-metal install medium**, `dd`'d / flashed to a USB stick (the "old laptop in the pantry" path). Same mkosi `Format=disk` rootfs; not optical media (no `.iso` — see # 2's 2026-06-17 resolution and `DECISIONS.md`). -- `moose-vX.Y.Z-amd64.raucb` + `.sha256` — the **OS update bundle** (# 1b # The bundle), signed by the release signer, what a running box downloads to update its OS. Built and attached since #562; nothing installs it yet (#563). `host-agent` ships inside it and inside the disk images; there is no `.deb` and no apt repo (# 4). +- `moose-vX.Y.Z-amd64.raucb` + `.sha256` — the **OS update bundle** (# 1b # The bundle), signed by the release signer, what a running box downloads to update its OS. Built and attached since #562; a hosted box installs it since #563, once its update target names it. `host-agent` ships inside it and inside the disk images; there is no `.deb` and no apt repo (# 4). - `registry.onmoose.io/moose/brain:vX.Y.Z` — the brain image, where `X.Y.Z` is the control-plane version. `latest` tag advances on stable channel. **The image tag has no `control-plane-` prefix**, only the git tag and the GitHub Release do: `vX.Y.Z` is the image tag shape every release before the split used, and the private control plane resolves digests by it. - `registry.onmoose.io/moose/ui:vX.Y.Z` — the dashboard image. Same control-plane `vX.Y.Z` as the brain; both bundled in the ISO for offline first-boot. - **The control-plane images are published publicly**, and `registry.onmoose.io` is a name we can point wherever later (the first realization is `ghcr.io/onmoose/…`, which costs nothing and has no egress bill for public packages). Public rather than private+credential because there is nothing to protect: the brain and UI are built from this public repo, and every secret a box holds is per-box and seeded at provision time (`ENVIRONMENT.md` # Provisioning), never baked into an image. A private registry would buy no confidentiality and would put a pull credential on every box — one more thing to seed, rotate, and fail at 03:00 on a machine nobody can SSH into. Boxes pull **by digest**, not by tag, using the same pinning the app installer already uses (`APP_LIFECYCLE.md`), so a public registry does not mean a mutable one. diff --git a/docs/specs/ENVIRONMENT.md b/docs/specs/ENVIRONMENT.md index b582a1e9..256bab8c 100644 --- a/docs/specs/ENVIRONMENT.md +++ b/docs/specs/ENVIRONMENT.md @@ -217,7 +217,7 @@ Owned by `UPDATES.md` # 8; indexed here because it is a hosted delta. The shape: - **The target version is per-box, held by the cloud control plane.** A hosted box never fetches the signed release manifest (`RELEASE_MANIFEST.md` is appliance-only). It asks the cloud what it should be running, on its existing **outbound** path — the cloud never connects in, consistent with the rest of the hosted network posture (# Networking & discovery, # Provisioning's metadata-egress block). - **Updates are pushed, not prompted.** We operate the box, so it patches itself in the window and the tenant admin is notified afterwards. The **app permission-expansion prompt is the carve-out** and still goes to the instance owner: operating someone's box does not transfer their data-access decisions to us. - **Everything else is shared with appliance** — both update streams, the apply transaction, the snapshots, the health-check-then-revert, the retention window. -- **Stream A is the A/B OS image on hosted too** (`UPDATES.md` # 1): the answer will name the OS releases on the way to the target (one per minor, `UPDATES.md` # 1) with their bundle digests, beside the brain and UI references. Planned (#486). Until it is built, a hosted box never updates its OS: the image carries no `apt`. +- **Stream A is the A/B OS image on hosted too** (`UPDATES.md` # 1): the answer names the OS releases on the way to the target (one per minor, `UPDATES.md` # 1) with their bundle digests, beside the brain and UI references. **The box side is built (#563)**, `UPDATES.md` # 1 # As built. The control plane does not send the OS part yet (a change on the private side), so until it does a hosted box stays on the OS it was built with: the image carries no `apt`. - **Auth for the box↔cloud channel is not designed yet.** The `enrollment` block in `seed.json` (# Owner sign-in & seed ingestion — as built) is scoped to acme-dns, not to a general control-plane API. Tracked in `NEXT.md`. - **Which target a box reads, and the window it applies in, are meant to be per-box** — there is no SSH (# Access & files), so nothing on the box can be hand-edited after boot. As shipped (#404) both are systemd credentials, which means neither is settable on a real box at all; #407 moves the target URL into the seed (fixed at create, which suits it) and #408 moves the window into the control plane's answer (changeable at any time, which the window needs). Note the downgrade property in `UPDATES.md` # 8.4: a box moves to whatever its target names, including backwards. diff --git a/docs/specs/NOTIFICATIONS.md b/docs/specs/NOTIFICATIONS.md index 58aee9e0..a0f01a30 100644 --- a/docs/specs/NOTIFICATIONS.md +++ b/docs/specs/NOTIFICATIONS.md @@ -126,7 +126,7 @@ The curated set of events that produce a notification. New entries are added by |---|---|---| | OS / host-agent / brain+UI update available | info | **Admin only** | | System update applied; reboot pending > 7 days | info | Admin | -| OS update applied, or reverted to the previous release (planned with the A/B OS image, #486; replaces the row above) | info, or warning when reverted | Admin | +| OS update applied, or reverted to the previous release (built on hosted, #563: one notification per outcome, read off host-agent's update-target report; replaces the row above) | info, or warning when reverted | Admin | | App auto-updated overnight | info | **Owner** (Tier-2 → Admin) | | App update needs permission approval | warning | **Owner** (mirrors the existing on-login modal) | | App update failed / rolled back | error | **Owner** (Tier-2 → Admin) | diff --git a/docs/specs/TESTING.md b/docs/specs/TESTING.md index 3751ecfc..22200f70 100644 --- a/docs/specs/TESTING.md +++ b/docs/specs/TESTING.md @@ -92,6 +92,8 @@ The hosted profile (`ENVIRONMENT.md`) has its own full-stack lane (`dev/cloud/ru A pair of **`remap` boots** (#531), on one fresh overlay + box-id of their own and seeded with a test-portal key like the `access` boot, is the net under the other user-namespace tiers (`APP_ISOLATION.md` # User-namespace tiers). The first, `remap`, installs one synthetic app per tier and checks the userns mode, the capabilities and the host owner of its data for each: `remapdrop` (folderless, default tier, root inside, so its process and its data are host uid `1000000`), `svcdrop` (`service_user`, default tier, at `1000000` plus its allocated uid), `rootsetup` (`root_setup`, caps tier: remapped, no `user:`, exactly the five capabilities, and a start that chowns its data as root and then drops to `www-data`, at `1000033`), `pgnote` (default tier, on the managed Postgres 16, which runs remapped with its data at `1000999`, and pgnote's row is read back from its database) and `filedrop` household (a folder app, `userns_mode: host`, the real `2000:2001` on the shared tree). It then recreates the caps-tier container and checks that its data is kept. The second, `remap-reboot`, is a real reboot of the same disk that runs every check again on what the box brought back, and checks that the caps-tier data survived. Both check that the new SSO owner has no range in `/etc/subuid` or `/etc/subgid`. The postgres image is a test-only tar that only the first remap boot loads. The pair is in the publish gate. A manual `CI / Cloud image` run can run it alone with `-f boots=remap -f publish=false` (a run with its own boot list refuses to publish). +Two **OS update boots** (#563), each on a fresh overlay + box-id of its own, are the net under stream A (`UPDATES.md` # 1). They are the only boots that reboot their box inside one QEMU run. `os-update` installs a test bundle (the boot-proof slot with a host-agent one patch release up, `dev/cloud/test/build-os-test-bundle.sh`) into slot B after refusing a wrong digest, switches in the window, boots slot B, marks it good, and checks the owner, an app, the SSH host keys, `machine-id` and a data file are unchanged. `os-revert` installs the same bundle with host-agent kept off slot B, so the image's trial timer reboots the box and it comes back on slot A on its own. Both run under both firmwares, in every full run; a run that publishes the OS runs only the refusal half of `os-update` (`../dev/hosted-boot-proof.md` # The OS update boots). + ## Slow lane — Soak + ISO end-to-end (nightly on `main`) **What it catches:** diff --git a/docs/specs/UPDATES.md b/docs/specs/UPDATES.md index c30f005c..7b548e28 100644 --- a/docs/specs/UPDATES.md +++ b/docs/specs/UPDATES.md @@ -8,7 +8,7 @@ This doc is **draft / option-survey**. Most sections present alternatives with a A box has **two update streams**. The split is not by component, it is by **what the thing runs on**: -- **Stream A — the box.** Debian base, kernel, firmware, and `host-agent`. This is the machine itself. It is slow, it needs a reboot, and it is **one atomic unit**: you do not get to have a new kernel with an old `host-agent`. That unit is an **A/B OS image** (# 1, # 2; `DECISIONS.md` 2026-10-01), designed and not yet built (#486). Its version is the moose release. +- **Stream A — the box.** Debian base, kernel, firmware, and `host-agent`. This is the machine itself. It is slow, it needs a reboot, and it is **one atomic unit**: you do not get to have a new kernel with an old `host-agent`. That unit is an **A/B OS image** (# 1, # 2; `DECISIONS.md` 2026-10-01), built on hosted (#561 to #563) and not yet on the appliance (#564). Its version is the moose release. - **Stream B — the containers.** `moose-brain`, `moose-ui`, apps, and managed services. These are images. They are frequent, they need no reboot, and each one already carries its own rollback story, because "keep the old image" is what a container registry is for. Within stream B the components still have their own policies — the control plane is admin-triggered or cloud-pushed, apps auto-apply unless permissions expand, managed services are invisible infrastructure. Those policies are in # 3, # 4, and # 5 below and are unchanged by the two-stream framing. What the framing fixes is the **unit of testing and the unit of rollback**: two streams means we ship and verify two combinations, not the cross-product of five. @@ -31,7 +31,7 @@ We are not copying their mechanism (they are image-based appliances; our stream The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host-agent` (# 2). Since `DECISIONS.md` 2026-10-01 this is **one A/B image**, on both profiles. There is no `apt` on the update path and no `unattended-upgrades`. The image layout, the per-box state rules and the build are in `BUILD.md` # 1b; this section is the update policy. -> **Status: half built (#486).** The hosted image is in the A/B layout (#561), and every OS release publishes a signed bundle (#562, `BUILD.md` # 1b # The bundle): step 2's signature check has its keyring and its release signer (`DECISIONS.md` 2026-10-02). Nothing installs a bundle yet (#563), so every box still never updates its OS. The engine was chosen by the spike in #485 (`../progress/ab-update-engine-spike.md`). Existing boxes are not migrated to the A/B layout: they are replaced. +> **Status: built on hosted (#563); the appliance waits for #564.** The hosted image is in the A/B layout (#561), every OS release publishes a signed bundle (#562, `BUILD.md` # 1b # The bundle), and `host-agent` applies it (#563, # As built below). The appliance image is not A/B yet, so its `host-agent` has no OS applier and reports stream A as `unsupported`. The engine was chosen by the spike in #485 (`../progress/ab-update-engine-spike.md`). Existing boxes are not migrated to the A/B layout: they are replaced. **The OS part of the hosted update-target answer is a change on the private side that is not made yet** (# As built), so until the control plane sends it no hosted box moves its OS. ### The update transaction @@ -54,8 +54,21 @@ The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host- **An OS downgrade by target follows the same reach.** The update loop applies whatever the target names, in either direction (# 8.4). For the OS, the box refuses a target that is more than one minor back (from 1.4.y: any 1.4 or 1.3 patch, never 1.2), and a target below the running control plane's floor (`minimumAgentVersion`, # 7 Compatibility matrix), since that would boot a `host-agent` the brain refuses to work with. A refusal is logged and changes nothing, like any other refused target. This is the same discipline the brain's own rollback already needs (`NEXT.md` # Brain state-migration framework). +**Which boots are on trial** (the maintainer's call, #563). Only the **first boot after an OS switch** is on trial: it is marked good once the brain answers, or given up. Every other boot is marked good as soon as `host-agent` starts. So a problem that has nothing to do with the OS (a broken control-plane update, a slow app) never moves a box back to its older OS, and a box with only one good slot never loops through reboots. + **A slot that hangs must still revert.** Boot counting only helps a slot that reboots. The spike saw a slot stop in the initrd's emergency shell and wait for ever. So the image reboots on emergency and rescue, sets `panic=` on the kernel command line, and runs a watchdog where the machine has one (`NEXT.md` # A/B OS image). +### As built (#563), on hosted + +`internal/hostagent/osupdate` is the transaction, `internal/hostagent/updatetarget` decides which release and when. Both run in `host-agent`; the brain only reports. + +- **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start, and a box with no such file has no floor to check. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. +- **Install, ahead of the window** (an `os-install` job, under the same lock as `system-update`). The bundle is downloaded to `/var/lib/moose/os-update/` on the state partition and its sha256 checked against the answer before RAUC sees it. Then `rauc install`, which checks the signature and writes the other slot. `system.conf` has `activate-installed=false`, so the boot order does not move. The file is deleted afterwards. A failed attempt is not repeated the same night for the same release and digest. +- **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. One switch per release per night. +- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad and stays. The same release is tried again the next night. +- **Report.** `GET /api/v1/system/version` carries `os_version` and `os_slot`; `GET /api/v1/system/update-target` carries an `os` object beside stream B's fields (`BRAIN_HOST_PROTOCOL.md`). The brain raises one admin notification per outcome (`NOTIFICATIONS.md` # Updates). Reporting back to the cloud (# 8.4 step 5) still waits for real box authentication. +- **What it costs a box** (CI, `../progress/host-agent-os-update.md`): one bundle download per update (438.5 MB for a real release), the same space on the state partition while it installs (1.1% of a 40 GB disk), and one reboot. + ### Policy: automatic on both profiles **The OS update applies by itself and reboots in the window, on appliance and on hosted** (`DECISIONS.md` 2026-10-01). An A/B update only takes effect after a reboot, so a box that waited for a click would never be patched. The automatic revert is what makes this safe; it is the rollback this doc used to say auto-apply was waiting for. @@ -356,13 +369,13 @@ Telemetry is a **signal that accelerates our reaction time**, not a gate. Boxes | Stream | Component | Rollback mechanism | |---|---|---| -| A — the box | The OS image, `host-agent` inside it | The previous slot, automatic when the new slot is not marked good (# 1). Designed, not built (#486) | +| A — the box | The OS image, `host-agent` inside it | The previous slot, automatic when the new slot is not marked good (# 1). Built on hosted (#563); the appliance with #564 | | B — containers | `moose-brain` + `moose-ui` | Previous image pair + SQLite snapshot; revert as a pair, automatic on health-check fail of either | | B — containers | App | Previous image + pre-update tar of `data_volumes` (+ `pg_dump` of managed-service DB if any), automatic on health-check fail; keep 7 days | | B — containers | Managed service (patch) | Previous image; data is shared so this is a tag-flip | | B — containers | Managed service (major migration) | Pre-migration dump, automatic on app-update fail | -The two streams roll back in different ways, and both do it on their own: stream B keeps the previous image, stream A keeps the previous slot. Until the A/B image is built, stream A has no rollback because it has no update at all (# 1 Status). +The two streams roll back in different ways, and both do it on their own: stream B keeps the previous image, stream A keeps the previous slot. On the appliance, until its image is A/B (#564), stream A has no rollback because it has no update at all (# 1 Status). --- @@ -439,7 +452,7 @@ The ledger is not bookkeeping. `host-agent` leaves an existing brain container a - **An unusable target is refused, not resolved away (#404, #407).** If the seed's `update_target_url` is set but is not an absolute `http` or `https` URL with a host, host-agent logs an error and **does not start the update loop at all**. It does not fall back to the fleet endpoint: a box that was deliberately pinned to a candidate must never quietly join stable because of a typo. The box keeps serving whatever it is already running. **A seed that will not parse is refused the same way**, because the bytes we could not read might have carried a target and we cannot tell; a seed that is simply absent is not an error and falls through, since that is the appliance and the un-steered hosted box. An unusable **window** is not treated like any of this and falls back to 03:00-04:00 with a warning, because a wrong hour can only apply an update at the wrong time, while a wrong target sends the box to the wrong version. - **The loop applies whatever the target names, in either direction. A box running something newer than its target rolls back to it.** The compare is "does the running pair differ from the target pair", with no notion of "forward", so setting a box's target to an older release is how a deliberate downgrade is performed, and pointing a box at a stale target is how one happens by accident. This is correct (per-box pinning wants deliberate downgrades) and it is surprising, so it is written here rather than left to be discovered. It nearly bit us the day the loop shipped: a box provisioned from the v0.7.0 image would have rolled itself back to v0.6.0 overnight had `stable` not been promoted the same hour. **Before pointing a box at a target, check which way it moves that box.** 3. **Stream B:** pull by digest, snapshot the brain's SQLite, write the staged compose, recreate the changed containers, health-check, revert both on failure of either (# 3). -4. **Stream A:** the A/B transaction of # 1, last in the window. The answer gains an OS version and its bundle digest beside the brain and UI references (planned, #486). A hosted VM reboot is cheaper than an appliance one: no user is physically waiting, and the window is ours. +4. **Stream A:** the A/B transaction of # 1, last in the window. **As built (#563)**, the answer carries an optional `os` list beside the brain and UI references: `"os": [{"version": "0.15.3", "bundle_url": "https://github.com/onmoose/os/releases/download/v0.15.3/moose-v0.15.3-amd64.raucb", "bundle_sha256": "<64 hex>"}, …]`, the newest patch of each minor from the oldest one still supported up to the target, which is last. Left out, the answer has no opinion about the OS and the box stays on its OS. The sha256 is the release's own `.raucb.sha256` asset. **The control plane does not send this yet**: that is a change on the private side, described for the maintainer in `../progress/host-agent-os-update.md`. A hosted VM reboot is cheaper than an appliance one: no user is physically waiting, and the window is ours. 5. The box reports the outcome back to the cloud: version now running, success or failure, and the failure mode if it rolled back. Step 5 is what the whole design is for. On appliance our visibility into a bad release is "GitHub issues and the forum, hours to days" (# 3). On hosted it is a fleet view that tells us a version is failing before the second box tries it. From 31418a831d65b40a7eb392e552ab2191c907a240 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 17:07:19 +0100 Subject: [PATCH 03/18] test: keep the in-guest update source across the OS boots' reboots (#563) --- .../updatetargetreport_test.go | 52 +++++++++++++++++++ dev/cloud/cloud-assertions.sh | 4 +- internal/api/systemupdate_test.go | 28 ++++++++++ 3 files changed, 83 insertions(+), 1 deletion(-) diff --git a/cmd/host-agent-real/updatetargetreport_test.go b/cmd/host-agent-real/updatetargetreport_test.go index 342afb8b..a5e7451a 100644 --- a/cmd/host-agent-real/updatetargetreport_test.go +++ b/cmd/host-agent-real/updatetargetreport_test.go @@ -287,3 +287,55 @@ func TestReport_RefusedTargetIsNotApplicable(t *testing.T) { t.Fatalf("target = %+v, want the refused answer verbatim", got.Target) } } + +// stubOS is stream A's facts for the report. +type stubOS struct { + peekState string +} + +func (s stubOS) Running() (string, string) { return "0.15.0", "A" } +func (s stubOS) Last() *protocol.OSOutcome { + return &protocol.OSOutcome{ID: "os-0.15.0-1", Outcome: protocol.OSOutcomeGood, Version: "0.15.0", At: "2026-10-01T03:10:00Z"} +} +func (s stubOS) Peek(updatetarget.OSRelease) (string, string, bool) { + return s.peekState, "", s.peekState != "" +} + +type stubOSApplier struct{ state string } + +func (stubOSApplier) Running() (string, string) { return "0.15.0", "A" } +func (stubOSApplier) Floor() string { return "" } +func (a stubOSApplier) Apply(updatetarget.OSRelease, bool, time.Time) updatetarget.OSDecision { + return updatetarget.OSDecision{State: a.state} +} + +func TestReportOS(t *testing.T) { + // No applier: stream A is reported, as unsupported. + r := updateTargetReport{loop: tickedLoop(t, stubSource{target: goodOffer()}, stubRunning{brain: runningBrain, ui: runningUI})} + if got := r.Read().OS; got == nil || got.State != protocol.OSUpdateUnsupported { + t.Fatalf("no applier: %+v", got) + } + + tgt := goodOffer() + tgt.OS = []updatetarget.OSRelease{{Version: "0.15.1", BundleURL: updatetarget.DefaultOSURLPrefix + "v0.15.1/b.raucb", BundleSHA256: strings.Repeat("e", 64)}} + l := &updatetarget.Loop{Source: stubSource{target: tgt}, Current: stubRunning{brain: runningBrain, ui: runningUI}, + Profile: "hosted", OS: stubOSApplier{state: protocol.OSUpdateInstalling}} + l.Tick(context.Background()) + + // The tick said installing; the job has finished since, and Peek says so. + r = updateTargetReport{loop: l, running: stubRunning{brain: runningBrain, ui: runningUI}, os: stubOS{peekState: protocol.OSUpdateInstalled}} + got := r.Read().OS + if got.State != protocol.OSUpdateInstalled || got.Running != "0.15.0" || got.Slot != "A" || got.Target == nil || got.Target.Version != "0.15.1" || got.Last == nil { + t.Fatalf("report: %+v", got) + } + // Peek with nothing newer leaves the loop's word. + r.os = stubOS{} + if got := r.Read().OS; got.State != protocol.OSUpdateInstalling { + t.Fatalf("without a newer state: %+v", got) + } + // A box with no loop still says what it runs. + r = updateTargetReport{disabledErr: "bad seed", os: stubOS{}} + if got := r.Read().OS; got.State != protocol.OSUpdateNone || got.Running != "0.15.0" || got.Detail == "" { + t.Fatalf("disabled: %+v", got) + } +} diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index aff5f2a1..8b9f81aa 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -2440,7 +2440,9 @@ UNIT # A shut window: twelve hours from now, one minute wide. shut="$(printf '%02d:00-%02d:01' $(( (10#$(date +%H) + 12) % 24 )) $(( (10#$(date +%H) + 12) % 24 )))" write_os_target "$(printf '%064d' 0)" "$shut" - docker run -d --name moose-test-target -p 127.0.0.1:5001:80 -v "$target_dir":/srv:ro "$caddy_image" \ + # --restart: the box reads its target again after each reboot, and the + # last stage checks what it decided then. + docker run -d --restart unless-stopped --name moose-test-target -p 127.0.0.1:5001:80 -v "$target_dir":/srv:ro "$caddy_image" \ caddy file-server --root /srv --listen :80 >/dev/null 2>&1 || fail "$MODE: could not start the in-guest file server" for _i in $(seq 1 60); do grep -q ' 200' <<<"$(http_status_addr 127.0.0.1 5001 /target.json 2>/dev/null)" && break; sleep 1; done diff --git a/internal/api/systemupdate_test.go b/internal/api/systemupdate_test.go index 24a11763..d5454d3f 100644 --- a/internal/api/systemupdate_test.go +++ b/internal/api/systemupdate_test.go @@ -6,6 +6,7 @@ import ( "net" "net/http" "path/filepath" + "strings" "testing" "time" @@ -376,3 +377,30 @@ func TestUpdateTarget_HostFailureIs502(t *testing.T) { _, err := h.readTarget(adminCtx("u_admin")) assertStatus(t, err, http.StatusBadGateway) } + +// Stream A rides the same read (#563): the host's os part reaches the caller +// unchanged, outcome included. +func TestUpdateTarget_OSPart(t *testing.T) { + h := newUpdateHarness(t) + h.target = protocol.UpdateTarget{ + State: protocol.UpdateTargetCurrent, + OS: &protocol.OSUpdate{ + State: protocol.OSUpdateInstalled, Running: "0.15.0", Slot: "A", + Target: &protocol.OSRelease{Version: "0.15.1", BundleURL: "https://github.com/onmoose/os/releases/download/v0.15.1/b.raucb", BundleSHA256: strings.Repeat("e", 64)}, + Last: &protocol.OSOutcome{ID: "os-0.15.0-1", Outcome: protocol.OSOutcomeGood, Version: "0.15.0", From: "0.14.9", At: "2026-10-01T03:10:00Z"}, + }, + } + out, err := h.readTarget(adminCtx("u_admin")) + if err != nil { + t.Fatalf("getSystemUpdateTarget: %v", err) + } + o := out.Body.OS + if o == nil || o.State != protocol.OSUpdateInstalled || o.Slot != "A" || o.Target == nil || o.Target.Version != "0.15.1" || + o.Last == nil || o.Last.Outcome != "good" || o.Last.From != "0.14.9" { + t.Fatalf("os = %+v", o) + } + h.target.OS = nil + if out, _ := h.readTarget(adminCtx("u_admin")); out.Body.OS != nil { + t.Fatalf("no os part on the host must leave it out, got %+v", out.Body.OS) + } +} From 3f775df6c826a60b13bb3f71241f4d37a6aa91cc Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 17:19:29 +0100 Subject: [PATCH 04/18] docs: the A/B OS update as built on hosted (#563) --- docs/progress/README.md | 3 +- docs/progress/host-agent-os-update.md | 100 ++++++++++++++++++++++++++ 2 files changed, 102 insertions(+), 1 deletion(-) create mode 100644 docs/progress/host-agent-os-update.md diff --git a/docs/progress/README.md b/docs/progress/README.md index 27144fd6..46950b19 100644 --- a/docs/progress/README.md +++ b/docs/progress/README.md @@ -20,7 +20,7 @@ The implementation slice queue, ordered. Each item links back to the progress en This is the **maintainer's critical-path** queue. Work carved off for **parallel contributors** lives in [GitHub Issues](https://github.com/onmoose/os/issues) (some items there are pulled from these "what's next" follow-ups). The two are kept from overlapping on purpose. See [`../dev/contributing.md`](../dev/contributing.md) for the contributor loop. -1. **A/B OS update (#486).** [ab-os-update-design.md](ab-os-update-design.md) designed it; [control-plane-version-line.md](control-plane-version-line.md) (#559) and [os-package-lock.md](os-package-lock.md) (#560) built the version split and the package lock. [hosted-ab-layout.md](hosted-ab-layout.md) (#561) built the hosted image in the A/B layout with the state inventory, and [rauc-bundle.md](rauc-bundle.md) (#562) a signed RAUC bundle per OS release. Before the next OS release the maintainer sets up the offline root CA and the `os-release` signer (`../dev/rauc-signing.md`); until then `release.yml` tags nothing for a `VERSION` bump. Next is **#563** (host-agent applies OS updates), then #564 (appliance). Until #563 lands, no box updates its OS. +1. **A/B OS update (#486).** [ab-os-update-design.md](ab-os-update-design.md) designed it; [control-plane-version-line.md](control-plane-version-line.md) (#559) and [os-package-lock.md](os-package-lock.md) (#560) built the version split and the package lock. [hosted-ab-layout.md](hosted-ab-layout.md) (#561) built the hosted image in the A/B layout with the state inventory, and [rauc-bundle.md](rauc-bundle.md) (#562) a signed RAUC bundle per OS release. Before the next OS release the maintainer sets up the offline root CA and the `os-release` signer (`../dev/rauc-signing.md`); until then `release.yml` tags nothing for a `VERSION` bump. [host-agent-os-update.md](host-agent-os-update.md) (#563) made host-agent apply OS updates on hosted: install into the other slot, switch in the window, trial boot, revert. **Next:** the maintainer adds the `os` part to the control plane's update-target answer (the exact wire is in that entry); until then no production box moves its OS. Then #486's proof on a provisioned box, and #564 (appliance). 2. **Persistent journal.** [container-logs-journald-driver.md](container-logs-journald-driver.md) made container logs reach journald, but neither image creates `/var/log/journal` or sets `Storage=persistent`, so the journal is volatile and app-log scrollback is lost on every reboot, while `LOGGING.md` # Per-app logs promises "scrollback up to journald's cap". Ship the journald drop-in from `LOGGING.md` # Tuning (`Storage=persistent`, `SystemMaxUse=1G`, `RuntimeMaxUse=128M`) on both images, sized against the partition `/var/log` lives on (on hosted, the state partition that grows at first boot, since #561). Also restores the size-based backpressure # Tuning assumes now that docker's per-unit rate limit is off. 3. **GPU + device capacity enforcement.** `install-permissions-enforcement.md` deferred `gpu` enforcement and device-existence validation (both need a host hardware-introspection endpoint). A 422 from the brain will surface correctly in the UI via the existing `dialogError` path ([install-consent-ui.md](install-consent-ui.md)) once the host endpoint lands. See `NEXT.md` # GPU. 4. **Tag the store manifests with `role` and `requires`.** [ai-bindings-after-install.md](ai-bindings-after-install.md) finished the `os` side of the install setup plan, but no real app shows the LLM provider pickers until its manifest in `onmoose/store` carries roles (`APP_MANIFEST.md` # D4 # Roles and requires, `docs/dev/authoring-apps-with-an-agent.md`). The last step of `INSTALL_SETUP.md`. @@ -312,3 +312,4 @@ Oldest first; append new entries to the bottom. | [hosted-ab-layout.md](hosted-ab-layout.md) — Closes #561, a slice of #486, after [docs-after-version-split.md](docs-after-version-split.md). **The hosted image is built in the A/B layout:** a 128 MiB ESP, a BIOS boot partition and slot A, a 1 GiB read-only squashfs-xz holding the kernel and initramfs, in the image; slot B (1 GiB) and a state partition (ext4, the rest of the disk, grown every boot) made by `systemd-repart` in the initramfs at first boot. The OS reserves 2.28 GB, 5.7% of a 40 GB disk. The maintainer rejected the designed 4 GiB ext4 slots (`DECISIONS.md` 2026-10-01). GRUB boots both firmwares from one `grub.cfg` and `grubenv`; the ESP holds no kernel (`Bootable=auto`) and is built with 512-byte sectors so 128 MiB can be FAT32. An initramfs-tools hook sets up the `/etc` overlay, four pinned files and the bind mounts; databases keep their own bind mounts and `/home` comes from `srv/moose/home`. `rauc` with the slot config, no keyring. **`dev/cloud/slotbudget`** fails the build when the squashfs fills over 60% of its slot (now 435.3 MB, 40.5%). host-agent's hosted build reports the state partition as "System". Every boot runs under UEFI and legacy BIOS and checks the layout; run 36932849491 passed all 14. **Gaps:** no bundle, install or mark-good (#562, #563); QEMU only; the control-plane tarballs are copied to the state partition (383 MB); the private smoke test needs a disk of 20 GB or more | done | | [ci-cloud-image-speedup.md](ci-cloud-image-speedup.md) — A slice of #486, after [hosted-ab-layout.md](hosted-ab-layout.md). **`CI / Cloud image` builds once and boots in parallel: a full run drops from 27.4 to 12.2 min** (run 37001109373). One `build` job makes both images and uploads the boot-proof image as a qcow2; a `boot` matrix runs one job per boot group and firmware (10 for the full list, `unseeded seeded` kept together because they share a disk); a `publish` job, the only one with write access, runs only after every boot passed and pushes the control-plane tarballs the image baked. Every publish guard, the PR `update`-only rule, the lock-bump full-list rule and the `boots` input are kept. The boot-proof build reuses the production tools tree (`MOOSE_MKOSI_TOOLS_TREE`, about 40 s), the QEMU packages are downloaded once and installed with `dpkg` (one slow mirror once cost a job 6 min), and runs that publish nothing build the pinned hosted Caddy image with a BuildKit `gha` layer cache (`MOOSE_BUILD_CACHE`; the brain and UI build plain, because their cache cost more than it saved). Runner minutes go from 27 to about 34, free on a public repo. **Gaps:** every new branch starts with an empty Caddy cache (about 1.2 min on its first run); no apt cache (measured at about 15 s, dropped by decision); the boot-proof image is still a second mkosi build; the publish job has not run yet | done | | [rauc-bundle.md](rauc-bundle.md) — Closes #562, a slice of #486, after [hosted-ab-layout.md](hosted-ab-layout.md) and [ci-cloud-image-speedup.md](ci-cloud-image-speedup.md). **Every OS release builds a signed RAUC bundle** (`moose-vX.Y.Z-amd64.raucb`, verity format): slot A of the image that ships, cut out by `dev/cloud/slotbudget -extract`, bundled by `dev/cloud/build-bundle.sh` with the box's own RAUC version from the snapshot, and checked against the `system.conf` and keyring read back out of the slot (the staged keyring, the throwaway root, a release keyring that must refuse the throwaway signer, a wrong key, and a rehearsal of the publish job's signing). The image bakes `/etc/rauc/keyring.pem` with `check-purpose=codesign`: the release root on an OS publish run, a throwaway root otherwise. **Key custody decided** (`DECISIONS.md` 2026-10-02): an offline root CA (`dev/release/rauc-ca.sh`, `docs/dev/rauc-signing.md`) and a signer in the `os-release` environment, so only a `sign` job of its own reads the signer, re-signs with `rauc resign` a bundle whose sha256 matches the build job's, and hands it to the publish job, which attaches the image, the bundle and their checksums as one set of four (it detects and refuses a mixed set). `release.yml` tags nothing for a `VERSION` bump until `release-ca.pem` is committed. **Numbers:** bundle 438.5 MB, 40.8% of the slot and half of an 874 MB OS release; about 1 min more per run. **Gaps:** no release path has run yet (no `release-ca.pem` committed; `release.yml` tags no OS release until the maintainer's setup); no `rauc install` proof (#563) | done | +| [host-agent-os-update.md](host-agent-os-update.md) — Closes #563, a slice of #486, after [hosted-ab-layout.md](hosted-ab-layout.md) and [rauc-bundle.md](rauc-bundle.md). **host-agent applies OS updates on hosted.** The update-target answer gains an optional `os` list; host-agent picks the next release (never skipping a minor, at most one minor back, never below the control plane's floor), downloads the bundle, checks its sha256 against the answer before RAUC sees it, installs it into the other slot ahead of the window (`activate-installed=false`), switches inside it after stream B (checking the grubenv reads `OK=1 TRY=0`), and on the next boot marks the slot good once the brain answers or reboots back; an image timer reboots a slot whose host-agent never started. Only the first boot after a switch is on trial. Stream A is reported on the version and update-target reads, and admins get one notification per outcome. New `os-update` and `os-revert` boots prove both paths under UEFI and BIOS, in every full run. Gaps: the private control plane does not send the `os` part yet (wire described for the maintainer), QEMU only, appliance waits for #564 | done | diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md new file mode 100644 index 00000000..9f2c93a2 --- /dev/null +++ b/docs/progress/host-agent-os-update.md @@ -0,0 +1,100 @@ +# host-agent applies OS updates + +- **Status:** done +- **Date:** 2026-10-02 +- **Specs touched:** `docs/specs/UPDATES.md`, `docs/specs/BRAIN_HOST_PROTOCOL.md`, `docs/specs/BUILD.md`, `docs/specs/ENVIRONMENT.md`, `docs/specs/NOTIFICATIONS.md`, `docs/specs/TESTING.md`, `docs/architecture.md`, `docs/dev/hosted-boot-proof.md`, `docs/dev/contributing.md`, `CLAUDE.md` (three log fields, approved by the maintainer) + +Closes #563, the fifth slice of #486. It follows [hosted-ab-layout.md](hosted-ab-layout.md) (#561), which built the hosted image in the A/B layout with RAUC and nothing that marks a slot good, and [rauc-bundle.md](rauc-bundle.md) (#562), which builds a signed bundle per OS release that nothing installed. Now `host-agent` installs that bundle into the other slot, switches inside the update window, keeps the new slot once the brain is healthy, and goes back to the old slot on its own when it is not. This is #486 steps 4 to 6 on the hosted profile. The appliance gets it with #564. + +## What was done + +### The answer's OS part (`internal/hostagent/updatetarget`) + +- **Wire.** The update-target answer gains an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}`. The last entry is the target. Left out, the answer has no opinion about the OS. +- **Checks** (`os.go`, `ValidateOS`): a plain `X.Y.Z` version, ascending order, a 64-hex lowercase sha256, an absolute http or https URL, and a URL that starts with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`, the same kind of check as the expected image repositories). +- **Which release** (`PickOS`), as `UPDATES.md` # 1 says: on a later minor, the first entry above the box's own minor (a box never skips a minor); on its own minor or one back, the target; further back, refused. A release below the running control plane's floor is refused too. The brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start for this; a box without the file has no floor to check. +- **Each stream is judged on its own.** A bad OS part refuses stream A only. A refused control-plane part, or a box that cannot read its running pair, still lets stream A go on. The test answer of the new boots carries only an OS part, which proves it. +- **Order in the window.** The loop runs stream B first. When stream B starts an update in this tick, or cannot because a job runs, stream A waits for the next tick. + +### The transaction (`internal/hostagent/osupdate`, new) + +- **Install, ahead of the window** (`os-install` job). Download to `/var/lib/moose/os-update/bundle.raucb.part` on the state partition, hashing as it writes. Only a file whose sha256 is the one the answer names is renamed and handed to `rauc install`, so RAUC never sees a bundle the answer did not name. RAUC then checks the signature against the image's keyring. `system.conf` now has `activate-installed=false`, so the install leaves the boot order alone. The file is deleted afterwards. A failed attempt is not repeated for the same release and digest the same night. +- **Switch, inside the window** (`os-switch` job). Check that RAUC reports the other slot holds the release, write the record, write the trial marker `/var/lib/moose/os-update/trial-`, run `rauc status mark-active other`, and check the grubenv reads `ORDER=" "`, `_OK=1`, `_TRY=0` before rebooting. RAUC 1.13's GRUB backend already resets `TRY` there (its `grub_set_primary`, read in the source), so the #570 gap is now a check, and a failing check puts the booted slot first again and does not reboot. One switch per release per night, kept in the record across the reboot. +- **The trial, at the next start of `host-agent`** (`Boot`). With the booted slot's marker: wait for the brain's `/healthz` up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min); healthy means `mark-good`, the marker goes, the outcome is `good`; not healthy means `mark-bad booted` and a reboot. Without a marker: `mark-good` at once. With the other slot's marker: that slot failed, so the outcome is `reverted`, the slot is marked bad, the marker goes. **Only the first boot after a switch is on trial** (the maintainer's call). +- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker, its `host-agent` never got there, so the script marks the slot bad and reboots. The boot-proof image sets it to 3 minutes. +- **Jobs.** `hostagent.Agent.StartJob` runs any job under the one lock `system-update` takes. The OS jobs carry no result. +- **Record.** `state.json`: what was installed where, the last install attempt, the last switch, the last outcome with an id. Fields only grow, because either release may read it. + +### Reporting + +- `GET /v1/system/status` and `GET /api/v1/system/version` carry `os_version` and `os_slot`. The OS version is `host-agent`'s own version, which is the version of the slot it ships in. +- `GET /v1/system/update-target`, and `GET /api/v1/system/update-target`, carry an `os` object: `state` (`unsupported`, `none`, `refused`, `current`, `installing`, `installed`, `waiting`, `rebooting`, `held`, `failed`), `running`, `slot`, `target`, `detail` and `last`. The loop decides once per tick, so between ticks `host-agent` refreshes the job states from the job and the record (`Peek`). +- **Admin notification.** A brain loop on the health-poll cadence reads `os.last` and raises one notification per outcome id: info "moose updated its system to X", warning "A system update did not work, so moose went back". It checks the store for the dedup key first, because a re-raise would mark a read notification unread again, and skips outcomes older than 7 days, so a pruned row is not raised again. +- The fake `host-agent` reports `unsupported`, or a canned state from `MOOSE_FAKE_OS_UPDATE`. The appliance build has no OS applier and reports `unsupported`. +- `CLAUDE.md` gains `os`, `slot` and `digest` in the standard log fields. + +### The boots (`os-update`, `os-revert`) + +- **The test bundle** (`dev/cloud/test/build-os-test-bundle.sh`, in the build job only when one of the two boots runs): slot A of the boot-proof image, repacked with a `host-agent` stamped one patch release above `VERSION`, the baked tarballs under `/var/lib/moose` left out (a slot's copy is read only at a box's first boot), squashfs gzip at level 1 (GRUB reads it), signed with the throwaway key. One bundle serves both boots. +- **The harness** puts it on an ext4 image attached as a second read-only disk, gives each boot its own overlay, box-id, test-portal key and owner assertion, and does not pass `-no-reboot`, so the box reboots between slots inside one QEMU run. The in-guest script keeps its stage on the state partition. +- **`os-update`**: on slot A, owner sign-in, a data file and the `whoami` app; a wrong digest is refused and slot B stays empty; the right digest with a shut window installs into slot B and leaves `ORDER="A B"`; an open window switches and reboots. On slot B: marked good, `ORDER="B A" B_OK=1 B_TRY=0`, `host-agent` at the new version, the version read and the `os` read right, the owner's session, password hash, SSH host key, `machine-id`, the file and the app unchanged, and the admin notification there. +- **`os-revert`**: the same install, with a drop-in that keeps `host-agent` off slot B. On slot B the script only checks that and waits. The safety net reboots the box, GRUB skips the slot still on trial, and on slot A `host-agent` records the revert and marks B bad. Then the same checks as `os-update`, with the `held` state and the warning notification. +- **On a run that publishes the OS**, `os-update` runs only its refusal half (the wrong digest, then RAUC refusing the throwaway-signed bundle with the right digest), and `os-revert` is left out of the matrix (the maintainer's call). +- **Both are in every full run**, so lock bumps and releases run them. A PR that touches the code still boots only `update` (the maintainer's call). + +## The private-side change this needs (for the maintainer) + +The box side is done. **No production box moves its OS until the control plane sends the OS part.** The exact wire change to `GET /api/updates/target`: + +- Add an optional top-level field `os`: a JSON array, oldest first, of objects `{"version": "X.Y.Z", "bundle_url": "", "bundle_sha256": "<64 lowercase hex>"}`. +- **Which entries:** the newest patch of each minor, from the oldest minor still supported up to the box's OS target, with the target last. For a box that should stay on its OS, either leave `os` out or send a list whose last entry is the release it runs. +- **`bundle_url`:** the Release asset, `https://github.com/onmoose/os/releases/download/vX.Y.Z/moose-vX.Y.Z-amd64.raucb`. Anything else must also start with that prefix, or the box refuses it. +- **`bundle_sha256`:** the content of the release's `moose-vX.Y.Z-amd64.raucb.sha256` asset (the hex digest only), read once when the OS release is recorded and stored with it. Never computed per request, and never "latest". +- **Leave it out entirely** when there is no OS target. An entry with a missing or malformed digest makes the box refuse stream A and log it; stream B is not affected. +- It is a per-box fact like the control-plane target, so per-box pinning and staged rollout work the same way. The control-plane part and the OS part are independent: a box may get either, both, or neither. + +## Numbers + +From runs on this branch (the final run is in # How it was verified): + +| | Value | +|---|---| +| Download per update (a real release) | 438.5 MB, one bundle (`rauc-bundle.md`) | +| Disk the box needs while it installs | 438.5 MB on the state partition, 1.1% of a 40 GB disk, deleted after the install | +| Time to apply, measured in the guest | NUMBERS-APPLY | +| Test bundle in CI | 366.4 MB, built in 27 s (unsquashfs 4 s, mksquashfs gzip 4 s), uploaded in 5 s | +| CI cost of the two boots | NUMBERS-CI | + +## How it was verified + +All in CI, every publish input false. Never built or booted locally. + +NUMBERS-RUNS + +- **Tests.** `internal/hostagent/updatetarget/os_test.go` (the checks, every pick rule, the loop: install outside the window and switch inside it, stream B first, a bad OS part refused with stream B still applying, an answer with only an OS part, current, none, unsupported). `internal/hostagent/osupdate/osupdate_test.go` against a fake two-slot RAUC (a normal boot marked good at once, install then switch, a wrong digest never reaching RAUC and not retried the same night, a stale `TRY` stopping the switch and putting the booted slot back, a busy lock, a good trial, a failed trial rebooting and the old slot recording the revert and holding, the floor file, the JSON and grubenv parsers, `Peek`). The report's `os` part, the brain's pass-through, the notification, the store lookup and the brain's outcome check have tests too. `make check` green. + +## How it maps to the specs + +- `UPDATES.md` # 1: steps 1 to 6 of the transaction, built on hosted. The section has an "As built (#563)" part and the trial rule. # 8.4 step 4 names the wire. The rollback table says built on hosted. +- `BRAIN_HOST_PROTOCOL.md`: `os_version`/`os_slot` on the status read, the `os` object on the update-target read, the two new job kinds under the one lock, and how rule D (drain before the reboot) is met. +- `BUILD.md` # 1b: the status, `activate-installed=false`, the trial timer, the install proof. +- `DECISIONS.md` 2026-10-02 ("a leaked signer alone cannot push an update"): realized by the digest check before RAUC. + +## Known gaps & deviations + +- **No production box moves its OS yet.** The private side has to send the `os` part (above). +- **QEMU only**, as the rest of #486. #486's "Done when" still needs the same proof on a provisioned box. +- **The test bundle is not a release bundle.** It is the boot-proof slot with one binary changed, gzip-compressed, signed with the throwaway key. The release bundle (xz, release signer) is checked by #562's build checks, not installed here. A release run installs nothing. +- **The trial checks the brain's `/healthz` only.** It does not check Caddy, the UI or the apps. The spec asks for `host-agent` and the brain. +- **A `held` release after a revert waits for the next night**, and the bundle is downloaded again then. The fleet halt that would stop a broken release (`UPDATES.md` # 8.5) is deferred. +- **No report back to the cloud.** `UPDATES.md` # 8.4 step 5 still waits for box authentication. +- **The floor needs a brain from this change.** An older brain writes no floor file, and `host-agent` then has no floor to check. +- **An `http` update target URL carries the digest unprotected**, as it already carries the control-plane digests. RAUC's signature check still applies. Production uses https. +- **Same-night key.** One attempt per night uses the window's calendar night. A test run that crosses midnight UTC between its stages would see a new night. +- **The version read degrades.** When the source cannot be read after a reboot, `os.state` is `none` with the last outcome still there; it does not say "unreachable" for stream A. + +## What's next + +1. The maintainer adds the `os` part to the control plane's answer (above), then #486's proof on a provisioned box under both firmwares. +2. #564: the appliance image in the A/B layout, which then gets this applier. +3. Box authentication, for the report back and the fleet halt (`UPDATES.md` # 8.4, # 8.5). From c00d66c5156c7f4e2775c30e169b7edbbde936ab Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 17:24:16 +0100 Subject: [PATCH 05/18] os update: shorter test safety net, and the revert says which path made it (#563) --- dev/cloud/cloud-assertions.sh | 4 +++- .../mkosi.extra/usr/lib/moose/os-trial-check | 2 ++ dev/cloud/test/bootstrap.sh | 11 ++++++----- docs/dev/hosted-boot-proof.md | 4 ++-- docs/progress/host-agent-os-update.md | 2 +- docs/specs/BUILD.md | 2 +- docs/specs/UPDATES.md | 2 +- internal/hostagent/osupdate/osupdate.go | 10 ++++++++++ internal/hostagent/osupdate/osupdate_test.go | 19 +++++++++++++++++++ 9 files changed, 45 insertions(+), 11 deletions(-) diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index 8b9f81aa..45140d0d 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -186,7 +186,7 @@ esac # on it (a drop-in the first stage planted), which is the case where nothing but # the image's own safety net can take the box back. So this stage checks only # that, and waits: every other check of this script would fail on a box with no -# host-agent, which is the point of the scenario. moose-os-trial.timer (3 min in +# host-agent, which is the point of the scenario. moose-os-trial.timer (90 s in # this image) must reboot the box; the next stage runs on slot A. if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then [ "$BOOTED" = B ] || fail "os-revert: stage 2 should boot slot B, booted '$BOOTED' (did GRUB skip the new slot?)" @@ -2513,6 +2513,8 @@ UNIT else [ "$stage" = 3 ] && [ "$BOOTED" = A ] || fail "os-revert: stage $stage booted slot '$BOOTED', want stage 3 back on slot A" wait_ha "the box went back to the old slot" 120 "host-agent did not record the revert" + # And it was the image's safety net that did it, the case under test. + wait_ha "the image's safety net rebooted the new slot" 10 "the revert was not made by moose-os-trial.timer" [ ! -e /var/lib/moose/os-update/trial-B ] || fail "os-revert: the trial marker for slot B is still there" [ "$(grubvar B_OK)" = 0 ] && [ "$(grubvar A_OK)" = 1 ] || fail "os-revert: grubenv after the revert: $(grub-editenv /efi/grub/grubenv list | tr '\n' ' ')" [ "$(/usr/lib/moose/host-agent-real --version | awk '{print $2}')" = "$base_ver" ] || fail "os-revert: back on slot A, host-agent is not $base_ver" diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check index 1fefdc31..7f49e06d 100755 --- a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check +++ b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check @@ -16,5 +16,7 @@ slot="$(sed -n 's/.*rauc\.slot=\([AB]\).*/\1/p' /proc/cmdline)" [ -n "$slot" ] || exit 0 [ -e "/var/lib/moose/os-update/trial-${slot}" ] || exit 0 echo "moose-os-trial: slot ${slot} was not marked good in time; rebooting so the box goes back to the other slot" +# A note for the old slot's host-agent, so the revert says which path made it. +date -u +%Y-%m-%dT%H:%M:%SZ > "/var/lib/moose/os-update/safety-net-${slot}" && sync rauc status mark-bad booted || echo "moose-os-trial: mark-bad failed; rebooting anyway, GRUB skips a slot still on trial" systemctl reboot diff --git a/dev/cloud/test/bootstrap.sh b/dev/cloud/test/bootstrap.sh index eced8b9d..b91b6e00 100755 --- a/dev/cloud/test/bootstrap.sh +++ b/dev/cloud/test/bootstrap.sh @@ -131,15 +131,16 @@ cp "${CLOUD_DIR}/cloud-assertions.sh" "$EXTRA/usr/local/bin/cloud-assertions.sh" chmod 0755 "$EXTRA/usr/local/bin/cloud-assertions.sh" cp "${TEST_DIR}/moose-cloud-assertions.service" "$EXTRA/etc/systemd/system/" -# The OS update trial's safety net fires after 3 minutes here instead of 15 -# (#563), so the os-revert boot does not sit out a quarter of an hour. Still -# longer than a healthy trial boot takes to mark its slot good, which the -# os-update boot proves under the same setting. The image that ships keeps 15. +# The OS update trial's safety net fires after 90 s here instead of 15 min +# (#563), so the os-revert boot does not sit out a quarter of an hour. A +# healthy trial boot marks its slot good about 17 s after the switch in CI +# (run 37031756819), which the os-update boot proves under the same setting. +# The image that ships keeps 15 min. mkdir -p "$EXTRA/etc/systemd/system/moose-os-trial.timer.d" cat > "$EXTRA/etc/systemd/system/moose-os-trial.timer.d/10-cloud-test.conf" <<'EOF' [Timer] OnBootSec= -OnBootSec=180s +OnBootSec=90s EOF # --- 3b. app-install fixtures for the access-mode e2e (#308) — TEST-LANE ONLY. The diff --git a/docs/dev/hosted-boot-proof.md b/docs/dev/hosted-boot-proof.md index 09424c51..46c2bfff 100644 --- a/docs/dev/hosted-boot-proof.md +++ b/docs/dev/hosted-boot-proof.md @@ -30,7 +30,7 @@ Net: a provisioned box logs both milestones, binds `:443`, and serves every ``, run `rauc status mark-active other`, and check the grubenv reads `ORDER=" "`, `_OK=1`, `_TRY=0` before rebooting. RAUC 1.13's GRUB backend already resets `TRY` there (its `grub_set_primary`, read in the source), so the #570 gap is now a check, and a failing check puts the booted slot first again and does not reboot. One switch per release per night, kept in the record across the reboot. - **The trial, at the next start of `host-agent`** (`Boot`). With the booted slot's marker: wait for the brain's `/healthz` up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min); healthy means `mark-good`, the marker goes, the outcome is `good`; not healthy means `mark-bad booted` and a reboot. Without a marker: `mark-good` at once. With the other slot's marker: that slot failed, so the outcome is `reverted`, the slot is marked bad, the marker goes. **Only the first boot after a switch is on trial** (the maintainer's call). -- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker, its `host-agent` never got there, so the script marks the slot bad and reboots. The boot-proof image sets it to 3 minutes. +- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker, its `host-agent` never got there, so the script marks the slot bad, leaves a note `safety-net-` and reboots. The old slot's `host-agent` logs from the note that the safety net made the revert, which the `os-revert` boot checks. The boot-proof image sets it to 90 s. - **Jobs.** `hostagent.Agent.StartJob` runs any job under the one lock `system-update` takes. The OS jobs carry no result. - **Record.** `state.json`: what was installed where, the last install attempt, the last switch, the last outcome with an id. Fields only grow, because either release may read it. diff --git a/docs/specs/BUILD.md b/docs/specs/BUILD.md index 7bcf7010..1f24f641 100644 --- a/docs/specs/BUILD.md +++ b/docs/specs/BUILD.md @@ -234,7 +234,7 @@ One **RAUC bundle** per OS release, in the `verity` format: the slot image in a - **Signing and its checks.** The build job always signs with the throwaway signer, and then checks the bundle against the `system.conf` and keyring read back out of the slot with `unsquashfs`, not against the copies in the repo: the slot carries the keyring the build staged; the bundle verifies against the throwaway root; with the throwaway keyring the image accepts it, and with the release keyring it must **refuse** it; and an unrelated CA (a wrong key) is refused. It also rehearses the `sign` job's re-signing with a throwaway "release" CA, so that script runs on every build. A **`sign` job** of its own, after every boot passed, alone re-signs the bundle with the release signer (`dev/release/sign-bundle.sh`, `rauc resign`; the payload stays the same bytes), but only a bundle whose sha256 matches the one the build job reported, and checks it against the image's own config and keyring. The publish job then attaches it, after checking the digest `sign` reported. So the signature vouches for what the build job produced; the box's check of the digest its update target names (#563) is the control for a bad build. It refuses when the signer secrets are empty, when the image keyring is not the committed release root, or when the image would accept the throwaway-signed bundle. - **Key custody** (`DECISIONS.md` 2026-10-02): an offline root CA with the maintainer, and a signer it issued as the secrets `RAUC_SIGNING_CERT` and `RAUC_SIGNING_KEY` of the GitHub Environment `os-release` (only `main` and `v*` tags; only the `sign` job enters it). No CRLs. Making the root, issuing and rotating a signer and replacing the root are in `docs/dev/rauc-signing.md`. **Until `release-ca.pem` is committed, `release.yml` tags nothing for a merge that bumps `VERSION`** (on either line, even when the merge bumps `CONTROL_PLANE_VERSION` too), and a dispatch that publishes the OS fails at its first step. Until the secrets exist, the `sign` job refuses and nothing is published. The exact rule is in `docs/dev/contributing.md` # Release model. - **Installed by `host-agent` since #563.** The `os-update` boot proves `rauc install` into slot B (RAUC's `raw` handler takes `rootfs.img`), the switch and the trial boot, under both firmwares; `os-revert` proves the revert (`docs/dev/hosted-boot-proof.md`). RAUC's install leaves the boot order alone (`activate-installed=false` in `system.conf`): `host-agent` switches with `rauc status mark-active other` only inside the window, and checks the grubenv reads `OK=1 TRY=0` for the new slot before it reboots (the stale try flag of #570). -- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`), its `host-agent` never marked it good, so it marks the slot bad and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 3 minutes (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15. +- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`), its `host-agent` never marked it good, so it marks the slot bad, leaves a note (`/var/lib/moose/os-update/safety-net-`, which the old slot's `host-agent` logs and removes) and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 90 s (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15 min. ### The OS package lock diff --git a/docs/specs/UPDATES.md b/docs/specs/UPDATES.md index 7b548e28..b48605c8 100644 --- a/docs/specs/UPDATES.md +++ b/docs/specs/UPDATES.md @@ -65,7 +65,7 @@ The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host- - **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start, and a box with no such file has no floor to check. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. - **Install, ahead of the window** (an `os-install` job, under the same lock as `system-update`). The bundle is downloaded to `/var/lib/moose/os-update/` on the state partition and its sha256 checked against the answer before RAUC sees it. Then `rauc install`, which checks the signature and writes the other slot. `system.conf` has `activate-installed=false`, so the boot order does not move. The file is deleted afterwards. A failed attempt is not repeated the same night for the same release and digest. - **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. One switch per release per night. -- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad and stays. The same release is tried again the next night. +- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. - **Report.** `GET /api/v1/system/version` carries `os_version` and `os_slot`; `GET /api/v1/system/update-target` carries an `os` object beside stream B's fields (`BRAIN_HOST_PROTOCOL.md`). The brain raises one admin notification per outcome (`NOTIFICATIONS.md` # Updates). Reporting back to the cloud (# 8.4 step 5) still waits for real box authentication. - **What it costs a box** (CI, `../progress/host-agent-os-update.md`): one bundle download per update (438.5 MB for a real release), the same space on the state partition while it installs (1.1% of a 40 GB disk), and one reboot. diff --git a/internal/hostagent/osupdate/osupdate.go b/internal/hostagent/osupdate/osupdate.go index 824d7722..c2860e99 100644 --- a/internal/hostagent/osupdate/osupdate.go +++ b/internal/hostagent/osupdate/osupdate.go @@ -336,6 +336,16 @@ func (a *Applier) Boot(ctx context.Context) { } slog.Warn("os update: the new slot did not come up healthy; the box went back to the old slot", "os", out.Version, "slot", booted) + // Which path made the revert: the new slot's host-agent (its trial timed + // out), or the image's timer, which leaves this note because the new + // slot's host-agent never got that far. + note := filepath.Join(a.dir(), "safety-net-"+oth) + if _, err := os.Stat(note); err == nil { + slog.Warn("os update: the image's safety net rebooted the new slot; its host-agent never marked it", "os", out.Version, "slot", oth) + if err := os.Remove(note); err != nil { + slog.Error("os update: could not remove the safety-net note", "err", err, "slot", oth) + } + } } func outcomeID(s *switched) string { diff --git a/internal/hostagent/osupdate/osupdate_test.go b/internal/hostagent/osupdate/osupdate_test.go index f4e9ed95..edf5d95d 100644 --- a/internal/hostagent/osupdate/osupdate_test.go +++ b/internal/hostagent/osupdate/osupdate_test.go @@ -349,3 +349,22 @@ func TestPeek(t *testing.T) { t.Fatalf("after the install: %s %v", st, ok) } } + +func TestRevertNotesTheSafetyNet(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + // The image's timer on slot B left its note and rebooted. + note := h.a.dir() + "/safety-net-B" + if err := os.WriteFile(note, []byte("2026-10-03T03:20:00Z\n"), 0o600); err != nil { + t.Fatal(err) + } + a := h.reboot(t, "A", "0.15.0") + a.Boot(context.Background()) + if a.Last() == nil || a.Last().Outcome != protocol.OSOutcomeReverted { + t.Fatalf("outcome %+v", a.Last()) + } + if _, err := os.Stat(note); !errors.Is(err, os.ErrNotExist) { + t.Fatal("the safety-net note was not removed") + } +} From db47ccc15648cfbacf8a05a009824543bd252b75 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 17:39:45 +0100 Subject: [PATCH 06/18] os update: review fixes (#563) Nothing installs or switches while the booted slot is on trial, and the box switches its OS at most once a night. A bad os part, a list that leaves a minor out, and a downgrade without a readable floor are refused for stream A only. A failed reboot undoes the switch; the image timer leaves a slot already marked good alone; rauc status is retried at start; an unreadable record is never written over; the download is capped. A local run builds the test bundle itself. --- cmd/brain/osupdate.go | 9 +- cmd/brain/osupdate_test.go | 3 +- cmd/host-agent-real/osupdate.go | 2 +- .../mkosi.extra/usr/lib/moose/os-trial-check | 8 ++ dev/cloud/run-cloud-tests.sh | 6 + docs/progress/host-agent-os-update.md | 15 ++- docs/specs/BUILD.md | 2 +- docs/specs/UPDATES.md | 6 +- internal/hostagent/osupdate/osupdate.go | 73 +++++++++++-- internal/hostagent/osupdate/osupdate_test.go | 103 ++++++++++++++++++ internal/hostagent/updatetarget/http.go | 17 ++- internal/hostagent/updatetarget/loop.go | 5 + internal/hostagent/updatetarget/os.go | 14 ++- internal/hostagent/updatetarget/os_test.go | 38 ++++++- internal/hostagent/updatetarget/target.go | 3 + internal/notify/notify.go | 10 +- 16 files changed, 276 insertions(+), 38 deletions(-) diff --git a/cmd/brain/osupdate.go b/cmd/brain/osupdate.go index eaf60388..0efa7a6f 100644 --- a/cmd/brain/osupdate.go +++ b/cmd/brain/osupdate.go @@ -26,7 +26,7 @@ const floorFileName = "minimum-host-agent" func writeFloorFile(stateDir string) { path := filepath.Join(stateDir, floorFileName) if err := os.WriteFile(path, []byte(minimumAgentVersion+"\n"), 0o644); err != nil { - slog.Warn("could not write the control-plane floor for host-agent; it will not check OS targets against it", "err", err, "dir", stateDir) + slog.Warn("could not write the control-plane floor for host-agent; it will not check OS targets against it", "err", err, "state_dir", stateDir) } } @@ -42,7 +42,7 @@ type notificationLookup interface { // osOutcomeNotifier is the slice of the notifier the check needs. type osOutcomeNotifier interface { - OSUpdateOutcome(outcomeID, outcome, version, from string) + OSUpdateOutcome(outcomeID, outcome, version, from string) bool } // osOutcomeMaxAge bounds which outcomes still get a notification. host-agent @@ -71,8 +71,9 @@ func checkOSOutcome(ctx context.Context, host osOutcomeReader, seen notification if done { return } - n.OSUpdateOutcome(last.ID, last.Outcome, last.Version, last.From) - slog.Info("os update: notified admins of the outcome", "os", last.Version) + if n.OSUpdateOutcome(last.ID, last.Outcome, last.Version, last.From) { + slog.Info("os update: notified admins of the outcome", "os", last.Version) + } } // osOutcomeLoop runs checkOSOutcome on the health-poll cadence. diff --git a/cmd/brain/osupdate_test.go b/cmd/brain/osupdate_test.go index b5d73208..4d29db9e 100644 --- a/cmd/brain/osupdate_test.go +++ b/cmd/brain/osupdate_test.go @@ -23,8 +23,9 @@ func (f fakeSeen) HasNotification(k string) (bool, error) { return f[k], nil } type fakeOutcomeNotifier struct{ raised []string } -func (f *fakeOutcomeNotifier) OSUpdateOutcome(id, outcome, version, from string) { +func (f *fakeOutcomeNotifier) OSUpdateOutcome(id, outcome, version, from string) bool { f.raised = append(f.raised, id+" "+outcome+" "+version+" "+from) + return true } func TestCheckOSOutcome(t *testing.T) { diff --git a/cmd/host-agent-real/osupdate.go b/cmd/host-agent-real/osupdate.go index 2d19b94b..b3740e5f 100644 --- a/cmd/host-agent-real/osupdate.go +++ b/cmd/host-agent-real/osupdate.go @@ -46,7 +46,7 @@ func osTrialTimeout() time.Duration { } d, err := time.ParseDuration(v) if err != nil || d <= 0 { - slog.Warn("os update: the trial timeout is not readable; using the default", "err", err, "window", v) + slog.Warn("os update: "+envOSTrialTimeout+" is not a positive duration; using the default", "err", err) return osupdate.DefaultTrialTimeout } return d diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check index 7f49e06d..98e36a8e 100755 --- a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check +++ b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check @@ -15,6 +15,14 @@ set -u slot="$(sed -n 's/.*rauc\.slot=\([AB]\).*/\1/p' /proc/cmdline)" [ -n "$slot" ] || exit 0 [ -e "/var/lib/moose/os-update/trial-${slot}" ] || exit 0 +# A slot already marked good (OK=1 and TRY=0; GRUB set TRY=1 when it booted +# it) keeps its marker only because host-agent could not remove it. That is +# not a failed trial: leave the box alone. +env="$(grub-editenv /efi/grub/grubenv list 2>/dev/null)" +if printf '%s\n' "$env" | grep -qx "${slot}_OK=1" && printf '%s\n' "$env" | grep -qx "${slot}_TRY=0"; then + echo "moose-os-trial: slot ${slot} is marked good but still has its trial marker; not rebooting" + exit 0 +fi echo "moose-os-trial: slot ${slot} was not marked good in time; rebooting so the box goes back to the other slot" # A note for the old slot's host-agent, so the revert says which path made it. date -u +%Y-%m-%dT%H:%M:%SZ > "/var/lib/moose/os-update/safety-net-${slot}" && sync diff --git a/dev/cloud/run-cloud-tests.sh b/dev/cloud/run-cloud-tests.sh index f3b214b2..570e6dd7 100755 --- a/dev/cloud/run-cloud-tests.sh +++ b/dev/cloud/run-cloud-tests.sh @@ -710,6 +710,12 @@ fi # call for #563). os_boot() { # NAME BOX_ID local name="$1" box="$2" dir="${MOOSE_CLOUD_OS_BUNDLE_DIR:-}" mint key token disk sum ver mode + # A local run that built the image here makes the bundle from it too, once. + if [ -z "$dir" ] && [ -f "$IMAGE_OUT" ]; then + dir="${WORK}/os-test" + [ "$dir/os-test.raucb" -nt "$IMAGE_OUT" ] || GO="$GO" "${REPO_ROOT}/dev/cloud/test/build-os-test-bundle.sh" "$IMAGE_OUT" "$dir" + MOOSE_CLOUD_OS_BUNDLE_DIR="$dir" + fi [ -n "$GO" ] && [ -x "$GO" ] || { echo "$name boot needs go to mint the owner assertion; none found (\$GO='${GO:-}')" >&2; exit 1; } [ -n "$dir" ] && [ -f "$dir/os-test.raucb" ] && [ -f "$dir/os-test.sha256" ] && [ -f "$dir/os-test.version" ] || { echo "$name boot needs the test bundle: set MOOSE_CLOUD_OS_BUNDLE_DIR to the output of dev/cloud/test/build-os-test-bundle.sh" >&2 diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 3a87ffaa..79889e9a 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -12,16 +12,16 @@ Closes #563, the fifth slice of #486. It follows [hosted-ab-layout.md](hosted-ab - **Wire.** The update-target answer gains an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}`. The last entry is the target. Left out, the answer has no opinion about the OS. - **Checks** (`os.go`, `ValidateOS`): a plain `X.Y.Z` version, ascending order, a 64-hex lowercase sha256, an absolute http or https URL, and a URL that starts with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`, the same kind of check as the expected image repositories). -- **Which release** (`PickOS`), as `UPDATES.md` # 1 says: on a later minor, the first entry above the box's own minor (a box never skips a minor); on its own minor or one back, the target; further back, refused. A release below the running control plane's floor is refused too. The brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start for this; a box without the file has no floor to check. -- **Each stream is judged on its own.** A bad OS part refuses stream A only. A refused control-plane part, or a box that cannot read its running pair, still lets stream A go on. The test answer of the new boots carries only an OS part, which proves it. +- **Which release** (`PickOS`), as `UPDATES.md` # 1 says: on a later minor, the first entry above the box's own minor (a box never skips a minor); on its own minor or one back, the target; further back, refused. A list that leaves a minor out is refused (the step must be the minor right after the box's own). A release below the running control plane's floor is refused too. The brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start for this; a box that cannot read the file refuses every move to an older release. +- **Each stream is judged on its own.** A bad OS part, or one of the wrong JSON shape (decoded apart from the rest), refuses stream A only. A refused control-plane part, or a box that cannot read its running pair, still lets stream A go on. The test answer of the new boots carries only an OS part, which proves it. - **Order in the window.** The loop runs stream B first. When stream B starts an update in this tick, or cannot because a job runs, stream A waits for the next tick. ### The transaction (`internal/hostagent/osupdate`, new) - **Install, ahead of the window** (`os-install` job). Download to `/var/lib/moose/os-update/bundle.raucb.part` on the state partition, hashing as it writes. Only a file whose sha256 is the one the answer names is renamed and handed to `rauc install`, so RAUC never sees a bundle the answer did not name. RAUC then checks the signature against the image's keyring. `system.conf` now has `activate-installed=false`, so the install leaves the boot order alone. The file is deleted afterwards. A failed attempt is not repeated for the same release and digest the same night. -- **Switch, inside the window** (`os-switch` job). Check that RAUC reports the other slot holds the release, write the record, write the trial marker `/var/lib/moose/os-update/trial-`, run `rauc status mark-active other`, and check the grubenv reads `ORDER=" "`, `_OK=1`, `_TRY=0` before rebooting. RAUC 1.13's GRUB backend already resets `TRY` there (its `grub_set_primary`, read in the source), so the #570 gap is now a check, and a failing check puts the booted slot first again and does not reboot. One switch per release per night, kept in the record across the reboot. +- **Switch, inside the window** (`os-switch` job). Check that RAUC reports the other slot holds the release, write the record, write the trial marker `/var/lib/moose/os-update/trial-`, run `rauc status mark-active other`, and check the grubenv reads `ORDER=" "`, `_OK=1`, `_TRY=0` before rebooting. RAUC 1.13's GRUB backend already resets `TRY` there (its `grub_set_primary`, read in the source), so the #570 gap is now a check, and a failing check puts the booted slot first again and does not reboot. A reboot that is not accepted undoes the switch. **One OS switch per night**, kept in the record across the reboot, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**: the other slot is then the way back. - **The trial, at the next start of `host-agent`** (`Boot`). With the booted slot's marker: wait for the brain's `/healthz` up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min); healthy means `mark-good`, the marker goes, the outcome is `good`; not healthy means `mark-bad booted` and a reboot. Without a marker: `mark-good` at once. With the other slot's marker: that slot failed, so the outcome is `reverted`, the slot is marked bad, the marker goes. **Only the first boot after a switch is on trial** (the maintainer's call). -- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker, its `host-agent` never got there, so the script marks the slot bad, leaves a note `safety-net-` and reboots. The old slot's `host-agent` logs from the note that the safety net made the revert, which the `os-revert` boot checks. The boot-proof image sets it to 90 s. +- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker and the grubenv does not already mark it good (a marker host-agent failed to remove), its `host-agent` never got there, so the script marks the slot bad, leaves a note `safety-net-` and reboots. The old slot's `host-agent` logs from the note that the safety net made the revert, which the `os-revert` boot checks. The boot-proof image sets it to 90 s. - **Jobs.** `hostagent.Agent.StartJob` runs any job under the one lock `system-update` takes. The OS jobs carry no result. - **Record.** `state.json`: what was installed where, the last install attempt, the last switch, the last outcome with an id. Fields only grow, because either release may read it. @@ -80,6 +80,11 @@ NUMBERS-RUNS - `BUILD.md` # 1b: the status, `activate-installed=false`, the trial timer, the install proof. - `DECISIONS.md` 2026-10-02 ("a leaked signer alone cannot push an update"): realized by the digest check before RAUC. +## Review + +- **The fresh review agent** found one Block and two Shoulds, all fixed: the loop could install into the old slot, the way back, while the new slot was still on trial, and a multi-step update could switch twice in one night (now: nothing moves during a trial, one switch per night); a failed `rauc status` at start left a trial to the image timer (now retried for about 30 s); a failed read of the record could be written over (now never). Its nits are fixed too: two log field names, a "notified" line for an outcome that raised nothing, and a 2 GiB cap on the download. +- **Greptile** found seven, all fixed: an `os` part of the wrong JSON shape failed the whole answer (now decoded apart, stream A only); a list that left a minor out let the box skip it (now refused); the trial-overwrite case above; a failed reboot after `mark-active` left the new slot first (now undone); a stale marker on a slot already marked good would make the timer revert it (now the script checks the grubenv first); a missing floor allowed a downgrade (now every downgrade needs the floor); and a plain local `make test-cloud-qemu` had no test bundle (now the harness builds it from the local image). + ## Known gaps & deviations - **No production box moves its OS yet.** The private side has to send the `os` part (above). @@ -88,7 +93,7 @@ NUMBERS-RUNS - **The trial checks the brain's `/healthz` only.** It does not check Caddy, the UI or the apps. The spec asks for `host-agent` and the brain. - **A `held` release after a revert waits for the next night**, and the bundle is downloaded again then. The fleet halt that would stop a broken release (`UPDATES.md` # 8.5) is deferred. - **No report back to the cloud.** `UPDATES.md` # 8.4 step 5 still waits for box authentication. -- **The floor needs a brain from this change.** An older brain writes no floor file, and `host-agent` then has no floor to check. +- **The floor needs a brain from this change.** An older brain writes no floor file, and `host-agent` then refuses every OS downgrade on that box. Upgrades are not affected. - **An `http` update target URL carries the digest unprotected**, as it already carries the control-plane digests. RAUC's signature check still applies. Production uses https. - **Same-night key.** One attempt per night uses the window's calendar night. A test run that crosses midnight UTC between its stages would see a new night. - **The version read degrades.** When the source cannot be read after a reboot, `os.state` is `none` with the last outcome still there; it does not say "unreachable" for stream A. diff --git a/docs/specs/BUILD.md b/docs/specs/BUILD.md index 1f24f641..1888e286 100644 --- a/docs/specs/BUILD.md +++ b/docs/specs/BUILD.md @@ -234,7 +234,7 @@ One **RAUC bundle** per OS release, in the `verity` format: the slot image in a - **Signing and its checks.** The build job always signs with the throwaway signer, and then checks the bundle against the `system.conf` and keyring read back out of the slot with `unsquashfs`, not against the copies in the repo: the slot carries the keyring the build staged; the bundle verifies against the throwaway root; with the throwaway keyring the image accepts it, and with the release keyring it must **refuse** it; and an unrelated CA (a wrong key) is refused. It also rehearses the `sign` job's re-signing with a throwaway "release" CA, so that script runs on every build. A **`sign` job** of its own, after every boot passed, alone re-signs the bundle with the release signer (`dev/release/sign-bundle.sh`, `rauc resign`; the payload stays the same bytes), but only a bundle whose sha256 matches the one the build job reported, and checks it against the image's own config and keyring. The publish job then attaches it, after checking the digest `sign` reported. So the signature vouches for what the build job produced; the box's check of the digest its update target names (#563) is the control for a bad build. It refuses when the signer secrets are empty, when the image keyring is not the committed release root, or when the image would accept the throwaway-signed bundle. - **Key custody** (`DECISIONS.md` 2026-10-02): an offline root CA with the maintainer, and a signer it issued as the secrets `RAUC_SIGNING_CERT` and `RAUC_SIGNING_KEY` of the GitHub Environment `os-release` (only `main` and `v*` tags; only the `sign` job enters it). No CRLs. Making the root, issuing and rotating a signer and replacing the root are in `docs/dev/rauc-signing.md`. **Until `release-ca.pem` is committed, `release.yml` tags nothing for a merge that bumps `VERSION`** (on either line, even when the merge bumps `CONTROL_PLANE_VERSION` too), and a dispatch that publishes the OS fails at its first step. Until the secrets exist, the `sign` job refuses and nothing is published. The exact rule is in `docs/dev/contributing.md` # Release model. - **Installed by `host-agent` since #563.** The `os-update` boot proves `rauc install` into slot B (RAUC's `raw` handler takes `rootfs.img`), the switch and the trial boot, under both firmwares; `os-revert` proves the revert (`docs/dev/hosted-boot-proof.md`). RAUC's install leaves the boot order alone (`activate-installed=false` in `system.conf`): `host-agent` switches with `rauc status mark-active other` only inside the window, and checks the grubenv reads `OK=1 TRY=0` for the new slot before it reboots (the stale try flag of #570). -- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`), its `host-agent` never marked it good, so it marks the slot bad, leaves a note (`/var/lib/moose/os-update/safety-net-`, which the old slot's `host-agent` logs and removes) and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 90 s (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15 min. +- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`) and the grubenv does not mark it good (`OK=1 TRY=0`), its `host-agent` never marked it good, so it marks the slot bad, leaves a note (`/var/lib/moose/os-update/safety-net-`, which the old slot's `host-agent` logs and removes) and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 90 s (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15 min. ### The OS package lock diff --git a/docs/specs/UPDATES.md b/docs/specs/UPDATES.md index b48605c8..29012b8e 100644 --- a/docs/specs/UPDATES.md +++ b/docs/specs/UPDATES.md @@ -62,10 +62,10 @@ The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host- `internal/hostagent/osupdate` is the transaction, `internal/hostagent/updatetarget` decides which release and when. Both run in `host-agent`; the brain only reports. -- **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start, and a box with no such file has no floor to check. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. +- **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a list whose next step is not the minor right after the box's own (a list that leaves a minor out), and a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start. **A box that cannot read that file refuses every move to an older release.** An `os` part of the wrong JSON shape is refused like a bad entry. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. - **Install, ahead of the window** (an `os-install` job, under the same lock as `system-update`). The bundle is downloaded to `/var/lib/moose/os-update/` on the state partition and its sha256 checked against the answer before RAUC sees it. Then `rauc install`, which checks the signature and writes the other slot. `system.conf` has `activate-installed=false`, so the boot order does not move. The file is deleted afterwards. A failed attempt is not repeated the same night for the same release and digest. -- **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. One switch per release per night. -- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. +- **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. A reboot that is not accepted undoes the switch. **One OS switch per night**, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**, because the other slot is then the way back. +- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker and the grubenv does not already mark it good, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. - **Report.** `GET /api/v1/system/version` carries `os_version` and `os_slot`; `GET /api/v1/system/update-target` carries an `os` object beside stream B's fields (`BRAIN_HOST_PROTOCOL.md`). The brain raises one admin notification per outcome (`NOTIFICATIONS.md` # Updates). Reporting back to the cloud (# 8.4 step 5) still waits for real box authentication. - **What it costs a box** (CI, `../progress/host-agent-os-update.md`): one bundle download per update (438.5 MB for a real release), the same space on the state partition while it installs (1.1% of a 40 GB disk), and one reboot. diff --git a/internal/hostagent/osupdate/osupdate.go b/internal/hostagent/osupdate/osupdate.go index c2860e99..c6517599 100644 --- a/internal/hostagent/osupdate/osupdate.go +++ b/internal/hostagent/osupdate/osupdate.go @@ -63,6 +63,18 @@ const DefaultDir = "/var/lib/moose/os-update" // a live host-agent always decides first. const DefaultTrialTimeout = 10 * time.Minute +// statusAttempts and statusRetry bound how long Boot waits for RAUC to +// answer: about 30 s. Vars, so a test does not wait. +var ( + statusAttempts = 10 + statusRetry = 3 * time.Second +) + +// maxBundleBytes caps a download. A bundle holds one slot image, and a slot is +// 1 GiB, so anything past 2 GiB is not a bundle and must not fill the state +// partition. +const maxBundleBytes = 2 << 30 + // Job bounds. A 438 MB download and install takes well under a minute on a // hosted box; the bound is there for a stalled transfer. const ( @@ -281,9 +293,22 @@ func (a *Applier) Last() *protocol.OSOutcome { // Boot reads the booted slot and decides what this boot is (see the package // comment). It returns at once; a trial runs in the background. func (a *Applier) Boot(ctx context.Context) { - st, err := a.RAUC.Status(ctx) + // Retried: rauc-service is D-Bus activated and can be slow on a busy + // boot. Giving up on a trial boot would leave the decision to the image's + // timer, which would revert a healthy slot. + var st Status + var err error + for i := 0; i < statusAttempts; i++ { + if st, err = a.RAUC.Status(ctx); err == nil { + break + } + select { + case <-ctx.Done(): + case <-time.After(statusRetry): + } + } if err != nil { - slog.Error("os update: cannot read the slots; this box will not update its OS until it can", "err", err) + slog.Error("os update: cannot read the slots; this box will not update its OS until host-agent restarts", "err", err) return } a.mu.Lock() @@ -368,7 +393,9 @@ func (a *Applier) trial(slot string) { if err := a.RAUC.Mark(bg, "bad", "booted"); err != nil { slog.Error("os update: could not mark the slot bad; rebooting anyway, GRUB skips a slot still on trial", "err", err) } - a.reboot() + if err := a.reboot(); err != nil { + slog.Error("os update: the trial cannot reboot; the image's safety net reboots the box", "err", err, "slot", slot) + } return } if err := a.RAUC.Mark(bg, "good", "booted"); err != nil { @@ -376,7 +403,9 @@ func (a *Applier) trial(slot string) { // inside the window, rather than leave a slot that will revert at // some random later reboot. slog.Error("os update: could not mark the new slot good; rebooting to the old slot", "err", err, "slot", slot) - a.reboot() + if err := a.reboot(); err != nil { + slog.Error("os update: the trial cannot reboot; the image's safety net reboots the box", "err", err, "slot", slot) + } return } // The marker goes first: once the slot is good, nothing may reboot it @@ -403,7 +432,7 @@ func (a *Applier) trial(slot string) { slog.Info("os update: the new slot is healthy and marked good", "os", a.Version, "slot", slot) } -func (a *Applier) reboot() { +func (a *Applier) reboot() error { var err error if a.Reboot != nil { err = a.Reboot() @@ -413,6 +442,7 @@ func (a *Applier) reboot() { if err != nil { slog.Error("os update: reboot failed", "err", err) } + return err } // Apply is the loop's call (updatetarget.OSApplier). @@ -429,6 +459,11 @@ func (a *Applier) Apply(rel updatetarget.OSRelease, open bool, night time.Time) if slot == "" { return updatetarget.OSDecision{State: protocol.OSUpdateUnsupported, Detail: "the booted slot is not known"} } + // While this boot is on trial the other slot is the way back. Nothing + // may write it, and nothing may switch, until the trial is decided. + if _, err := os.Stat(TrialMarker(a.dir(), slot)); err == nil { + return updatetarget.OSDecision{State: protocol.OSUpdateWaiting, Detail: "this boot is on trial; nothing moves until the new slot is marked good"} + } r, err := a.load() if err != nil { return updatetarget.OSDecision{State: protocol.OSUpdateFailed, Detail: err.Error()} @@ -441,6 +476,12 @@ func (a *Applier) Apply(rel updatetarget.OSRelease, open bool, night time.Time) if !open { return updatetarget.OSDecision{State: protocol.OSUpdateInstalled} } + // One OS switch per night: a box several minors behind takes the + // next step in a later window, not right after this one (UPDATES.md # 1). + if s := r.Switch; s != nil && s.Night.Equal(night) { + return updatetarget.OSDecision{State: protocol.OSUpdateHeld, + Detail: "the box already switched its OS tonight; the next window switches again"} + } return a.start(protocol.JobKindOSSwitch, switchMaxDuration, protocol.OSUpdateRebooting, func(ctx context.Context) error { return a.doSwitch(ctx, rel, night) }) @@ -484,7 +525,13 @@ func (a *Applier) doInstall(ctx context.Context, rel updatetarget.OSRelease, nig defer func() { r, lerr := a.load() if lerr != nil { - slog.Error("os update: cannot read the record", "err", lerr) + // Never write over a record we could not read: it holds the + // switch guard and the last outcome. + slog.Error("os update: cannot read the record; not recording this attempt", "err", lerr) + if err == nil { + err = lerr + } + return } r.Attempt = &attempt{Version: rel.Version, Digest: rel.BundleSHA256, Night: night} if err != nil { @@ -556,7 +603,10 @@ func (a *Applier) download(ctx context.Context, rel updatetarget.OSRelease, path return err } h := sha256.New() - _, err = io.Copy(io.MultiWriter(f, h), resp.Body) + n, err := io.Copy(io.MultiWriter(f, h), io.LimitReader(resp.Body, maxBundleBytes+1)) + if err == nil && n > maxBundleBytes { + err = fmt.Errorf("the bundle is larger than %d bytes", int64(maxBundleBytes)) + } if cerr := f.Close(); err == nil { err = cerr } @@ -615,7 +665,11 @@ func (a *Applier) doSwitch(ctx context.Context, rel updatetarget.OSRelease, nigh env["ORDER"], target, env[target+"_OK"], target, env[target+"_TRY"])) } slog.Info("os update: switching slots and rebooting", "os", rel.Version, "slot", target) - a.reboot() + // A reboot that was not accepted must not leave the new slot first: an + // unrelated reboot later would then switch the OS outside the window. + if err := a.reboot(); err != nil { + return undo(fmt.Errorf("reboot: %w", err)) + } return nil } @@ -634,6 +688,9 @@ func (a *Applier) Peek(rel updatetarget.OSRelease) (state, detail string, ok boo case protocol.JobKindOSSwitch: return protocol.OSUpdateRebooting, "", true } + if _, err := os.Stat(TrialMarker(a.dir(), slot)); err == nil { + return protocol.OSUpdateWaiting, "this boot is on trial; nothing moves until the new slot is marked good", true + } r, err := a.load() if err != nil { return "", "", false diff --git a/internal/hostagent/osupdate/osupdate_test.go b/internal/hostagent/osupdate/osupdate_test.go index edf5d95d..93053f2f 100644 --- a/internal/hostagent/osupdate/osupdate_test.go +++ b/internal/hostagent/osupdate/osupdate_test.go @@ -368,3 +368,106 @@ func TestRevertNotesTheSafetyNet(t *testing.T) { t.Fatal("the safety-net note was not removed") } } + +// While the booted slot is on trial, the other slot is the way back: nothing +// installs into it and nothing switches (review of #563). +func TestNothingMovesDuringATrial(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + h.healthy = errors.New("not yet") + b := h.reboot(t, "B", "0.15.1") + b.TrialTimeout = time.Hour // the trial stays open for the test + b.Boot(context.Background()) + next := updatetarget.OSRelease{Version: "0.16.0", BundleURL: h.rel.BundleURL, BundleSHA256: h.rel.BundleSHA256} + jobs := len(h.jobs.errs) + if d := b.Apply(next, true, h.night); d.State != protocol.OSUpdateWaiting { + t.Fatalf("during a trial: %+v", d) + } + if len(h.jobs.errs) != jobs { + t.Fatal("a job started during the trial") + } +} + +// After a good switch tonight, the next step waits for the next window. +func TestOneSwitchPerNight(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + b := h.reboot(t, "B", "0.15.1") + b.Boot(context.Background()) + waitFor(t, func() bool { return b.Last() != nil }) + // The next step is installed into slot A the same night. + body := "0.16.0\n" + sum := sha256.Sum256([]byte(body)) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { _, _ = w.Write([]byte(body)) })) + defer srv.Close() + next := updatetarget.OSRelease{Version: "0.16.0", BundleURL: srv.URL + "/n.raucb", BundleSHA256: hex.EncodeToString(sum[:])} + if d := b.Apply(next, true, h.night); d.State != protocol.OSUpdateInstalling { + t.Fatalf("install of the next step: %+v", d) + } + if d := b.Apply(next, true, h.night); d.State != protocol.OSUpdateHeld { + t.Fatalf("a second switch the same night must wait: %+v", d) + } + if d := b.Apply(next, true, h.night.Add(24*time.Hour)); d.State != protocol.OSUpdateRebooting { + t.Fatalf("the next night switches: %+v", d) + } +} + +// flakyRAUC fails Status a few times, the way a slow rauc-service does. +type flakyRAUC struct { + *fakeRAUC + fails int +} + +func (f *flakyRAUC) Status(ctx context.Context) (Status, error) { + if f.fails > 0 { + f.fails-- + return Status{}, errors.New("rauc-service not ready") + } + return f.fakeRAUC.Status(ctx) +} + +func TestBootRetriesStatus(t *testing.T) { + old := statusRetry + statusRetry = time.Millisecond + defer func() { statusRetry = old }() + r := &flakyRAUC{fakeRAUC: newFakeRAUC("A"), fails: 3} + a := &Applier{RAUC: r, Jobs: &syncJobs{}, Version: "0.15.0", Dir: t.TempDir()} + a.Boot(context.Background()) + if _, slot := a.Running(); slot != "A" { + t.Fatalf("Boot gave up on a slow RAUC: slot %q", slot) + } +} + +// A record that cannot be read is never written over. +func TestInstallKeepsAnUnreadableRecord(t *testing.T) { + h := newHarness(t, "A") + path := h.a.dir() + "/state.json" + if err := os.WriteFile(path, []byte("{not json"), 0o600); err != nil { + t.Fatal(err) + } + h.a.Apply(h.rel, false, h.night) + b, err := os.ReadFile(path) + if err != nil || string(b) != "{not json" { + t.Fatalf("the record was overwritten: %q %v", b, err) + } +} + +// A reboot that was not accepted puts the booted slot first again, so a later +// unrelated reboot does not switch the OS outside the window. +func TestFailedRebootUndoesTheSwitch(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Reboot = func() error { return errors.New("systemctl: no") } + h.a.Apply(h.rel, true, h.night) + if h.jobs.errs[1] == nil { + t.Fatal("the switch reported success with no reboot") + } + if h.rauc.env["ORDER"] != "A B" { + t.Fatalf("ORDER=%q after a failed reboot, want the booted slot first", h.rauc.env["ORDER"]) + } + if _, err := os.Stat(TrialMarker(h.a.dir(), "B")); !errors.Is(err, os.ErrNotExist) { + t.Fatal("the trial marker was left behind") + } +} diff --git a/internal/hostagent/updatetarget/http.go b/internal/hostagent/updatetarget/http.go index 765bf51a..1d88173f 100644 --- a/internal/hostagent/updatetarget/http.go +++ b/internal/hostagent/updatetarget/http.go @@ -150,7 +150,10 @@ type wireTarget struct { // OS is optional too: the OS releases on the way to this box's OS target, // oldest first (os.go, UPDATES.md # 1). Left out, the answer has no // opinion about the OS. - OS []wireOS `json:"os"` + // + // It is decoded on its own (RawMessage), so an OS part of the wrong shape + // refuses stream A only and never the control-plane pair beside it. + OS json.RawMessage `json:"os"` } // wireOS is one OS release in the answer. Part of the same contract. @@ -210,11 +213,19 @@ func (s HTTPSource) Target(ctx context.Context) (Target, error) { return Target{}, fmt.Errorf("updatetarget: parse the answer from %s: %w", RedactURL(url), err) } var osList []OSRelease - for _, o := range w.OS { - osList = append(osList, OSRelease{Version: o.Version, BundleURL: o.BundleURL, BundleSHA256: o.BundleSHA256}) + var osErr error + if len(w.OS) > 0 && string(w.OS) != "null" { + var raw []wireOS + if err := json.Unmarshal(w.OS, &raw); err != nil { + osErr = fmt.Errorf("%w: the os part is not a list of releases: %v", ErrOSRefused, err) + } + for _, o := range raw { + osList = append(osList, OSRelease{Version: o.Version, BundleURL: o.BundleURL, BundleSHA256: o.BundleSHA256}) + } } return Target{ OS: osList, + OSErr: osErr, Version: w.Version, BrainImage: w.BrainImage, BrainDigest: w.BrainDigest, diff --git a/internal/hostagent/updatetarget/loop.go b/internal/hostagent/updatetarget/loop.go index 65bab56a..d47236f3 100644 --- a/internal/hostagent/updatetarget/loop.go +++ b/internal/hostagent/updatetarget/loop.go @@ -445,6 +445,11 @@ func (l *Loop) tickOS(t Target, w Window, cpBusy bool) { l.recordOS(OSSnapshot{State: protocol.OSUpdateUnsupported, Detail: "this host-agent cannot update the OS"}) return } + if t.OSErr != nil { + l.recordOS(OSSnapshot{State: protocol.OSUpdateRefused, Detail: t.OSErr.Error()}) + l.quietOS(slog.LevelError, "refused:"+t.OSErr.Error(), "os update: refusing the OS part of the answer; nothing downloaded", "err", t.OSErr) + return + } if len(t.OS) == 0 { l.recordOS(OSSnapshot{State: protocol.OSUpdateNone}) l.quietOS(slog.LevelInfo, "none", "os update: the answer names no OS release; staying on this OS") diff --git a/internal/hostagent/updatetarget/os.go b/internal/hostagent/updatetarget/os.go index 3d10dd95..d6380edd 100644 --- a/internal/hostagent/updatetarget/os.go +++ b/internal/hostagent/updatetarget/os.go @@ -114,8 +114,8 @@ func (a line) less(b line) bool { // - target further back, or below the running control plane's floor: refused. // // running is this box's OS version and floor the running control plane's -// minimum_host_agent ("" when the box cannot read it, and then there is no -// floor check). current is true when the box already runs the target. +// minimum_host_agent ("" when the box cannot read it, and then every move +// to an older release is refused). current is true when the box already runs the target. func PickOS(running, floor string, list []OSRelease) (rel OSRelease, current bool, err error) { if len(list) == 0 { return OSRelease{}, false, fmt.Errorf("%w: the answer names no OS release", ErrOSRefused) @@ -138,6 +138,11 @@ func PickOS(running, floor string, list []OSRelease) (rel OSRelease, current boo break } } + // The step must be the line right after the box's own: a list that + // leaves a minor out would make the box skip it. + if nl := lineOf(canonical(rel.Version)); !(nl == line{rl.major, rl.minor + 1} || nl == line{rl.major + 1, 0}) { + return OSRelease{}, false, fmt.Errorf("%w: the next step %s is not the minor right after %s; the list leaves a minor out", ErrOSRefused, rel.Version, running) + } case tl == rl: rel = target case oneLineBack(tl, rl, list): @@ -145,6 +150,11 @@ func PickOS(running, floor string, list []OSRelease) (rel OSRelease, current boo default: return OSRelease{}, false, fmt.Errorf("%w: the target %s is more than one minor back from %s", ErrOSRefused, target.Version, running) } + // A downgrade needs the floor: without it the box cannot tell whether the + // older host-agent is one the running brain still works with. + if canonical(floor) == "" && semver.Compare(canonical(rel.Version), run) < 0 { + return OSRelease{}, false, fmt.Errorf("%w: %s is older than %s, and this box cannot read the running control plane's floor", ErrOSRefused, rel.Version, running) + } if f := canonical(floor); f != "" && semver.Compare(canonical(rel.Version), f) < 0 { return OSRelease{}, false, fmt.Errorf("%w: %s is below the running control plane's floor %s", ErrOSRefused, rel.Version, floor) } diff --git a/internal/hostagent/updatetarget/os_test.go b/internal/hostagent/updatetarget/os_test.go index a361931e..71ed8160 100644 --- a/internal/hostagent/updatetarget/os_test.go +++ b/internal/hostagent/updatetarget/os_test.go @@ -3,6 +3,8 @@ package updatetarget import ( "context" "errors" + "net/http" + "net/http/httptest" "strings" "testing" "time" @@ -54,15 +56,18 @@ func TestPickOS(t *testing.T) { }{ {name: "current", running: "1.4.2", list: []string{"1.3.9", "1.4.2"}, want: "1.4.2", current: true}, {name: "patch up", running: "1.4.0", list: []string{"1.4.2"}, want: "1.4.2"}, - {name: "patch down", running: "1.4.2", list: []string{"1.4.0"}, want: "1.4.0"}, + {name: "patch down", running: "1.4.2", floor: "0.1.0", list: []string{"1.4.0"}, want: "1.4.0"}, + {name: "a downgrade without a floor", running: "1.4.2", list: []string{"1.4.0"}, refused: true}, {name: "next minor", running: "1.3.1", list: []string{"1.3.9", "1.4.2"}, want: "1.4.2"}, {name: "never skips a minor", running: "1.2.0", list: []string{"1.2.5", "1.3.9", "1.4.2"}, want: "1.3.9"}, {name: "across a major", running: "1.9.3", list: []string{"1.9.4", "2.0.1"}, want: "2.0.1"}, - {name: "one minor back", running: "1.4.2", list: []string{"1.3.9"}, want: "1.3.9"}, - {name: "two minors back", running: "1.4.2", list: []string{"1.2.9"}, refused: true}, - {name: "one minor back across a major", running: "2.0.1", list: []string{"1.9.4"}, want: "1.9.4"}, - {name: "not the last line of the major", running: "2.0.1", list: []string{"1.8.4", "1.9.4"}, refused: false, want: "1.9.4"}, - {name: "a later line of the old major exists", running: "2.0.1", list: []string{"1.8.4", "1.9.0", "1.8.9"}, refused: true}, + {name: "a list that leaves a minor out", running: "1.2.0", list: []string{"1.4.2"}, refused: true}, + {name: "a list that leaves a major's first minor out", running: "1.9.3", list: []string{"2.1.0"}, refused: true}, + {name: "one minor back", running: "1.4.2", floor: "0.1.0", list: []string{"1.3.9"}, want: "1.3.9"}, + {name: "two minors back", running: "1.4.2", floor: "0.1.0", list: []string{"1.2.9"}, refused: true}, + {name: "one minor back across a major", running: "2.0.1", floor: "0.1.0", list: []string{"1.9.4"}, want: "1.9.4"}, + {name: "not the last line of the major", running: "2.0.1", floor: "0.1.0", list: []string{"1.8.4", "1.9.4"}, refused: false, want: "1.9.4"}, + {name: "a later line of the old major exists", running: "2.0.1", floor: "0.1.0", list: []string{"1.8.4", "1.9.0", "1.8.9"}, refused: true}, {name: "below the floor", running: "1.4.2", floor: "1.4.0", list: []string{"1.3.9"}, refused: true}, {name: "above the floor", running: "1.4.2", floor: "1.3.0", list: []string{"1.3.9"}, want: "1.3.9"}, {name: "running is not a version", running: "dev", list: []string{"1.4.2"}, refused: true}, @@ -212,3 +217,24 @@ func TestTickOS(t *testing.T) { } }) } + +func TestOSPartOfTheWrongShape(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + _, _ = w.Write([]byte(`{"version":"v1","brain_image":"` + brainRef + `","ui_image":"` + uiRef + `","os":"invalid"}`)) + })) + defer srv.Close() + tgt, err := HTTPSource{URL: srv.URL}.Target(context.Background()) + if err != nil { + t.Fatalf("a bad os part must not fail the whole answer: %v", err) + } + if tgt.BrainImage != brainRef || !errors.Is(tgt.OSErr, ErrOSRefused) { + t.Fatalf("got %+v", tgt) + } + w, _ := ParseWindow("03:00-04:00") + l, f, ap := osLoop(tgt, "0.15.0", w, time.Date(2026, 10, 3, 3, 10, 0, 0, time.Local)) + l.Current = fakeRunning{brain: oldBrain, ui: uiRef} + l.Tick(context.Background()) + if len(ap.calls) != 1 || len(f.calls) != 0 || l.Snapshot().OS.State != protocol.OSUpdateRefused { + t.Fatalf("stream B must apply and stream A be refused: %d %v %+v", len(ap.calls), f.calls, l.Snapshot().OS) + } +} diff --git a/internal/hostagent/updatetarget/target.go b/internal/hostagent/updatetarget/target.go index 49fea064..3cbf501e 100644 --- a/internal/hostagent/updatetarget/target.go +++ b/internal/hostagent/updatetarget/target.go @@ -85,6 +85,9 @@ type Target struct { // the OS, and the box stays on the OS it runs. It is checked on its own // (ValidateOS), so a bad OS part never blocks the control-plane pair. OS []OSRelease + // OSErr is set when the source could not read its own OS part (the wrong + // JSON shape). Stream A is then refused; stream B is not affected. + OSErr error } // Source is the seam: one call, one answer. diff --git a/internal/notify/notify.go b/internal/notify/notify.go index d3ef215d..2da9547d 100644 --- a/internal/notify/notify.go +++ b/internal/notify/notify.go @@ -397,8 +397,9 @@ func OSUpdateDedupKey(outcomeID string) string { return "os-update:" + outcomeID // OSUpdateOutcome tells admins what the last OS update did (UPDATES.md # 1 // step 6, NOTIFICATIONS.md # Updates): info when the box moved to the new // release, warning when the new release did not come up and the box went back -// on its own. The caller makes sure each outcome is raised once. -func (n *Notifier) OSUpdateOutcome(outcomeID, outcome, version, from string) { +// on its own. The caller makes sure each outcome is raised once. It reports +// whether a notification was raised. +func (n *Notifier) OSUpdateOutcome(outcomeID, outcome, version, from string) bool { note := Notification{ TS: n.now().UnixMilli(), Category: CategoryUpdates, @@ -420,15 +421,16 @@ func (n *Notifier) OSUpdateOutcome(outcomeID, outcome, version, from string) { note.Summary = "A system update did not work, so moose went back" note.Body = "Your moose tried to install system version " + version + " overnight. It did not start correctly, so moose went back to version " + from + " on its own. Nothing was lost. It tries again in a later night." default: - return + return false } if err := n.store.RaiseNotification(note); err != nil { slog.Error("notify: raise failed", "source_id", outcomeID, "err", err) - return + return false } n.publish(events.NotificationCreated, map[string]any{ "dedup_key": note.DedupKey, "category": string(note.Category), "severity": string(note.Severity), }) + return true } From 3a1de6a0adf1fe7443755091965ff68264adcfb7 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 18:10:33 +0100 Subject: [PATCH 07/18] test: show the grubenv and the trial timer on the broken slot (#563) --- dev/cloud/cloud-assertions.sh | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index 45140d0d..0d633f01 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -194,10 +194,10 @@ if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then for _ in $(seq 1 20); do systemctl is-active -q host-agent.service && break; sleep 1; done systemctl is-active -q host-agent.service && fail "os-revert: host-agent runs on the slot it was meant to be kept off" systemctl list-timers --all --no-pager 2>/dev/null | grep -q moose-os-trial.timer || fail "os-revert: moose-os-trial.timer is not scheduled on the trial boot" - echo "cloud-assertions: os-revert: on slot B, host-agent cannot start, trial marker present; waiting for the safety net to reboot the box" + echo "cloud-assertions: os-revert: on slot B, host-agent cannot start, trial marker present; grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' '); waiting for the safety net to reboot the box" set_os_stage 3 - sleep 900 - fail "os-revert: the safety net never rebooted the box off the broken slot" + sleep 300 + fail "os-revert: the safety net never rebooted the box off the broken slot. grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ') timer: $(systemctl list-timers --all --no-pager 2>&1 | grep moose-os-trial) service: $(journalctl -u moose-os-trial.service -b --no-pager 2>&1 | tail -5 | tr '\n' ' ')" fi # --- 1. no control-plane unit has failed. From 9bfa7bd449daa38e7b5d88e7586684c4e42ddd89 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 18:29:46 +0100 Subject: [PATCH 08/18] image: mask rauc-mark-good.service; host-agent alone marks a slot good (#563) Debian's rauc-service marks every booted slot good at the end of boot, so a slot whose host-agent never started was marked good and the box could not fall back. The os-revert boot caught it. --- dev/cloud/cloud-assertions.sh | 4 ++++ dev/cloud/mkosi.postinst.chroot | 8 ++++++++ docs/progress/host-agent-os-update.md | 1 + docs/specs/BUILD.md | 1 + 4 files changed, 14 insertions(+) diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index 0d633f01..c3456d52 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -332,6 +332,10 @@ grub-editenv /efi/grub/grubenv list | grep -qx "${BOOTED}_OK=1" || layout_fail " for u in emergency.service rescue.service; do systemctl cat "$u" 2>/dev/null | grep -q 'systemctl --no-block reboot' || layout_fail "$u has no reboot drop-in" done +# Only host-agent marks a slot good (#563): RAUC's own mark-good unit would mark +# a slot good whose host-agent never started. +[ "$(systemctl is-enabled rauc-mark-good.service 2>&1)" = masked ] \ + || layout_fail "rauc-mark-good.service is '$(systemctl is-enabled rauc-mark-good.service 2>&1)', want masked: it would mark every booted slot good" dmesg 2>/dev/null | grep -q 'moose-state: bind mounts done' || journalctl -k -b --no-pager 2>/dev/null | grep -q 'moose-state: bind mounts done' \ || layout_fail "no 'moose-state: bind mounts done' in the kernel log" # state-setup looked for the state partition on the boot disk only. diff --git a/dev/cloud/mkosi.postinst.chroot b/dev/cloud/mkosi.postinst.chroot index c13bcff4..9adfdd22 100755 --- a/dev/cloud/mkosi.postinst.chroot +++ b/dev/cloud/mkosi.postinst.chroot @@ -159,6 +159,14 @@ EOF # are committed static files under mkosi.extra/ (etc/moose/metadata-firewall.nft). enable_unit /etc/systemd/system/moose-metadata-firewall.service moose-metadata-firewall.service +# host-agent alone marks a slot good, and on a trial boot only once the brain +# answers (UPDATES.md # 1, #563). Debian's rauc-service ships +# rauc-mark-good.service, which marks every booted slot good at the end of +# boot; left on, it marked a slot good whose host-agent never started (CI run +# 37038924785), and the box could never fall back. Masked, not disabled, so a +# preset can not turn it back on. +systemctl --root=/ mask rauc-mark-good.service >/dev/null + # The OS update trial's safety net (UPDATES.md # 1, #563): reboots a box whose # new slot was never marked good, when host-agent itself cannot. A timer, so # it goes under timers.target. diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 79889e9a..3a1ba0cd 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -22,6 +22,7 @@ Closes #563, the fifth slice of #486. It follows [hosted-ab-layout.md](hosted-ab - **Switch, inside the window** (`os-switch` job). Check that RAUC reports the other slot holds the release, write the record, write the trial marker `/var/lib/moose/os-update/trial-`, run `rauc status mark-active other`, and check the grubenv reads `ORDER=" "`, `_OK=1`, `_TRY=0` before rebooting. RAUC 1.13's GRUB backend already resets `TRY` there (its `grub_set_primary`, read in the source), so the #570 gap is now a check, and a failing check puts the booted slot first again and does not reboot. A reboot that is not accepted undoes the switch. **One OS switch per night**, kept in the record across the reboot, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**: the other slot is then the way back. - **The trial, at the next start of `host-agent`** (`Boot`). With the booted slot's marker: wait for the brain's `/healthz` up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min); healthy means `mark-good`, the marker goes, the outcome is `good`; not healthy means `mark-bad booted` and a reboot. Without a marker: `mark-good` at once. With the other slot's marker: that slot failed, so the outcome is `reverted`, the slot is marked bad, the marker goes. **Only the first boot after a switch is on trial** (the maintainer's call). - **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker and the grubenv does not already mark it good (a marker host-agent failed to remove), its `host-agent` never got there, so the script marks the slot bad, leaves a note `safety-net-` and reboots. The old slot's `host-agent` logs from the note that the safety net made the revert, which the `os-revert` boot checks. The boot-proof image sets it to 90 s. +- **Only `host-agent` marks a slot good.** Debian's `rauc-service` ships `rauc-mark-good.service`, which marks every booted slot good at the end of boot. The `os-revert` boot caught it: in run 37038924785 the broken slot read `B_TRY=1` when it came up and `B_TRY=0` 90 s later with `host-agent` dead, so the safety net saw a "good" slot and did not reboot. The unit is masked in the image now, and every boot checks it (`cloud-assertions.sh` 1b). Earlier green `os-revert` runs passed only because the first safety-net script marked the slot bad without looking at the grubenv. - **Jobs.** `hostagent.Agent.StartJob` runs any job under the one lock `system-update` takes. The OS jobs carry no result. - **Record.** `state.json`: what was installed where, the last install attempt, the last switch, the last outcome with an id. Fields only grow, because either release may read it. diff --git a/docs/specs/BUILD.md b/docs/specs/BUILD.md index 1888e286..3ee094c8 100644 --- a/docs/specs/BUILD.md +++ b/docs/specs/BUILD.md @@ -234,6 +234,7 @@ One **RAUC bundle** per OS release, in the `verity` format: the slot image in a - **Signing and its checks.** The build job always signs with the throwaway signer, and then checks the bundle against the `system.conf` and keyring read back out of the slot with `unsquashfs`, not against the copies in the repo: the slot carries the keyring the build staged; the bundle verifies against the throwaway root; with the throwaway keyring the image accepts it, and with the release keyring it must **refuse** it; and an unrelated CA (a wrong key) is refused. It also rehearses the `sign` job's re-signing with a throwaway "release" CA, so that script runs on every build. A **`sign` job** of its own, after every boot passed, alone re-signs the bundle with the release signer (`dev/release/sign-bundle.sh`, `rauc resign`; the payload stays the same bytes), but only a bundle whose sha256 matches the one the build job reported, and checks it against the image's own config and keyring. The publish job then attaches it, after checking the digest `sign` reported. So the signature vouches for what the build job produced; the box's check of the digest its update target names (#563) is the control for a bad build. It refuses when the signer secrets are empty, when the image keyring is not the committed release root, or when the image would accept the throwaway-signed bundle. - **Key custody** (`DECISIONS.md` 2026-10-02): an offline root CA with the maintainer, and a signer it issued as the secrets `RAUC_SIGNING_CERT` and `RAUC_SIGNING_KEY` of the GitHub Environment `os-release` (only `main` and `v*` tags; only the `sign` job enters it). No CRLs. Making the root, issuing and rotating a signer and replacing the root are in `docs/dev/rauc-signing.md`. **Until `release-ca.pem` is committed, `release.yml` tags nothing for a merge that bumps `VERSION`** (on either line, even when the merge bumps `CONTROL_PLANE_VERSION` too), and a dispatch that publishes the OS fails at its first step. Until the secrets exist, the `sign` job refuses and nothing is published. The exact rule is in `docs/dev/contributing.md` # Release model. - **Installed by `host-agent` since #563.** The `os-update` boot proves `rauc install` into slot B (RAUC's `raw` handler takes `rootfs.img`), the switch and the trial boot, under both firmwares; `os-revert` proves the revert (`docs/dev/hosted-boot-proof.md`). RAUC's install leaves the boot order alone (`activate-installed=false` in `system.conf`): `host-agent` switches with `rauc status mark-active other` only inside the window, and checks the grubenv reads `OK=1 TRY=0` for the new slot before it reboots (the stale try flag of #570). +- **Only `host-agent` marks a slot good.** Debian's `rauc-service` ships `rauc-mark-good.service`, which marks every booted slot good at the end of boot. It is masked in the image (`mkosi.postinst.chroot`), and the boot lane checks it: left on, it marked a slot good whose `host-agent` never started (CI run 37038924785), so the box could not fall back. - **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`) and the grubenv does not mark it good (`OK=1 TRY=0`), its `host-agent` never marked it good, so it marks the slot bad, leaves a note (`/var/lib/moose/os-update/safety-net-`, which the old slot's `host-agent` logs and removes) and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 90 s (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15 min. ### The OS package lock From 41f89514680c4eee977d25aacbae192d0749b332 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 18:47:05 +0100 Subject: [PATCH 09/18] test: trace grubenv changes on the broken slot (#563) --- dev/cloud/cloud-assertions.sh | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index c3456d52..4f69d204 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -196,7 +196,16 @@ if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then systemctl list-timers --all --no-pager 2>/dev/null | grep -q moose-os-trial.timer || fail "os-revert: moose-os-trial.timer is not scheduled on the trial boot" echo "cloud-assertions: os-revert: on slot B, host-agent cannot start, trial marker present; grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' '); waiting for the safety net to reboot the box" set_os_stage 3 - sleep 300 + # Trace: report every change of the grubenv while waiting, with what ran. + prev="$(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ')" + for _i in $(seq 1 150); do + cur="$(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ')" + if [ "$cur" != "$prev" ]; then + echo "cloud-assertions: os-revert: TRACE grubenv changed at uptime $(cut -d' ' -f1 /proc/uptime): [$prev] -> [$cur]; journal: $(journalctl -b --no-pager --since '-20s' 2>&1 | grep -v CONTAINER | tail -25 | cut -c1-200 | tr '\n' '|')" > /dev/console + prev="$cur" + fi + sleep 2 + done fail "os-revert: the safety net never rebooted the box off the broken slot. grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ') timer: $(systemctl list-timers --all --no-pager 2>&1 | grep moose-os-trial) service: $(journalctl -u moose-os-trial.service -b --no-pager 2>&1 | tail -5 | tr '\n' ' ')" fi From a6543fd023c5f7d83ea506cf6add2a0c6eb317fe Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 19:06:04 +0100 Subject: [PATCH 10/18] test: log the grubenv GRUB left for each boot (#563) --- dev/cloud/cloud-assertions.sh | 5 ++++- dev/cloud/test/bootstrap.sh | 17 +++++++++++++++++ dev/cloud/test/mkosi.postinst.chroot | 1 + 3 files changed, 22 insertions(+), 1 deletion(-) diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index 4f69d204..cec2a313 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -190,6 +190,7 @@ esac # this image) must reboot the box; the next stage runs on slot A. if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then [ "$BOOTED" = B ] || fail "os-revert: stage 2 should boot slot B, booted '$BOOTED' (did GRUB skip the new slot?)" + boot_env="$(journalctl -u moose-test-grubenv.service -b --no-pager -o cat 2>/dev/null | grep 'grubenv at boot' | tail -1)" [ -e /var/lib/moose/os-update/trial-B ] || fail "os-revert: no trial marker for slot B on its trial boot" for _ in $(seq 1 20); do systemctl is-active -q host-agent.service && break; sleep 1; done systemctl is-active -q host-agent.service && fail "os-revert: host-agent runs on the slot it was meant to be kept off" @@ -206,7 +207,7 @@ if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then fi sleep 2 done - fail "os-revert: the safety net never rebooted the box off the broken slot. grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ') timer: $(systemctl list-timers --all --no-pager 2>&1 | grep moose-os-trial) service: $(journalctl -u moose-os-trial.service -b --no-pager 2>&1 | tail -5 | tr '\n' ' ')" + fail "os-revert: the safety net never rebooted the box off the broken slot. $boot_env. grubenv now: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ') timer: $(systemctl list-timers --all --no-pager 2>&1 | grep moose-os-trial) service: $(journalctl -u moose-os-trial.service -b --no-pager 2>&1 | tail -5 | tr '\n' ' ')" fi # --- 1. no control-plane unit has failed. @@ -353,6 +354,8 @@ journalctl -k -b --no-pager 2>/dev/null | grep -q "moose-state: boot disk is $ro || layout_fail "state-setup did not report $root_disk as the boot disk" [ "$(lsblk -no PKNAME "$state_dev")" = "$(basename "$root_disk")" ] || layout_fail "the state partition $state_dev is not on the boot disk $root_disk" echo "cloud-assertions: layout: rauc sees slot $BOOTED booted, grubenv has ${BOOTED}_OK=1, emergency and rescue reboot" +boot_env="$(journalctl -u moose-test-grubenv.service -b --no-pager -o cat 2>/dev/null | grep 'grubenv at boot' | tail -1)" +echo "cloud-assertions: layout: $boot_env" # --- 1c. the baked host-agent carries a real build stamp (BUILD.md # Versioning: # "every build stamps two fields"). An unstamped build reports internal/version's diff --git a/dev/cloud/test/bootstrap.sh b/dev/cloud/test/bootstrap.sh index b91b6e00..c2d10245 100755 --- a/dev/cloud/test/bootstrap.sh +++ b/dev/cloud/test/bootstrap.sh @@ -136,6 +136,23 @@ cp "${TEST_DIR}/moose-cloud-assertions.service" "$EXTRA/etc/systemd/system/" # healthy trial boot marks its slot good about 17 s after the switch in CI # (run 37031756819), which the os-update boot proves under the same setting. # The image that ships keeps 15 min. +# What GRUB left in the grubenv for this boot, logged before host-agent can +# mark anything (#563): GRUB sets the booted slot's TRY=1, and the boot lane +# checks it did, under both firmwares. +cat > "$EXTRA/etc/systemd/system/moose-test-grubenv.service" <<'EOF' +[Unit] +Description=moose test: log the grubenv GRUB left for this boot +Before=host-agent.service +After=local-fs.target +DefaultDependencies=no + +[Service] +Type=oneshot +ExecStart=/bin/sh -c 'echo "grubenv at boot: $(grub-editenv /efi/grub/grubenv list | tr "\n" " ")"' + +[Install] +WantedBy=multi-user.target +EOF mkdir -p "$EXTRA/etc/systemd/system/moose-os-trial.timer.d" cat > "$EXTRA/etc/systemd/system/moose-os-trial.timer.d/10-cloud-test.conf" <<'EOF' [Timer] diff --git a/dev/cloud/test/mkosi.postinst.chroot b/dev/cloud/test/mkosi.postinst.chroot index 7f44c74d..c2c1f57e 100755 --- a/dev/cloud/test/mkosi.postinst.chroot +++ b/dev/cloud/test/mkosi.postinst.chroot @@ -21,6 +21,7 @@ enable_unit() { # polls it, and writes its verdict to /dev/console for the QEMU harness to grep. # Test-lane only — the production image ships no assertions oneshot. enable_unit /etc/systemd/system/moose-cloud-assertions.service moose-cloud-assertions.service +enable_unit /etc/systemd/system/moose-test-grubenv.service moose-test-grubenv.service # Make the staged self-check executable (ExtraTrees preserves mode, but be # defensive — a clobbered mode would otherwise fail silently at boot). From cd29634ce73c56d68f8f821fcccfcac2be264a6b Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 19:24:36 +0100 Subject: [PATCH 11/18] os update: the safety net trusts the marker, not the grubenv (#563) GRUB under UEFI does not save the booted slot's TRY flag (CI run 37045179497), so a grubenv check made the safety net never fire under UEFI. host-agent now retries removing the marker after mark-good instead. The rauc-mark-good mask is dropped: Debian ships no such unit. --- dev/cloud/cloud-assertions.sh | 15 +------------- .../mkosi.extra/usr/lib/moose/os-trial-check | 12 ++++------- dev/cloud/mkosi.postinst.chroot | 8 -------- dev/cloud/run-cloud-tests.sh | 6 ++++++ docs/progress/host-agent-os-update.md | 5 ++--- docs/specs/BUILD.md | 3 +-- docs/specs/UPDATES.md | 2 +- internal/hostagent/osupdate/osupdate.go | 20 +++++++++++++++---- 8 files changed, 31 insertions(+), 40 deletions(-) diff --git a/dev/cloud/cloud-assertions.sh b/dev/cloud/cloud-assertions.sh index cec2a313..1b3adec5 100755 --- a/dev/cloud/cloud-assertions.sh +++ b/dev/cloud/cloud-assertions.sh @@ -197,16 +197,7 @@ if [ "$MODE" = os-revert ] && [ "$(os_stage)" = 2 ]; then systemctl list-timers --all --no-pager 2>/dev/null | grep -q moose-os-trial.timer || fail "os-revert: moose-os-trial.timer is not scheduled on the trial boot" echo "cloud-assertions: os-revert: on slot B, host-agent cannot start, trial marker present; grubenv: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' '); waiting for the safety net to reboot the box" set_os_stage 3 - # Trace: report every change of the grubenv while waiting, with what ran. - prev="$(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ')" - for _i in $(seq 1 150); do - cur="$(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ')" - if [ "$cur" != "$prev" ]; then - echo "cloud-assertions: os-revert: TRACE grubenv changed at uptime $(cut -d' ' -f1 /proc/uptime): [$prev] -> [$cur]; journal: $(journalctl -b --no-pager --since '-20s' 2>&1 | grep -v CONTAINER | tail -25 | cut -c1-200 | tr '\n' '|')" > /dev/console - prev="$cur" - fi - sleep 2 - done + sleep 300 fail "os-revert: the safety net never rebooted the box off the broken slot. $boot_env. grubenv now: $(grub-editenv /efi/grub/grubenv list 2>&1 | tr '\n' ' ') timer: $(systemctl list-timers --all --no-pager 2>&1 | grep moose-os-trial) service: $(journalctl -u moose-os-trial.service -b --no-pager 2>&1 | tail -5 | tr '\n' ' ')" fi @@ -342,10 +333,6 @@ grub-editenv /efi/grub/grubenv list | grep -qx "${BOOTED}_OK=1" || layout_fail " for u in emergency.service rescue.service; do systemctl cat "$u" 2>/dev/null | grep -q 'systemctl --no-block reboot' || layout_fail "$u has no reboot drop-in" done -# Only host-agent marks a slot good (#563): RAUC's own mark-good unit would mark -# a slot good whose host-agent never started. -[ "$(systemctl is-enabled rauc-mark-good.service 2>&1)" = masked ] \ - || layout_fail "rauc-mark-good.service is '$(systemctl is-enabled rauc-mark-good.service 2>&1)', want masked: it would mark every booted slot good" dmesg 2>/dev/null | grep -q 'moose-state: bind mounts done' || journalctl -k -b --no-pager 2>/dev/null | grep -q 'moose-state: bind mounts done' \ || layout_fail "no 'moose-state: bind mounts done' in the kernel log" # state-setup looked for the state partition on the boot disk only. diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check index 98e36a8e..baac2c47 100755 --- a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check +++ b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check @@ -15,14 +15,10 @@ set -u slot="$(sed -n 's/.*rauc\.slot=\([AB]\).*/\1/p' /proc/cmdline)" [ -n "$slot" ] || exit 0 [ -e "/var/lib/moose/os-update/trial-${slot}" ] || exit 0 -# A slot already marked good (OK=1 and TRY=0; GRUB set TRY=1 when it booted -# it) keeps its marker only because host-agent could not remove it. That is -# not a failed trial: leave the box alone. -env="$(grub-editenv /efi/grub/grubenv list 2>/dev/null)" -if printf '%s\n' "$env" | grep -qx "${slot}_OK=1" && printf '%s\n' "$env" | grep -qx "${slot}_TRY=0"; then - echo "moose-os-trial: slot ${slot} is marked good but still has its trial marker; not rebooting" - exit 0 -fi +# The marker alone decides, not the grubenv: under UEFI, GRUB does not save the +# booted slot's TRY flag (CI run 37045179497), so the grubenv cannot tell a +# slot on trial from a good one. host-agent removes the marker, with retries, +# right after it marks the slot good. echo "moose-os-trial: slot ${slot} was not marked good in time; rebooting so the box goes back to the other slot" # A note for the old slot's host-agent, so the revert says which path made it. date -u +%Y-%m-%dT%H:%M:%SZ > "/var/lib/moose/os-update/safety-net-${slot}" && sync diff --git a/dev/cloud/mkosi.postinst.chroot b/dev/cloud/mkosi.postinst.chroot index 9adfdd22..c13bcff4 100755 --- a/dev/cloud/mkosi.postinst.chroot +++ b/dev/cloud/mkosi.postinst.chroot @@ -159,14 +159,6 @@ EOF # are committed static files under mkosi.extra/ (etc/moose/metadata-firewall.nft). enable_unit /etc/systemd/system/moose-metadata-firewall.service moose-metadata-firewall.service -# host-agent alone marks a slot good, and on a trial boot only once the brain -# answers (UPDATES.md # 1, #563). Debian's rauc-service ships -# rauc-mark-good.service, which marks every booted slot good at the end of -# boot; left on, it marked a slot good whose host-agent never started (CI run -# 37038924785), and the box could never fall back. Masked, not disabled, so a -# preset can not turn it back on. -systemctl --root=/ mask rauc-mark-good.service >/dev/null - # The OS update trial's safety net (UPDATES.md # 1, #563): reboots a box whose # new slot was never marked good, when host-agent itself cannot. A timer, so # it goes under timers.target. diff --git a/dev/cloud/run-cloud-tests.sh b/dev/cloud/run-cloud-tests.sh index 570e6dd7..916bbfc7 100755 --- a/dev/cloud/run-cloud-tests.sh +++ b/dev/cloud/run-cloud-tests.sh @@ -183,6 +183,9 @@ dump_serial() { grep -niE 'cloud-assertions|moose|docker|caddy|brain|host-agent|networkd|fail' "$QEMU_SERIAL" 2>/dev/null | tail -40 >&2 || true # The whole diag block too. The tail above keeps only 40 lines, and a red # boot with many steps (the remap boots, #531) cuts the brain log out of it. + # GRUB's own messages (OVMF mirrors the EFI console to the serial port). + echo "--- serial: GRUB errors ---" >&2 + grep -aE '^error:|grub.*error|save_env|grubenv' "$QEMU_SERIAL" 2>/dev/null | head -20 >&2 || true echo "--- serial: diag block ---" >&2 sed -n '/=== MOOSE_CLOUD_DIAG ===/,/=== END MOOSE_CLOUD_DIAG ===/p' "$QEMU_SERIAL" 2>/dev/null | grep -v '^-A \|^:\|^\*' | cut -c1-2000 >&2 || true echo "--- serial: tail 30 ---" >&2 @@ -429,6 +432,9 @@ run_boot() { # skipped, a proof that never ran — was indistinguishable from one that # held. These lines are the evidence, and they belong in the CI log. grep -o 'cloud-assertions:.*' "$QEMU_SERIAL" 2>/dev/null | tr -d '\r' | sed 's/^/ /' || true + # GRUB's own error lines, if any (#563: under UEFI GRUB does not + # save the try flag; this is where it would say why). + grep -aE '^error:|save_env|grubenv' "$QEMU_SERIAL" 2>/dev/null | tr -d '\r' | head -10 | sed 's/^/ grub: /' || true # On PASS the guest powers itself off (cloud-assertions.sh ok()); wait # for QEMU to exit so the overlay write (box-id + admin) flushes before # the next boot reads it. Bounded — kill if the clean shutdown hangs. diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 3a1ba0cd..b580bd14 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -21,8 +21,7 @@ Closes #563, the fifth slice of #486. It follows [hosted-ab-layout.md](hosted-ab - **Install, ahead of the window** (`os-install` job). Download to `/var/lib/moose/os-update/bundle.raucb.part` on the state partition, hashing as it writes. Only a file whose sha256 is the one the answer names is renamed and handed to `rauc install`, so RAUC never sees a bundle the answer did not name. RAUC then checks the signature against the image's keyring. `system.conf` now has `activate-installed=false`, so the install leaves the boot order alone. The file is deleted afterwards. A failed attempt is not repeated for the same release and digest the same night. - **Switch, inside the window** (`os-switch` job). Check that RAUC reports the other slot holds the release, write the record, write the trial marker `/var/lib/moose/os-update/trial-`, run `rauc status mark-active other`, and check the grubenv reads `ORDER=" "`, `_OK=1`, `_TRY=0` before rebooting. RAUC 1.13's GRUB backend already resets `TRY` there (its `grub_set_primary`, read in the source), so the #570 gap is now a check, and a failing check puts the booted slot first again and does not reboot. A reboot that is not accepted undoes the switch. **One OS switch per night**, kept in the record across the reboot, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**: the other slot is then the way back. - **The trial, at the next start of `host-agent`** (`Boot`). With the booted slot's marker: wait for the brain's `/healthz` up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min); healthy means `mark-good`, the marker goes, the outcome is `good`; not healthy means `mark-bad booted` and a reboot. Without a marker: `mark-good` at once. With the other slot's marker: that slot failed, so the outcome is `reverted`, the slot is marked bad, the marker goes. **Only the first boot after a switch is on trial** (the maintainer's call). -- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker and the grubenv does not already mark it good (a marker host-agent failed to remove), its `host-agent` never got there, so the script marks the slot bad, leaves a note `safety-net-` and reboots. The old slot's `host-agent` logs from the note that the safety net made the revert, which the `os-revert` boot checks. The boot-proof image sets it to 90 s. -- **Only `host-agent` marks a slot good.** Debian's `rauc-service` ships `rauc-mark-good.service`, which marks every booted slot good at the end of boot. The `os-revert` boot caught it: in run 37038924785 the broken slot read `B_TRY=1` when it came up and `B_TRY=0` 90 s later with `host-agent` dead, so the safety net saw a "good" slot and did not reboot. The unit is masked in the image now, and every boot checks it (`cloud-assertions.sh` 1b). Earlier green `os-revert` runs passed only because the first safety-net script marked the slot bad without looking at the grubenv. +- **The safety net in the image.** `moose-os-trial.timer` runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its marker, its `host-agent` never got there, so the script marks the slot bad, leaves a note `safety-net-` and reboots. The old slot's `host-agent` logs from the note that the safety net made the revert, which the `os-revert` boot checks. The boot-proof image sets it to 90 s. - **Jobs.** `hostagent.Agent.StartJob` runs any job under the one lock `system-update` takes. The OS jobs carry no result. - **Record.** `state.json`: what was installed where, the last install attempt, the last switch, the last outcome with an id. Fields only grow, because either release may read it. @@ -84,7 +83,7 @@ NUMBERS-RUNS ## Review - **The fresh review agent** found one Block and two Shoulds, all fixed: the loop could install into the old slot, the way back, while the new slot was still on trial, and a multi-step update could switch twice in one night (now: nothing moves during a trial, one switch per night); a failed `rauc status` at start left a trial to the image timer (now retried for about 30 s); a failed read of the record could be written over (now never). Its nits are fixed too: two log field names, a "notified" line for an outcome that raised nothing, and a 2 GiB cap on the download. -- **Greptile** found seven, all fixed: an `os` part of the wrong JSON shape failed the whole answer (now decoded apart, stream A only); a list that left a minor out let the box skip it (now refused); the trial-overwrite case above; a failed reboot after `mark-active` left the new slot first (now undone); a stale marker on a slot already marked good would make the timer revert it (now the script checks the grubenv first); a missing floor allowed a downgrade (now every downgrade needs the floor); and a plain local `make test-cloud-qemu` had no test bundle (now the harness builds it from the local image). +- **Greptile** found seven, all fixed: an `os` part of the wrong JSON shape failed the whole answer (now decoded apart, stream A only); a list that left a minor out let the box skip it (now refused); the trial-overwrite case above; a failed reboot after `mark-active` left the new slot first (now undone); a stale marker on a slot already marked good would make the timer revert it (now `host-agent` retries the removal for a few seconds; the script cannot use the grubenv for this, because GRUB under UEFI does not save the try flag, see # Known gaps); a missing floor allowed a downgrade (now every downgrade needs the floor); and a plain local `make test-cloud-qemu` had no test bundle (now the harness builds it from the local image). ## Known gaps & deviations diff --git a/docs/specs/BUILD.md b/docs/specs/BUILD.md index 3ee094c8..1f24f641 100644 --- a/docs/specs/BUILD.md +++ b/docs/specs/BUILD.md @@ -234,8 +234,7 @@ One **RAUC bundle** per OS release, in the `verity` format: the slot image in a - **Signing and its checks.** The build job always signs with the throwaway signer, and then checks the bundle against the `system.conf` and keyring read back out of the slot with `unsquashfs`, not against the copies in the repo: the slot carries the keyring the build staged; the bundle verifies against the throwaway root; with the throwaway keyring the image accepts it, and with the release keyring it must **refuse** it; and an unrelated CA (a wrong key) is refused. It also rehearses the `sign` job's re-signing with a throwaway "release" CA, so that script runs on every build. A **`sign` job** of its own, after every boot passed, alone re-signs the bundle with the release signer (`dev/release/sign-bundle.sh`, `rauc resign`; the payload stays the same bytes), but only a bundle whose sha256 matches the one the build job reported, and checks it against the image's own config and keyring. The publish job then attaches it, after checking the digest `sign` reported. So the signature vouches for what the build job produced; the box's check of the digest its update target names (#563) is the control for a bad build. It refuses when the signer secrets are empty, when the image keyring is not the committed release root, or when the image would accept the throwaway-signed bundle. - **Key custody** (`DECISIONS.md` 2026-10-02): an offline root CA with the maintainer, and a signer it issued as the secrets `RAUC_SIGNING_CERT` and `RAUC_SIGNING_KEY` of the GitHub Environment `os-release` (only `main` and `v*` tags; only the `sign` job enters it). No CRLs. Making the root, issuing and rotating a signer and replacing the root are in `docs/dev/rauc-signing.md`. **Until `release-ca.pem` is committed, `release.yml` tags nothing for a merge that bumps `VERSION`** (on either line, even when the merge bumps `CONTROL_PLANE_VERSION` too), and a dispatch that publishes the OS fails at its first step. Until the secrets exist, the `sign` job refuses and nothing is published. The exact rule is in `docs/dev/contributing.md` # Release model. - **Installed by `host-agent` since #563.** The `os-update` boot proves `rauc install` into slot B (RAUC's `raw` handler takes `rootfs.img`), the switch and the trial boot, under both firmwares; `os-revert` proves the revert (`docs/dev/hosted-boot-proof.md`). RAUC's install leaves the boot order alone (`activate-installed=false` in `system.conf`): `host-agent` switches with `rauc status mark-active other` only inside the window, and checks the grubenv reads `OK=1 TRY=0` for the new slot before it reboots (the stale try flag of #570). -- **Only `host-agent` marks a slot good.** Debian's `rauc-service` ships `rauc-mark-good.service`, which marks every booted slot good at the end of boot. It is masked in the image (`mkosi.postinst.chroot`), and the boot lane checks it: left on, it marked a slot good whose `host-agent` never started (CI run 37038924785), so the box could not fall back. -- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`) and the grubenv does not mark it good (`OK=1 TRY=0`), its `host-agent` never marked it good, so it marks the slot bad, leaves a note (`/var/lib/moose/os-update/safety-net-`, which the old slot's `host-agent` logs and removes) and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 90 s (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15 min. +- **The trial's safety net.** `moose-os-trial.timer` (`/etc/systemd/system/`, enabled in `timers.target`) runs `/usr/lib/moose/os-trial-check` 15 minutes after boot. If the booted slot still has its trial marker (`/var/lib/moose/os-update/trial-`), its `host-agent` never marked it good, so it marks the slot bad, leaves a note (`/var/lib/moose/os-update/safety-net-`, which the old slot's `host-agent` logs and removes) and reboots, and GRUB boots the other slot. The boot-proof image sets the timer to 90 s (`dev/cloud/test/bootstrap.sh`); the image that ships keeps 15 min. ### The OS package lock diff --git a/docs/specs/UPDATES.md b/docs/specs/UPDATES.md index 29012b8e..e1a7554a 100644 --- a/docs/specs/UPDATES.md +++ b/docs/specs/UPDATES.md @@ -65,7 +65,7 @@ The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host- - **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a list whose next step is not the minor right after the box's own (a list that leaves a minor out), and a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start. **A box that cannot read that file refuses every move to an older release.** An `os` part of the wrong JSON shape is refused like a bad entry. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. - **Install, ahead of the window** (an `os-install` job, under the same lock as `system-update`). The bundle is downloaded to `/var/lib/moose/os-update/` on the state partition and its sha256 checked against the answer before RAUC sees it. Then `rauc install`, which checks the signature and writes the other slot. `system.conf` has `activate-installed=false`, so the boot order does not move. The file is deleted afterwards. A failed attempt is not repeated the same night for the same release and digest. - **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. A reboot that is not accepted undoes the switch. **One OS switch per night**, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**, because the other slot is then the way back. -- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker and the grubenv does not already mark it good, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. +- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. - **Report.** `GET /api/v1/system/version` carries `os_version` and `os_slot`; `GET /api/v1/system/update-target` carries an `os` object beside stream B's fields (`BRAIN_HOST_PROTOCOL.md`). The brain raises one admin notification per outcome (`NOTIFICATIONS.md` # Updates). Reporting back to the cloud (# 8.4 step 5) still waits for real box authentication. - **What it costs a box** (CI, `../progress/host-agent-os-update.md`): one bundle download per update (438.5 MB for a real release), the same space on the state partition while it installs (1.1% of a 40 GB disk), and one reboot. diff --git a/internal/hostagent/osupdate/osupdate.go b/internal/hostagent/osupdate/osupdate.go index c6517599..5cb73c79 100644 --- a/internal/hostagent/osupdate/osupdate.go +++ b/internal/hostagent/osupdate/osupdate.go @@ -409,10 +409,10 @@ func (a *Applier) trial(slot string) { return } // The marker goes first: once the slot is good, nothing may reboot it - // away, and the image's timer reboots any slot that still has one. - if err := os.Remove(TrialMarker(a.dir(), slot)); err != nil { - slog.Error("os update: could not remove the trial marker", "err", err, "slot", slot) - } + // away, and the image's timer reboots any slot that still has one. Tried + // a few times, because a marker left behind turns a good slot into a + // revert. + removeMarker(TrialMarker(a.dir(), slot), slot) r, err := a.load() if err != nil { slog.Error("os update: cannot read the record", "err", err) @@ -703,3 +703,15 @@ func (a *Applier) Peek(rel updatetarget.OSRelease) (state, detail string, ok boo } return "", "", false } + +// removeMarker removes a trial marker, retrying for a few seconds. +func removeMarker(path, slot string) { + var err error + for i := 0; i < 5; i++ { + if err = os.Remove(path); err == nil || errors.Is(err, os.ErrNotExist) { + return + } + time.Sleep(time.Second) + } + slog.Error("os update: could not remove the trial marker; the image's safety net will revert this good slot", "err", err, "slot", slot) +} From fe3eaa827a21953a4f372421a16a8c49bc39e0dd Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 19:39:21 +0100 Subject: [PATCH 12/18] docs: GRUB under UEFI does not save the try flag (#575) --- docs/progress/host-agent-os-update.md | 2 ++ docs/specs/BUILD.md | 2 +- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index b580bd14..768c2d28 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -87,6 +87,8 @@ NUMBERS-RUNS ## Known gaps & deviations +- **Under UEFI, GRUB does not save the try flag (#575).** Found by this slice's `os-revert` boot. The boot-proof image now logs the grubenv before `host-agent` starts (`moose-test-grubenv.service`), and every boot prints it (`layout: grubenv at boot:`). Under BIOS the booted slot has `TRY=1`; under UEFI it has `TRY=0`, on every boot, with no GRUB error (run 37047235688). This slice does not depend on it: `host-agent` and the trial timer both mark a failed slot bad (`OK=0`), so the revert works on both firmwares, which the boots prove. What it leaves open is a new slot that panics before userspace (a bad kernel or initramfs): under UEFI, `panic=10` reboots it into the same slot again. #575 tracks it; Hetzner CPX and every UEFI provider boot that way. + - **No production box moves its OS yet.** The private side has to send the `os` part (above). - **QEMU only**, as the rest of #486. #486's "Done when" still needs the same proof on a provisioned box. - **The test bundle is not a release bundle.** It is the boot-proof slot with one binary changed, gzip-compressed, signed with the throwaway key. The release bundle (xz, release signer) is checked by #562's build checks, not installed here. A release run installs nothing. diff --git a/docs/specs/BUILD.md b/docs/specs/BUILD.md index 1f24f641..4adf6489 100644 --- a/docs/specs/BUILD.md +++ b/docs/specs/BUILD.md @@ -160,7 +160,7 @@ Every byte the OS reserves is taken from every box for life, so the budget is me GRUB boots the box under UEFI (`x86_64-efi`) and under legacy BIOS (`i386-pc`), from **one** `grub.cfg` and **one** `grubenv` on the ESP. This replaces the hosted image's split of systemd-boot for UEFI and GRUB for BIOS. -- The slot choice is RAUC's GRUB backend: `ORDER`, `_OK` and `_TRY` in `grubenv`. GRUB boots the first slot in `ORDER` that is good and not yet tried, and sets its `TRY` before booting it. `rauc status mark-good` clears it. A slot that never gets there is skipped on the next boot. **One attempt per update.** +- The slot choice is RAUC's GRUB backend: `ORDER`, `_OK` and `_TRY` in `grubenv`. GRUB boots the first slot in `ORDER` that is good and not yet tried, and sets its `TRY` before booting it. `rauc status mark-good` clears it. A slot that never gets there is skipped on the next boot. **One attempt per update.** **Under UEFI, GRUB does not save `TRY` today** (CI run 37045179497, #575): the OS update's own paths do not need it, because `host-agent` and the trial timer mark a failed slot bad (#563), but a slot that panics before userspace is booted again under UEFI until #575 is fixed. - When no slot is left to try, GRUB falls back to the **known-good slot** rather than stopping at a prompt nobody will see: the last good slot in `ORDER`, because an install puts the new slot first and the slot it came from last. **Only the fallback slot's `TRY` is cleared.** A slot that failed keeps `TRY=1` and is never booted again until a new install into it resets its flag. As built (#563), the old slot's `host-agent` also marks a failed slot bad (`OK=0`), and the switch to a newly installed slot (`rauc status mark-active other`) sets its `OK=1 TRY=0`. - The kernel and initramfs live **inside** each slot, where GRUB reads them from the slot's squashfs (`squash4` and `xzio`). A slot is one self-contained unit: the kernel always matches the root it boots. - A slot that hangs instead of rebooting would never revert, so the image reboots on emergency and rescue, sets `panic=`, and runs a watchdog where the machine has one (`UPDATES.md` # 1). From b107ba82d768f182541ce8bf41d70237dd7d9d25 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 19:55:23 +0100 Subject: [PATCH 13/18] docs: numbers and runs for #563 --- docs/progress/host-agent-os-update.md | 29 +++++++++++++++++++++++---- 1 file changed, 25 insertions(+), 4 deletions(-) diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 768c2d28..9da5421e 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -61,15 +61,36 @@ From runs on this branch (the final run is in # How it was verified): |---|---| | Download per update (a real release) | 438.5 MB, one bundle (`rauc-bundle.md`) | | Disk the box needs while it installs | 438.5 MB on the state partition, 1.1% of a 40 GB disk, deleted after the install | -| Time to apply, measured in the guest | NUMBERS-APPLY | -| Test bundle in CI | 366.4 MB, built in 27 s (unsquashfs 4 s, mksquashfs gzip 4 s), uploaded in 5 s | -| CI cost of the two boots | NUMBERS-CI | +| Time to apply, measured in the guest (run 37048902317) | download and digest 0.5 to 0.7 s, `rauc install` 2.8 to 4.8 s, switch to marked good 16 to 18 s with the reboot. The guest reads the bundle from a local file server, so a real box adds the download: about 4 s at 1 Gbit/s, 35 s at 100 Mbit/s | +| Time to revert on its own (`os-revert`) | 109 to 117 s from the switch: two reboots and the 90 s test safety net. With the 15 min timer that ships, about 16 min; with a live `host-agent` that finds the brain unhealthy, the 10 min trial | +| Test bundle in CI | 366.4 MB, built in 26 to 27 s (unsquashfs 4 s, mksquashfs gzip 4 s), uploaded in 5 s | + +**CI cost**, before and after (wall time from the first job's start to the last job's end; runner minutes summed over jobs): + +| Run | What | Wall | Runner minutes | +|---|---|---|---| +| 37001109373 | full list before #562 (`ci-cloud-image-speedup.md`) | 12.2 min | 33.7 | +| 37017805364 | full list of #562 | 12.7 min | 33.0 | +| 37019713707 | full list of #562, another run | 15.9 min | 37.8 | +| 37033699687 | full list of this branch, before the review fixes | 14.9 min | 48.2 | +| **37048902317** | **full list of this branch, final head** | **14.8 min** | **49.8** | + +The two new boot groups add four jobs: `os-update` 3.0 to 3.5 min and `os-revert` 3.4 to 5.2 min each, about 15 runner minutes, plus the test bundle (about 0.5 min in the build job). The wall time moves by the bundle step and by `os-revert` when it is the longest job: about +2 min against #562's 12.7 min run, inside the +3 to 4 min the maintainer asked for. The build job itself varies more than that between runs (8.2 to 10.7 min). ## How it was verified All in CI, every publish input false. Never built or booted locally. -NUMBERS-RUNS +| Run | Boots | Result | +|---|---|---| +| 37028201385 | `os-update os-revert` | red: the bundle script looked for `rauc.sh` one directory too high | +| 37029567427 | `os-update os-revert` | red at the last check, after every step worked under both firmwares: install, switch, trial good, safety-net revert. The in-guest update source had no restart policy, so after the reboot the box could not read its target and `os.state` read `none` | +| 37031756819 | `os-update os-revert` | green, all four jobs | +| 37033699687 | the full list | green, all 14 jobs | +| 37035445476 | the full list, after the review fixes | red: `os-revert` under UEFI. The new grubenv check in the safety net never fired | +| 37038924785, 37043066156, 37045179497 | `os-revert`, then `unseeded os-revert`, with traces | found the cause: under UEFI, GRUB leaves the booted slot's `TRY` at 0 (#575) | +| 37047235688 | `unseeded os-revert` | green, after the safety net went back to trusting the marker | +| **37048902317** | **the full list, final head `fe3eaa8`** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS** | - **Tests.** `internal/hostagent/updatetarget/os_test.go` (the checks, every pick rule, the loop: install outside the window and switch inside it, stream B first, a bad OS part refused with stream B still applying, an answer with only an OS part, current, none, unsupported). `internal/hostagent/osupdate/osupdate_test.go` against a fake two-slot RAUC (a normal boot marked good at once, install then switch, a wrong digest never reaching RAUC and not retried the same night, a stale `TRY` stopping the switch and putting the booted slot back, a busy lock, a good trial, a failed trial rebooting and the old slot recording the revert and holding, the floor file, the JSON and grubenv parsers, `Peek`). The report's `os` part, the brain's pass-through, the notification, the store lookup and the brain's outcome check have tests too. `make check` green. From c9952cf7e9399fc141de80b5d97fee17e6ca67d5 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 19:55:47 +0100 Subject: [PATCH 14/18] docs: record the dismissed Greptile point (#563) --- docs/progress/host-agent-os-update.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 9da5421e..0285c8d0 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -106,6 +106,8 @@ All in CI, every publish input false. Never built or booted locally. - **The fresh review agent** found one Block and two Shoulds, all fixed: the loop could install into the old slot, the way back, while the new slot was still on trial, and a multi-step update could switch twice in one night (now: nothing moves during a trial, one switch per night); a failed `rauc status` at start left a trial to the image timer (now retried for about 30 s); a failed read of the record could be written over (now never). Its nits are fixed too: two log field names, a "notified" line for an outcome that raised nothing, and a 2 GiB cap on the download. - **Greptile** found seven, all fixed: an `os` part of the wrong JSON shape failed the whole answer (now decoded apart, stream A only); a list that left a minor out let the box skip it (now refused); the trial-overwrite case above; a failed reboot after `mark-active` left the new slot first (now undone); a stale marker on a slot already marked good would make the timer revert it (now `host-agent` retries the removal for a few seconds; the script cannot use the grubenv for this, because GRUB under UEFI does not save the try flag, see # Known gaps); a missing floor allowed a downgrade (now every downgrade needs the floor); and a plain local `make test-cloud-qemu` had no test bundle (now the harness builds it from the local image). +- **Greptile, on the later commits, dismissed:** the `grubenv at boot` line is printed but not checked. On purpose: checking `TRY=1` would fail every UEFI boot until #575 is fixed, and #575's "Done when" adds that check. + ## Known gaps & deviations - **Under UEFI, GRUB does not save the try flag (#575).** Found by this slice's `os-revert` boot. The boot-proof image now logs the grubenv before `host-agent` starts (`moose-test-grubenv.service`), and every boot prints it (`layout: grubenv at boot:`). Under BIOS the booted slot has `TRY=1`; under UEFI it has `TRY=0`, on every boot, with no GRUB error (run 37047235688). This slice does not depend on it: `host-agent` and the trial timer both mark a failed slot bad (`OK=0`), so the revert works on both firmwares, which the boots prove. What it leaves open is a new slot that panics before userspace (a bad kernel or initramfs): under UEFI, `panic=10` reboots it into the same slot again. #575 tracks it; Hetzner CPX and every UEFI provider boot that way. From b63859456e198587d5056a8ec7f0fa9bed7e7ec8 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 20:02:17 +0100 Subject: [PATCH 15/18] fixup! os update: review fixes (#563) Second review: a trial that cannot mark its slot reboots at most once, the image timer reboots a slot at most once, a switch counts only once activated, leftover downloads are removed, the trial timeout is capped under the image timer, and a stale marker on a good slot is only removed. --- cmd/host-agent-real/osupdate.go | 6 + .../mkosi.extra/usr/lib/moose/os-trial-check | 7 + docs/progress/host-agent-os-update.md | 1 + docs/specs/UPDATES.md | 2 +- internal/hostagent/osupdate/osupdate.go | 139 +++++++++++++---- internal/hostagent/osupdate/osupdate_test.go | 140 +++++++++++++++++- 6 files changed, 257 insertions(+), 38 deletions(-) diff --git a/cmd/host-agent-real/osupdate.go b/cmd/host-agent-real/osupdate.go index b3740e5f..979f518e 100644 --- a/cmd/host-agent-real/osupdate.go +++ b/cmd/host-agent-real/osupdate.go @@ -49,6 +49,12 @@ func osTrialTimeout() time.Duration { slog.Warn("os update: "+envOSTrialTimeout+" is not a positive duration; using the default", "err", err) return osupdate.DefaultTrialTimeout } + // Under the image's 15 min timer, so a slow but healthy slot is decided + // by host-agent and never reverted by the timer. + if d > osupdate.MaxTrialTimeout { + slog.Warn("os update: "+envOSTrialTimeout+" is longer than the image's trial timer allows; using the longest allowed", "err", fmt.Errorf("%s > %s", d, osupdate.MaxTrialTimeout)) + return osupdate.MaxTrialTimeout + } return d } diff --git a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check index baac2c47..dee7e7be 100755 --- a/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check +++ b/dev/cloud/mkosi.extra/usr/lib/moose/os-trial-check @@ -19,6 +19,13 @@ slot="$(sed -n 's/.*rauc\.slot=\([AB]\).*/\1/p' /proc/cmdline)" # booted slot's TRY flag (CI run 37045179497), so the grubenv cannot tell a # slot on trial from a good one. host-agent removes the marker, with retries, # right after it marks the slot good. +# Once per slot: if this script already rebooted this slot and the box came +# back to it (the mark-bad failed, and under UEFI GRUB does not save the try +# flag, #575), rebooting again would only loop. Stay up for a person. +if [ -e "/var/lib/moose/os-update/safety-net-${slot}" ]; then + echo "moose-os-trial: slot ${slot} was already rebooted once by this safety net and booted again; not rebooting again" + exit 0 +fi echo "moose-os-trial: slot ${slot} was not marked good in time; rebooting so the box goes back to the other slot" # A note for the old slot's host-agent, so the revert says which path made it. date -u +%Y-%m-%dT%H:%M:%SZ > "/var/lib/moose/os-update/safety-net-${slot}" && sync diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 0285c8d0..426fc42c 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -106,6 +106,7 @@ All in CI, every publish input false. Never built or booted locally. - **The fresh review agent** found one Block and two Shoulds, all fixed: the loop could install into the old slot, the way back, while the new slot was still on trial, and a multi-step update could switch twice in one night (now: nothing moves during a trial, one switch per night); a failed `rauc status` at start left a trial to the image timer (now retried for about 30 s); a failed read of the record could be written over (now never). Its nits are fixed too: two log field names, a "notified" line for an outcome that raised nothing, and a 2 GiB cap on the download. - **Greptile** found seven, all fixed: an `os` part of the wrong JSON shape failed the whole answer (now decoded apart, stream A only); a list that left a minor out let the box skip it (now refused); the trial-overwrite case above; a failed reboot after `mark-active` left the new slot first (now undone); a stale marker on a slot already marked good would make the timer revert it (now `host-agent` retries the removal for a few seconds; the script cannot use the grubenv for this, because GRUB under UEFI does not save the try flag, see # Known gaps); a missing floor allowed a downgrade (now every downgrade needs the floor); and a plain local `make test-cloud-qemu` had no test bundle (now the harness builds it from the local image). +- **Second review (a fresh agent and Greptile, no Block), all fixed:** a trial whose RAUC mark failed rebooted into the same slot for ever under UEFI (#575): now a slot it cannot mark bad is rebooted once (`trial_reboots` in the record), a slot it cannot mark good is not rebooted, and the image's timer reboots a slot at most once (its `safety-net-` note). A power cut between the trial marker and `mark-active` read as a revert and marked the freshly installed slot bad: the record now has `activated`, written after `mark-active`, and a marker without it is only removed. A `bundle.raucb` or `.part` left by a kill is deleted at start and before each install. `MOOSE_OS_TRIAL_TIMEOUT` is capped at 14 min, under the 15 min timer. A marker whose removal failed is retried for about 13 min, and a later boot that finds the good outcome for it only removes the marker. Greptile's "OS switches on a later reboot" was already fixed in `db47ccc` (the switch is undone when the reboot is not accepted). - **Greptile, on the later commits, dismissed:** the `grubenv at boot` line is printed but not checked. On purpose: checking `TRY=1` would fail every UEFI boot until #575 is fixed, and #575's "Done when" adds that check. ## Known gaps & deviations diff --git a/docs/specs/UPDATES.md b/docs/specs/UPDATES.md index e1a7554a..342e1e8a 100644 --- a/docs/specs/UPDATES.md +++ b/docs/specs/UPDATES.md @@ -65,7 +65,7 @@ The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host- - **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a list whose next step is not the minor right after the box's own (a list that leaves a minor out), and a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start. **A box that cannot read that file refuses every move to an older release.** An `os` part of the wrong JSON shape is refused like a bad entry. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. - **Install, ahead of the window** (an `os-install` job, under the same lock as `system-update`). The bundle is downloaded to `/var/lib/moose/os-update/` on the state partition and its sha256 checked against the answer before RAUC sees it. Then `rauc install`, which checks the signature and writes the other slot. `system.conf` has `activate-installed=false`, so the boot order does not move. The file is deleted afterwards. A failed attempt is not repeated the same night for the same release and digest. - **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. A reboot that is not accepted undoes the switch. **One OS switch per night**, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**, because the other slot is then the way back. -- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. +- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min, at most 14 so it always decides before the image's timer). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. Only a switch that took effect counts (the record's `activated` flag is written after `mark-active` succeeds), so a power cut before it never reads as a revert. When RAUC cannot mark the slot, the trial does not loop: a slot it cannot mark bad is rebooted once, a slot it cannot mark good is not rebooted, and the image's timer reboots a slot at most once; after that the box stays up for a person. A download left by a kill or a power cut is deleted at the next start. - **Report.** `GET /api/v1/system/version` carries `os_version` and `os_slot`; `GET /api/v1/system/update-target` carries an `os` object beside stream B's fields (`BRAIN_HOST_PROTOCOL.md`). The brain raises one admin notification per outcome (`NOTIFICATIONS.md` # Updates). Reporting back to the cloud (# 8.4 step 5) still waits for real box authentication. - **What it costs a box** (CI, `../progress/host-agent-os-update.md`): one bundle download per update (438.5 MB for a real release), the same space on the state partition while it installs (1.1% of a 40 GB disk), and one reboot. diff --git a/internal/hostagent/osupdate/osupdate.go b/internal/hostagent/osupdate/osupdate.go index 5cb73c79..7f9cf792 100644 --- a/internal/hostagent/osupdate/osupdate.go +++ b/internal/hostagent/osupdate/osupdate.go @@ -63,6 +63,45 @@ const DefaultDir = "/var/lib/moose/os-update" // a live host-agent always decides first. const DefaultTrialTimeout = 10 * time.Minute +// MaxTrialTimeout is the longest trial allowed. It stays under the image's +// moose-os-trial.timer (15 min), so a live host-agent always decides first +// and the timer never reverts a slot that is slow but healthy. +const MaxTrialTimeout = 14 * time.Minute + +// markerRetries and markerRetry bound removeMarker. Vars for tests. +var ( + markerRetries = 80 + markerRetry = 10 * time.Second +) + +// markAttempts and markRetry bound the retries of one rauc mark. Vars for tests. +var ( + markAttempts = 3 + markRetry = 2 * time.Second +) + +// mark runs a rauc mark, retried a few times. +func (a *Applier) mark(state, which string) error { + var err error + for i := 0; i < markAttempts; i++ { + if err = a.RAUC.Mark(context.Background(), state, which); err == nil { + return nil + } + time.Sleep(markRetry) + } + return err +} + +// removeLeftovers deletes a bundle a kill or a power cut left behind mid-download +// or mid-install: about 438 MB on the state partition. +func (a *Applier) removeLeftovers() { + for _, f := range []string{"bundle.raucb", "bundle.raucb.part"} { + if err := os.Remove(filepath.Join(a.dir(), f)); err != nil && !errors.Is(err, os.ErrNotExist) { + slog.Error("os update: could not remove a leftover download", "err", err, "src", f) + } + } +} + // statusAttempts and statusRetry bound how long Boot waits for RAUC to // answer: about 30 s. Vars, so a test does not wait. var ( @@ -174,6 +213,15 @@ type switched struct { To string `json:"to"` Night time.Time `json:"night"` At time.Time `json:"at"` + // Activated is set once mark-active succeeded and the grubenv was + // checked. A marker for the other slot without it is left by a power cut + // before the switch took effect, not by a failed trial. + Activated bool `json:"activated,omitempty"` + // TrialReboots counts the reboots a trial made without managing to mark + // the slot bad. Without the mark a slot can boot again (GRUB does not save + // the try flag under UEFI, #575), so after one the trial stays up and + // leaves the decision to the image's timer. + TrialReboots int `json:"trial_reboots,omitempty"` } func (a *Applier) dir() string { @@ -315,12 +363,23 @@ func (a *Applier) Boot(ctx context.Context) { a.slot = st.Booted a.mu.Unlock() booted, oth := st.Booted, other(st.Booted) + a.removeLeftovers() + r, rerr := a.load() + if rerr != nil { + slog.Error("os update: cannot read the record", "err", rerr) + } if _, err := os.Stat(TrialMarker(a.dir(), booted)); err == nil { - slog.Info("os update: this boot is on trial; waiting for the brain before marking the slot good", - "os", a.Version, "slot", booted) - go a.trial(booted) - return + if s := r.Switch; s != nil && r.Last != nil && r.Last.Outcome == protocol.OSOutcomeGood && r.Last.ID == outcomeID(s) && s.To == booted { + // The trial already passed; only the marker removal failed. + slog.Warn("os update: removing a trial marker left on a slot already marked good", "slot", booted) + go removeMarker(TrialMarker(a.dir(), booted), booted) + } else { + slog.Info("os update: this boot is on trial; waiting for the brain before marking the slot good", + "os", a.Version, "slot", booted) + go a.trial(booted) + return + } } // Not on trial: the slot is good. Marked at once, so an unrelated problem @@ -335,11 +394,15 @@ func (a *Applier) Boot(ctx context.Context) { if _, err := os.Stat(TrialMarker(a.dir(), oth)); err != nil { return } - // The other slot was on trial and the box is back here: it reverted. - r, err := a.load() - if err != nil { - slog.Error("os update: cannot read the record", "err", err) + if s := r.Switch; rerr == nil && (s == nil || !s.Activated || s.To != oth) { + // The marker was written but the switch never took effect (a power + // cut before mark-active): nothing was tried, nothing reverted. The + // slot keeps what was installed into it. + slog.Warn("os update: removing a trial marker from a switch that never took effect", "slot", oth) + go removeMarker(TrialMarker(a.dir(), oth), oth) + return } + // The other slot was on trial and the box is back here: it reverted. out := &protocol.OSOutcome{Outcome: protocol.OSOutcomeReverted, From: a.Version, At: a.now().UTC().Format(time.RFC3339)} if r.Switch != nil { out.Version = r.Switch.Version @@ -348,7 +411,7 @@ func (a *Applier) Boot(ctx context.Context) { } else { out.ID = "os-reverted-" + a.now().UTC().Format("20060102T150405Z") } - if err := a.RAUC.Mark(ctx, "bad", "other"); err != nil { + if err := a.mark("bad", "other"); err != nil { slog.Error("os update: could not mark the failed slot bad", "err", err, "slot", oth) } r.Last = out @@ -356,9 +419,7 @@ func (a *Applier) Boot(ctx context.Context) { if err := a.save(r); err != nil { slog.Error("os update: could not write the record", "err", err) } - if err := os.Remove(TrialMarker(a.dir(), oth)); err != nil { - slog.Error("os update: could not remove the trial marker", "err", err, "slot", oth) - } + go removeMarker(TrialMarker(a.dir(), oth), oth) slog.Warn("os update: the new slot did not come up healthy; the box went back to the old slot", "os", out.Version, "slot", booted) // Which path made the revert: the new slot's host-agent (its trial timed @@ -383,36 +444,45 @@ func (a *Applier) trial(slot string) { if timeout <= 0 { timeout = DefaultTrialTimeout } + if timeout > MaxTrialTimeout { + timeout = MaxTrialTimeout + } ctx, cancel := context.WithTimeout(context.Background(), timeout) err := a.Healthy(ctx) cancel() - bg := context.Background() if err != nil { slog.Error("os update: the new slot is not healthy; marking it bad and rebooting to the old slot", "err", err, "os", a.Version, "slot", slot) - if err := a.RAUC.Mark(bg, "bad", "booted"); err != nil { - slog.Error("os update: could not mark the slot bad; rebooting anyway, GRUB skips a slot still on trial", "err", err) + if merr := a.mark("bad", "booted"); merr != nil { + // Without the mark this slot can boot again (#575), so reboot + // once at most; after that stay up and let the image's timer + // decide, rather than reboot into the same slot for ever. + r, lerr := a.load() + if lerr != nil || r.Switch == nil || r.Switch.TrialReboots >= 1 { + slog.Error("os update: could not mark the slot bad, and already rebooted once for it; staying up, the image's safety net decides", "err", merr, "slot", slot) + return + } + r.Switch.TrialReboots++ + if serr := a.save(r); serr != nil { + slog.Error("os update: could not write the record; staying up, the image's safety net decides", "err", serr, "slot", slot) + return + } + slog.Error("os update: could not mark the slot bad; rebooting once, GRUB skips a slot still on trial under BIOS", "err", merr, "slot", slot) } if err := a.reboot(); err != nil { slog.Error("os update: the trial cannot reboot; the image's safety net reboots the box", "err", err, "slot", slot) } return } - if err := a.RAUC.Mark(bg, "good", "booted"); err != nil { - // Without the mark GRUB skips this slot on the next boot. Reboot now, - // inside the window, rather than leave a slot that will revert at - // some random later reboot. - slog.Error("os update: could not mark the new slot good; rebooting to the old slot", "err", err, "slot", slot) - if err := a.reboot(); err != nil { - slog.Error("os update: the trial cannot reboot; the image's safety net reboots the box", "err", err, "slot", slot) - } + if err := a.mark("good", "booted"); err != nil { + // Do not reboot: with RAUC failing the next boot could be this slot + // again, for ever. Stay up with the marker, and the image's timer + // gives the slot up. + slog.Error("os update: could not mark the new slot good; staying up, the image's safety net decides", "err", err, "slot", slot) return } - // The marker goes first: once the slot is good, nothing may reboot it - // away, and the image's timer reboots any slot that still has one. Tried - // a few times, because a marker left behind turns a good slot into a - // revert. - removeMarker(TrialMarker(a.dir(), slot), slot) + // The outcome is recorded before the marker goes: if the removal fails, + // the next boot finds the good outcome and only removes the marker. r, err := a.load() if err != nil { slog.Error("os update: cannot read the record", "err", err) @@ -429,6 +499,7 @@ func (a *Applier) trial(slot string) { if err := a.save(r); err != nil { slog.Error("os update: could not write the record", "err", err) } + removeMarker(TrialMarker(a.dir(), slot), slot) slog.Info("os update: the new slot is healthy and marked good", "os", a.Version, "slot", slot) } @@ -551,6 +622,7 @@ func (a *Applier) doInstall(ctx context.Context, rel updatetarget.OSRelease, nig if err := os.MkdirAll(a.dir(), 0o700); err != nil { return err } + a.removeLeftovers() path := filepath.Join(a.dir(), "bundle.raucb") defer os.Remove(path) slog.Info("os update: downloading the bundle", "os", rel.Version, "slot", target, "url", updatetarget.RedactURL(rel.BundleURL)) @@ -664,6 +736,10 @@ func (a *Applier) doSwitch(ctx context.Context, rel updatetarget.OSRelease, nigh return undo(fmt.Errorf("after mark-active the grubenv is ORDER=%q %s_OK=%q %s_TRY=%q; not rebooting", env["ORDER"], target, env[target+"_OK"], target, env[target+"_TRY"])) } + r.Switch.Activated = true + if err := a.save(r); err != nil { + return undo(fmt.Errorf("record the switch: %w", err)) + } slog.Info("os update: switching slots and rebooting", "os", rel.Version, "slot", target) // A reboot that was not accepted must not leave the new slot first: an // unrelated reboot later would then switch the OS outside the window. @@ -704,14 +780,15 @@ func (a *Applier) Peek(rel updatetarget.OSRelease) (state, detail string, ok boo return "", "", false } -// removeMarker removes a trial marker, retrying for a few seconds. +// removeMarker removes a trial marker, retrying until it is gone or the image's +// timer would act (markerRetries x markerRetry, about 13 min). func removeMarker(path, slot string) { var err error - for i := 0; i < 5; i++ { + for i := 0; i < markerRetries; i++ { if err = os.Remove(path); err == nil || errors.Is(err, os.ErrNotExist) { return } - time.Sleep(time.Second) + time.Sleep(markerRetry) } slog.Error("os update: could not remove the trial marker; the image's safety net will revert this good slot", "err", err, "slot", slot) } diff --git a/internal/hostagent/osupdate/osupdate_test.go b/internal/hostagent/osupdate/osupdate_test.go index 93053f2f..72fde8b5 100644 --- a/internal/hostagent/osupdate/osupdate_test.go +++ b/internal/hostagent/osupdate/osupdate_test.go @@ -293,9 +293,7 @@ func TestTrialUnhealthyRebootsAndOldSlotRecordsRevert(t *testing.T) { if last == nil || last.Outcome != protocol.OSOutcomeReverted || last.Version != "0.15.1" { t.Fatalf("outcome %+v", last) } - if _, err := os.Stat(TrialMarker(a.dir(), "B")); !errors.Is(err, os.ErrNotExist) { - t.Fatal("the trial marker is still there") - } + waitFor(t, func() bool { _, err := os.Stat(TrialMarker(a.dir(), "B")); return errors.Is(err, os.ErrNotExist) }) if d := a.Apply(h.rel, true, h.night); d.State != protocol.OSUpdateHeld { t.Fatalf("after a revert the same release waits for the next night, got %+v", d) } @@ -429,9 +427,6 @@ func (f *flakyRAUC) Status(ctx context.Context) (Status, error) { } func TestBootRetriesStatus(t *testing.T) { - old := statusRetry - statusRetry = time.Millisecond - defer func() { statusRetry = old }() r := &flakyRAUC{fakeRAUC: newFakeRAUC("A"), fails: 3} a := &Applier{RAUC: r, Jobs: &syncJobs{}, Version: "0.15.0", Dir: t.TempDir()} a.Boot(context.Background()) @@ -471,3 +466,136 @@ func TestFailedRebootUndoesTheSwitch(t *testing.T) { t.Fatal("the trial marker was left behind") } } + +// fastRetries is a no-op kept for readability: TestMain makes every retry +// fast for the whole package, once, so no goroutine a test leaves behind ever +// races a restore. +func fastRetries(*testing.T) {} + +func TestMain(m *testing.M) { + markAttempts, markRetry, markerRetries, markerRetry, statusRetry = 2, time.Millisecond, 3, time.Millisecond, time.Millisecond + os.Exit(m.Run()) +} + +// failingMark makes one rauc mark state fail. +type failingMark struct { + *fakeRAUC + state string +} + +func (f failingMark) Mark(ctx context.Context, state, which string) error { + if state == f.state { + return errors.New("rauc: d-bus timeout") + } + return f.fakeRAUC.Mark(ctx, state, which) +} + +// A power cut between the marker and mark-active leaves a marker for a switch +// that never took effect: no revert, no bad mark, no warning (review of #563). +func TestMarkerWithoutActivationIsNoRevert(t *testing.T) { + fastRetries(t) + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + r, _ := h.a.load() + r.Switch = &switched{Version: h.rel.Version, From: "A", To: "B", Night: h.night, At: time.Now()} + if err := h.a.save(r); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(TrialMarker(h.a.dir(), "B"), []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + a := h.reboot(t, "A", "0.15.0") + a.Boot(context.Background()) + if a.Last() != nil { + t.Fatalf("a switch that never took effect was recorded as %+v", a.Last()) + } + for _, m := range h.rauc.marks { + if m == "bad:B" { + t.Fatal("the installed slot was marked bad") + } + } + waitFor(t, func() bool { _, err := os.Stat(TrialMarker(a.dir(), "B")); return errors.Is(err, os.ErrNotExist) }) +} + +// A marker left on a slot whose trial already passed is only removed: no new +// trial, and the image's timer then finds nothing. +func TestStaleMarkerOnAGoodSlot(t *testing.T) { + fastRetries(t) + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + b := h.reboot(t, "B", "0.15.1") + b.Boot(context.Background()) + waitFor(t, func() bool { return b.Last() != nil }) + waitFor(t, func() bool { _, err := os.Stat(TrialMarker(b.dir(), "B")); return errors.Is(err, os.ErrNotExist) }) + if err := os.WriteFile(TrialMarker(b.dir(), "B"), []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + h.healthy = errors.New("would fail a new trial") + before := h.reboots.Load() + b2 := h.reboot(t, "B", "0.15.1") + b2.Boot(context.Background()) + waitFor(t, func() bool { _, err := os.Stat(TrialMarker(b2.dir(), "B")); return errors.Is(err, os.ErrNotExist) }) + if h.reboots.Load() != before { + t.Fatal("a stale marker started a new trial that rebooted") + } +} + +// When RAUC cannot mark the slot bad, the trial reboots once, then stays up. +func TestTrialRebootsOnceWithoutAMark(t *testing.T) { + fastRetries(t) + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + h.healthy = errors.New("brain down") + fm := failingMark{fakeRAUC: h.rauc, state: "bad"} + boot := func() int32 { + before := h.reboots.Load() + b := h.reboot(t, "B", "0.15.1") + b.RAUC = fm + b.Boot(context.Background()) + time.Sleep(50 * time.Millisecond) + return h.reboots.Load() - before + } + if n := boot(); n != 1 { + t.Fatalf("first trial: %d reboots, want 1", n) + } + if n := boot(); n != 0 { + t.Fatalf("second trial on the same slot: %d reboots, want 0 (stay up for the safety net)", n) + } +} + +// When RAUC cannot mark the slot good, the trial does not reboot. +func TestTrialStaysUpWhenMarkGoodFails(t *testing.T) { + fastRetries(t) + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + before := h.reboots.Load() + b := h.reboot(t, "B", "0.15.1") + b.RAUC = failingMark{fakeRAUC: h.rauc, state: "good"} + b.Boot(context.Background()) + time.Sleep(50 * time.Millisecond) + if h.reboots.Load() != before || b.Last() != nil { + t.Fatalf("reboots %d, outcome %+v", h.reboots.Load()-before, b.Last()) + } + if _, err := os.Stat(TrialMarker(b.dir(), "B")); err != nil { + t.Fatal("the marker must stay for the image's timer") + } +} + +func TestBootRemovesLeftovers(t *testing.T) { + h := newHarness(t, "A") + for _, f := range []string{"bundle.raucb", "bundle.raucb.part"} { + if err := os.WriteFile(h.a.dir()+"/"+f, []byte("x"), 0o600); err != nil { + t.Fatal(err) + } + } + a := h.reboot(t, "A", "0.15.0") + a.Boot(context.Background()) + for _, f := range []string{"bundle.raucb", "bundle.raucb.part"} { + if _, err := os.Stat(h.a.dir() + "/" + f); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("%s was left", f) + } + } +} From db4923e94a32f66d75797cac6ef6d86fa0516762 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 20:32:09 +0100 Subject: [PATCH 16/18] fixup! docs: numbers and runs for #563 --- docs/progress/host-agent-os-update.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 426fc42c..7224f605 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -73,7 +73,8 @@ From runs on this branch (the final run is in # How it was verified): | 37017805364 | full list of #562 | 12.7 min | 33.0 | | 37019713707 | full list of #562, another run | 15.9 min | 37.8 | | 37033699687 | full list of this branch, before the review fixes | 14.9 min | 48.2 | -| **37048902317** | **full list of this branch, final head** | **14.8 min** | **49.8** | +| 37048902317 | full list of this branch, before the second review | 14.8 min | 49.8 | +| **37053183525** | **full list of this branch, final head** | **13.6 min** | **47.2** | The two new boot groups add four jobs: `os-update` 3.0 to 3.5 min and `os-revert` 3.4 to 5.2 min each, about 15 runner minutes, plus the test bundle (about 0.5 min in the build job). The wall time moves by the bundle step and by `os-revert` when it is the longest job: about +2 min against #562's 12.7 min run, inside the +3 to 4 min the maintainer asked for. The build job itself varies more than that between runs (8.2 to 10.7 min). @@ -90,7 +91,9 @@ All in CI, every publish input false. Never built or booted locally. | 37035445476 | the full list, after the review fixes | red: `os-revert` under UEFI. The new grubenv check in the safety net never fired | | 37038924785, 37043066156, 37045179497 | `os-revert`, then `unseeded os-revert`, with traces | found the cause: under UEFI, GRUB leaves the booted slot's `TRY` at 0 (#575) | | 37047235688 | `unseeded os-revert` | green, after the safety net went back to trusting the marker | -| **37048902317** | **the full list, final head `fe3eaa8`** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS** | +| 37048902317 | the full list, head `fe3eaa8` before the second review | green, all 14 jobs | +| 37051462028 | `os-update os-revert`, after the second review's fixes | green, all four jobs | +| **37053183525** | **the full list, final head `b638594`** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS; 13.6 min wall, 47.2 runner minutes** | - **Tests.** `internal/hostagent/updatetarget/os_test.go` (the checks, every pick rule, the loop: install outside the window and switch inside it, stream B first, a bad OS part refused with stream B still applying, an answer with only an OS part, current, none, unsupported). `internal/hostagent/osupdate/osupdate_test.go` against a fake two-slot RAUC (a normal boot marked good at once, install then switch, a wrong digest never reaching RAUC and not retried the same night, a stale `TRY` stopping the switch and putting the booted slot back, a busy lock, a good trial, a failed trial rebooting and the old slot recording the revert and holding, the floor file, the JSON and grubenv parsers, `Peek`). The report's `os` part, the brain's pass-through, the notification, the store lookup and the brain's outcome check have tests too. `make check` green. From 02ca9f4404b56ff88616d7168b975983793581de Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 20:50:18 +0100 Subject: [PATCH 17/18] fixup! os update: review fixes (#563) Third round: the trial ends 13 min after boot at the latest, 2 min before the image timer; a new switch clears the slot's old notes; a slot that gives up its trial leaves trial-failed-, so a power cut between mark-active and the activated flag still records the revert. --- docs/progress/host-agent-os-update.md | 5 +- docs/specs/UPDATES.md | 2 +- internal/hostagent/osupdate/osupdate.go | 80 ++++++++++++++++++-- internal/hostagent/osupdate/osupdate_test.go | 78 ++++++++++++++++++- 4 files changed, 154 insertions(+), 11 deletions(-) diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 7224f605..0989f624 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -74,7 +74,7 @@ From runs on this branch (the final run is in # How it was verified): | 37019713707 | full list of #562, another run | 15.9 min | 37.8 | | 37033699687 | full list of this branch, before the review fixes | 14.9 min | 48.2 | | 37048902317 | full list of this branch, before the second review | 14.8 min | 49.8 | -| **37053183525** | **full list of this branch, final head** | **13.6 min** | **47.2** | +| **37053183525** | **full list of this branch, last head tested** | **13.6 min** | **47.2** | The two new boot groups add four jobs: `os-update` 3.0 to 3.5 min and `os-revert` 3.4 to 5.2 min each, about 15 runner minutes, plus the test bundle (about 0.5 min in the build job). The wall time moves by the bundle step and by `os-revert` when it is the longest job: about +2 min against #562's 12.7 min run, inside the +3 to 4 min the maintainer asked for. The build job itself varies more than that between runs (8.2 to 10.7 min). @@ -93,7 +93,7 @@ All in CI, every publish input false. Never built or booted locally. | 37047235688 | `unseeded os-revert` | green, after the safety net went back to trusting the marker | | 37048902317 | the full list, head `fe3eaa8` before the second review | green, all 14 jobs | | 37051462028 | `os-update os-revert`, after the second review's fixes | green, all four jobs | -| **37053183525** | **the full list, final head `b638594`** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS; 13.6 min wall, 47.2 runner minutes** | +| **37053183525** | **the full list, the last head tested, `b638594` (later commits change only docs)** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS; 13.6 min wall, 47.2 runner minutes** | - **Tests.** `internal/hostagent/updatetarget/os_test.go` (the checks, every pick rule, the loop: install outside the window and switch inside it, stream B first, a bad OS part refused with stream B still applying, an answer with only an OS part, current, none, unsupported). `internal/hostagent/osupdate/osupdate_test.go` against a fake two-slot RAUC (a normal boot marked good at once, install then switch, a wrong digest never reaching RAUC and not retried the same night, a stale `TRY` stopping the switch and putting the booted slot back, a busy lock, a good trial, a failed trial rebooting and the old slot recording the revert and holding, the floor file, the JSON and grubenv parsers, `Peek`). The report's `os` part, the brain's pass-through, the notification, the store lookup and the brain's outcome check have tests too. `make check` green. @@ -110,6 +110,7 @@ All in CI, every publish input false. Never built or booted locally. - **Greptile** found seven, all fixed: an `os` part of the wrong JSON shape failed the whole answer (now decoded apart, stream A only); a list that left a minor out let the box skip it (now refused); the trial-overwrite case above; a failed reboot after `mark-active` left the new slot first (now undone); a stale marker on a slot already marked good would make the timer revert it (now `host-agent` retries the removal for a few seconds; the script cannot use the grubenv for this, because GRUB under UEFI does not save the try flag, see # Known gaps); a missing floor allowed a downgrade (now every downgrade needs the floor); and a plain local `make test-cloud-qemu` had no test bundle (now the harness builds it from the local image). - **Second review (a fresh agent and Greptile, no Block), all fixed:** a trial whose RAUC mark failed rebooted into the same slot for ever under UEFI (#575): now a slot it cannot mark bad is rebooted once (`trial_reboots` in the record), a slot it cannot mark good is not rebooted, and the image's timer reboots a slot at most once (its `safety-net-` note). A power cut between the trial marker and `mark-active` read as a revert and marked the freshly installed slot bad: the record now has `activated`, written after `mark-active`, and a marker without it is only removed. A `bundle.raucb` or `.part` left by a kill is deleted at start and before each install. `MOOSE_OS_TRIAL_TIMEOUT` is capped at 14 min, under the 15 min timer. A marker whose removal failed is retried for about 13 min, and a later boot that finds the good outcome for it only removes the marker. Greptile's "OS switches on a later reboot" was already fixed in `db47ccc` (the switch is undone when the reboot is not accepted). +- **Third round (Greptile on `db4923e`), all fixed:** the trial's deadline counted from the trial's start while the image's timer counts from boot, so a slow start could let the timer end a healthy trial: the trial now also ends 13 min after boot (`/proc/uptime`), 2 min before the timer. An old `safety-net-` note could make the timer skip a later trial of the same slot: a new switch now clears the slot's notes. A power cut between `mark-active` and the `activated` flag lost the revert's record: a slot that gives up its trial now leaves `trial-failed-`, and the old slot counts a slot with either note as tried. - **Greptile, on the later commits, dismissed:** the `grubenv at boot` line is printed but not checked. On purpose: checking `TRY=1` would fail every UEFI boot until #575 is fixed, and #575's "Done when" adds that check. ## Known gaps & deviations diff --git a/docs/specs/UPDATES.md b/docs/specs/UPDATES.md index 342e1e8a..effdb0b5 100644 --- a/docs/specs/UPDATES.md +++ b/docs/specs/UPDATES.md @@ -65,7 +65,7 @@ The OS underneath us: kernel, libc, OpenSSL, firmware, Docker itself, and `host- - **The answer's OS part** is an optional `os` list, oldest first, each entry `{version, bundle_url, bundle_sha256}` (# 8.4 has the wire). The box refuses an entry with no 64-hex sha256, a version that is not a plain `X.Y.Z`, a list out of order, or a `bundle_url` that does not start with the expected prefix (`MOOSE_UPDATE_OS_URL_PREFIX`, default `https://github.com/onmoose/os/releases/download/`). It then picks as # The update transaction says: the next minor's entry, the target in its own or the previous minor, or a refusal. It also refuses a list whose next step is not the minor right after the box's own (a list that leaves a minor out), and a release below the running control plane's floor: the brain writes `minimumAgentVersion` to `/var/lib/moose/state/minimum-host-agent` at start. **A box that cannot read that file refuses every move to an older release.** An `os` part of the wrong JSON shape is refused like a bad entry. **Each stream is judged on its own**: a bad OS part holds back stream A only, and a bad control-plane part holds back stream B only. The running OS release is `host-agent`'s own version, which is the version of the slot it ships in. - **Install, ahead of the window** (an `os-install` job, under the same lock as `system-update`). The bundle is downloaded to `/var/lib/moose/os-update/` on the state partition and its sha256 checked against the answer before RAUC sees it. Then `rauc install`, which checks the signature and writes the other slot. `system.conf` has `activate-installed=false`, so the boot order does not move. The file is deleted afterwards. A failed attempt is not repeated the same night for the same release and digest. - **Switch, inside the window** (an `os-switch` job), and only when stream B does not hold the window (it did not just start an update, and no job runs). `rauc status mark-active other` puts the new slot first with `OK=1 TRY=0`; RAUC's GRUB backend resets the slot's try flag there, and `host-agent` checks the grubenv says so before it reboots. It writes a trial marker for the new slot (`/var/lib/moose/os-update/trial-`) and the record (`state.json`), then reboots. A reboot that is not accepted undoes the switch. **One OS switch per night**, so a box several minors behind takes one step per window. **Nothing installs or switches while the booted slot is on trial**, because the other slot is then the way back. -- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min, at most 14 so it always decides before the image's timer). Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. Only a switch that took effect counts (the record's `activated` flag is written after `mark-active` succeeds), so a power cut before it never reads as a revert. When RAUC cannot mark the slot, the trial does not loop: a slot it cannot mark bad is rebooted once, a slot it cannot mark good is not rebooted, and the image's timer reboots a slot at most once; after that the box stays up for a person. A download left by a kill or a power cut is deleted at the next start. +- **The trial.** On the new slot, `host-agent` waits for the brain's `/healthz` for up to `MOOSE_OS_TRIAL_TIMEOUT` (10 min, at most 14), and never past 13 min after boot: the image's timer counts 15 min from boot, and the 2 min between are for the final RAUC mark, so `host-agent` always decides first. Healthy: `mark-good`, the marker goes, the outcome is `good`. Not healthy: `mark-bad booted` and a reboot, and GRUB boots the old slot. **A `host-agent` that never starts** cannot do either, so the image carries `moose-os-trial.timer` (15 min after boot): if the booted slot still has its marker, it marks the slot bad, leaves a note (`safety-net-`) and reboots. On the old slot, `host-agent` finds the new slot's marker, records the outcome `reverted`, marks that slot bad, logs whether the safety net made the revert, and stays. The same release is tried again the next night. Only a switch that took effect counts: the record's `activated` flag is written after `mark-active` succeeds, and a slot that gives up its trial leaves a note (`trial-failed-` from `host-agent`, `safety-net-` from the timer). So a power cut before `mark-active` never reads as a revert, and a power cut right after it still records one. A new switch clears the old notes of its slot. When RAUC cannot mark the slot, the trial does not loop: a slot it cannot mark bad is rebooted once, a slot it cannot mark good is not rebooted, and the image's timer reboots a slot at most once; after that the box stays up for a person. A download left by a kill or a power cut is deleted at the next start. - **Report.** `GET /api/v1/system/version` carries `os_version` and `os_slot`; `GET /api/v1/system/update-target` carries an `os` object beside stream B's fields (`BRAIN_HOST_PROTOCOL.md`). The brain raises one admin notification per outcome (`NOTIFICATIONS.md` # Updates). Reporting back to the cloud (# 8.4 step 5) still waits for real box authentication. - **What it costs a box** (CI, `../progress/host-agent-os-update.md`): one bundle download per update (438.5 MB for a real release), the same space on the state partition while it installs (1.1% of a 40 GB disk), and one reboot. diff --git a/internal/hostagent/osupdate/osupdate.go b/internal/hostagent/osupdate/osupdate.go index 7f9cf792..cb15f3fc 100644 --- a/internal/hostagent/osupdate/osupdate.go +++ b/internal/hostagent/osupdate/osupdate.go @@ -63,11 +63,39 @@ const DefaultDir = "/var/lib/moose/os-update" // a live host-agent always decides first. const DefaultTrialTimeout = 10 * time.Minute -// MaxTrialTimeout is the longest trial allowed. It stays under the image's -// moose-os-trial.timer (15 min), so a live host-agent always decides first -// and the timer never reverts a slot that is slow but healthy. +// MaxTrialTimeout is the longest trial allowed. const MaxTrialTimeout = 14 * time.Minute +// TrialEndSinceBoot is when a trial must have decided, counted from boot like +// moose-os-trial.timer (OnBootSec=15min). The 2 min between them are for the +// final RAUC mark and its retries, so a live host-agent always decides before +// the timer and the timer never reverts a slot that is slow but healthy. +const TrialEndSinceBoot = 13 * time.Minute + +// minTrialWait is the shortest health wait a trial gets, even when the box +// took long to reach it. +const minTrialWait = 10 * time.Second + +// bootUptime reads the time since boot from /proc/uptime. +func bootUptime() (time.Duration, error) { + b, err := os.ReadFile("/proc/uptime") + if err != nil { + return 0, err + } + var secs float64 + if _, err := fmt.Sscanf(string(b), "%f", &secs); err != nil { + return 0, fmt.Errorf("read /proc/uptime: %w", err) + } + return time.Duration(secs * float64(time.Second)), nil +} + +// FailedNote is the file a trial writes before it gives its slot up, so the +// old slot can tell a tried slot from one whose switch never took effect. +func FailedNote(dir, slot string) string { return filepath.Join(dir, "trial-failed-"+slot) } + +// SafetyNetNote is the file moose-os-trial.timer leaves when it reboots a slot. +func SafetyNetNote(dir, slot string) string { return filepath.Join(dir, "safety-net-"+slot) } + // markerRetries and markerRetry bound removeMarker. Vars for tests. var ( markerRetries = 80 @@ -165,8 +193,11 @@ type Applier struct { Reboot func() error // HTTP downloads the bundle. Nil means a plain client. HTTP Doer - // TrialTimeout bounds the trial; zero means DefaultTrialTimeout. + // TrialTimeout bounds the trial; zero means DefaultTrialTimeout. The + // trial also ends TrialEndSinceBoot after boot, whichever comes first. TrialTimeout time.Duration + // Uptime is the time since boot; nil means /proc/uptime. + Uptime func() (time.Duration, error) // Now is the clock; nil means time.Now. Now func() time.Time @@ -394,7 +425,14 @@ func (a *Applier) Boot(ctx context.Context) { if _, err := os.Stat(TrialMarker(a.dir(), oth)); err != nil { return } - if s := r.Switch; rerr == nil && (s == nil || !s.Activated || s.To != oth) { + // Tried means the switch took effect: the record says so, or the new + // slot left a note when it gave up (its host-agent, or the image's + // timer). The notes cover a power cut between mark-active and the + // record's activated flag. + _, ferr := os.Stat(FailedNote(a.dir(), oth)) + _, serr := os.Stat(SafetyNetNote(a.dir(), oth)) + tried := ferr == nil || serr == nil + if s := r.Switch; rerr == nil && !tried && (s == nil || !s.Activated || s.To != oth) { // The marker was written but the switch never took effect (a power // cut before mark-active): nothing was tried, nothing reverted. The // slot keeps what was installed into it. @@ -425,7 +463,10 @@ func (a *Applier) Boot(ctx context.Context) { // Which path made the revert: the new slot's host-agent (its trial timed // out), or the image's timer, which leaves this note because the new // slot's host-agent never got that far. - note := filepath.Join(a.dir(), "safety-net-"+oth) + if err := os.Remove(FailedNote(a.dir(), oth)); err != nil && !errors.Is(err, os.ErrNotExist) { + slog.Error("os update: could not remove the failed-trial note", "err", err, "slot", oth) + } + note := SafetyNetNote(a.dir(), oth) if _, err := os.Stat(note); err == nil { slog.Warn("os update: the image's safety net rebooted the new slot; its host-agent never marked it", "os", out.Version, "slot", oth) if err := os.Remove(note); err != nil { @@ -447,12 +488,31 @@ func (a *Applier) trial(slot string) { if timeout > MaxTrialTimeout { timeout = MaxTrialTimeout } + // Counted from boot, as the image's timer is. + up := a.Uptime + if up == nil { + up = bootUptime + } + if since, err := up(); err == nil { + if left := TrialEndSinceBoot - since; left < timeout { + timeout = left + } + } else { + slog.Error("os update: cannot read the time since boot; the trial uses its own bound", "err", err) + } + if timeout < minTrialWait { + timeout = minTrialWait + } ctx, cancel := context.WithTimeout(context.Background(), timeout) err := a.Healthy(ctx) cancel() if err != nil { slog.Error("os update: the new slot is not healthy; marking it bad and rebooting to the old slot", "err", err, "os", a.Version, "slot", slot) + // Left for the old slot: this slot was tried and failed. + if werr := writeSynced(FailedNote(a.dir(), slot), []byte(a.Version+"\n")); werr != nil { + slog.Error("os update: could not leave the failed-trial note", "err", werr, "slot", slot) + } if merr := a.mark("bad", "booted"); merr != nil { // Without the mark this slot can boot again (#575), so reboot // once at most; after that stay up and let the image's timer @@ -719,6 +779,14 @@ func (a *Applier) doSwitch(ctx context.Context, rel updatetarget.OSRelease, nigh os.Remove(TrialMarker(a.dir(), target)) return cause } + // A new trial of this slot starts clean: the notes of an earlier trial + // would make the image's timer skip it (its once-per-trial guard) or + // make the old slot read it as tried. + for _, n := range []string{SafetyNetNote(a.dir(), target), FailedNote(a.dir(), target)} { + if err := os.Remove(n); err != nil && !errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("clear %s: %w", n, err) + } + } if err := writeSynced(TrialMarker(a.dir(), target), []byte(rel.Version+"\n")); err != nil { return err } diff --git a/internal/hostagent/osupdate/osupdate_test.go b/internal/hostagent/osupdate/osupdate_test.go index 72fde8b5..1194113e 100644 --- a/internal/hostagent/osupdate/osupdate_test.go +++ b/internal/hostagent/osupdate/osupdate_test.go @@ -134,7 +134,8 @@ func newHarness(t *testing.T, booted string) *harness { h := &harness{rauc: newFakeRAUC(booted), jobs: &syncJobs{}, night: time.Date(2026, 10, 3, 0, 0, 0, 0, time.UTC)} h.rel = updatetarget.OSRelease{Version: "0.15.1", BundleURL: srv.URL + "/b.raucb", BundleSHA256: hex.EncodeToString(sum[:])} h.a = &Applier{ - RAUC: h.rauc, Jobs: h.jobs, Version: "0.15.0", Dir: t.TempDir(), + Uptime: func() (time.Duration, error) { return time.Minute, nil }, + RAUC: h.rauc, Jobs: h.jobs, Version: "0.15.0", Dir: t.TempDir(), Healthy: func(context.Context) error { return h.healthy }, Reboot: func() error { h.reboots.Add(1); return nil }, } @@ -235,7 +236,7 @@ func (h *harness) reboot(t *testing.T, booted, version string) *Applier { h.rauc.booted = booted h.rauc.marks = nil h.rauc.mu.Unlock() - a := &Applier{RAUC: h.rauc, Jobs: h.jobs, Version: version, Dir: h.a.dir(), + a := &Applier{Uptime: func() (time.Duration, error) { return time.Minute, nil }, RAUC: h.rauc, Jobs: h.jobs, Version: version, Dir: h.a.dir(), Healthy: func(context.Context) error { return h.healthy }, Reboot: func() error { h.reboots.Add(1); return nil }} return a @@ -376,7 +377,11 @@ func TestNothingMovesDuringATrial(t *testing.T) { h.healthy = errors.New("not yet") b := h.reboot(t, "B", "0.15.1") b.TrialTimeout = time.Hour // the trial stays open for the test + before := h.reboots.Load() b.Boot(context.Background()) + // The failing trial reboots; wait for it, so it does not write into the + // test's directory after the test ends. + defer waitFor(t, func() bool { return h.reboots.Load() > before }) next := updatetarget.OSRelease{Version: "0.16.0", BundleURL: h.rel.BundleURL, BundleSHA256: h.rel.BundleSHA256} jobs := len(h.jobs.errs) if d := b.Apply(next, true, h.night); d.State != protocol.OSUpdateWaiting { @@ -599,3 +604,72 @@ func TestBootRemovesLeftovers(t *testing.T) { } } } + +// The trial ends TrialEndSinceBoot after boot, like the image's timer, not +// TrialTimeout after the trial starts (Greptile, #574). +func TestTrialDeadlineCountsFromBoot(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + b := h.reboot(t, "B", "0.15.1") + b.Uptime = func() (time.Duration, error) { return TrialEndSinceBoot - time.Second, nil } + var left time.Duration + done := make(chan struct{}) + b.Healthy = func(ctx context.Context) error { + d, _ := ctx.Deadline() + left = time.Until(d) + close(done) + return nil + } + b.Boot(context.Background()) + <-done + waitFor(t, func() bool { + _, err := os.Stat(TrialMarker(b.dir(), "B")) + return b.Last() != nil && errors.Is(err, os.ErrNotExist) + }) + if left > minTrialWait+time.Second || left < minTrialWait-time.Second { + t.Fatalf("a trial 1 s before the end got %s, want about %s", left, minTrialWait) + } +} + +// A power cut between mark-active and the activated flag: the new slot was +// tried and failed, and left its note; the old slot records the revert. +func TestRevertWithoutActivatedFlag(t *testing.T) { + h := newHarness(t, "A") + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + r, _ := h.a.load() + r.Switch.Activated = false + if err := h.a.save(r); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(FailedNote(h.a.dir(), "B"), []byte("0.15.1\n"), 0o600); err != nil { + t.Fatal(err) + } + a := h.reboot(t, "A", "0.15.0") + a.Boot(context.Background()) + if a.Last() == nil || a.Last().Outcome != protocol.OSOutcomeReverted { + t.Fatalf("outcome %+v", a.Last()) + } + if _, err := os.Stat(FailedNote(a.dir(), "B")); !errors.Is(err, os.ErrNotExist) { + t.Fatal("the failed-trial note was left") + } +} + +// A new trial of a slot starts without the notes of an earlier one, so the +// image's once-per-trial guard does not skip it. +func TestSwitchClearsOldNotes(t *testing.T) { + h := newHarness(t, "A") + for _, n := range []string{SafetyNetNote(h.a.dir(), "B"), FailedNote(h.a.dir(), "B")} { + if err := os.WriteFile(n, []byte("old"), 0o600); err != nil { + t.Fatal(err) + } + } + h.a.Apply(h.rel, false, h.night) + h.a.Apply(h.rel, true, h.night) + for _, n := range []string{SafetyNetNote(h.a.dir(), "B"), FailedNote(h.a.dir(), "B")} { + if _, err := os.Stat(n); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("%s survived the new switch", n) + } + } +} From 59be822d7fb4bffa93b2a23e17b20a3d21cf3557 Mon Sep 17 00:00:00 2001 From: Andrei Date: Fri, 2 Oct 2026 21:18:30 +0100 Subject: [PATCH 18/18] fixup! docs: numbers and runs for #563 --- docs/progress/host-agent-os-update.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/progress/host-agent-os-update.md b/docs/progress/host-agent-os-update.md index 0989f624..4061319d 100644 --- a/docs/progress/host-agent-os-update.md +++ b/docs/progress/host-agent-os-update.md @@ -74,7 +74,8 @@ From runs on this branch (the final run is in # How it was verified): | 37019713707 | full list of #562, another run | 15.9 min | 37.8 | | 37033699687 | full list of this branch, before the review fixes | 14.9 min | 48.2 | | 37048902317 | full list of this branch, before the second review | 14.8 min | 49.8 | -| **37053183525** | **full list of this branch, last head tested** | **13.6 min** | **47.2** | +| 37053183525 | full list of this branch, after the second review | 13.6 min | 47.2 | +| **37058171441** | **full list of this branch, last head tested** | **13.7 min** | **47.0** | The two new boot groups add four jobs: `os-update` 3.0 to 3.5 min and `os-revert` 3.4 to 5.2 min each, about 15 runner minutes, plus the test bundle (about 0.5 min in the build job). The wall time moves by the bundle step and by `os-revert` when it is the longest job: about +2 min against #562's 12.7 min run, inside the +3 to 4 min the maintainer asked for. The build job itself varies more than that between runs (8.2 to 10.7 min). @@ -93,7 +94,9 @@ All in CI, every publish input false. Never built or booted locally. | 37047235688 | `unseeded os-revert` | green, after the safety net went back to trusting the marker | | 37048902317 | the full list, head `fe3eaa8` before the second review | green, all 14 jobs | | 37051462028 | `os-update os-revert`, after the second review's fixes | green, all four jobs | -| **37053183525** | **the full list, the last head tested, `b638594` (later commits change only docs)** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS; 13.6 min wall, 47.2 runner minutes** | +| 37053183525 | the full list, `b638594`, after the second review | green, all 14 jobs | +| 37056704788 | `os-update os-revert`, after the third round | green, all four jobs | +| **37058171441** | **the full list, the last head tested, `02ca9f4` (later commits change only docs)** | **green, all 14 jobs: 7 boot groups under UEFI and under legacy BIOS; 13.7 min wall, 47.0 runner minutes** | - **Tests.** `internal/hostagent/updatetarget/os_test.go` (the checks, every pick rule, the loop: install outside the window and switch inside it, stream B first, a bad OS part refused with stream B still applying, an answer with only an OS part, current, none, unsupported). `internal/hostagent/osupdate/osupdate_test.go` against a fake two-slot RAUC (a normal boot marked good at once, install then switch, a wrong digest never reaching RAUC and not retried the same night, a stale `TRY` stopping the switch and putting the booted slot back, a busy lock, a good trial, a failed trial rebooting and the old slot recording the revert and holding, the floor file, the JSON and grubenv parsers, `Peek`). The report's `os` part, the brain's pass-through, the notification, the store lookup and the brain's outcome check have tests too. `make check` green.