diff --git a/.gitignore b/.gitignore index 725a634..24c0922 100644 --- a/.gitignore +++ b/.gitignore @@ -8,6 +8,7 @@ /mavcaldav /mavwaked /mavmaild +/mavupdate # Certs (private keys, don't commit) certs/ diff --git a/Makefile b/Makefile index eacc5a9..28f34fd 100644 --- a/Makefile +++ b/Makefile @@ -20,7 +20,7 @@ PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data all: build -build: build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav build-mail +build: build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav build-mail build-update build-stt: CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \ @@ -53,6 +53,12 @@ build-caldav: build-mail: $(GO) build $(GOFLAGS) -o mavmaild ./cmd/mavmaild/ +# mavupdate is an operator CLI, not a daemon: nothing runs it but a human on the +# box. It is built with the rest so a broken update path is caught by `make +# build` rather than the first time it is needed. +build-update: + $(GO) build $(GOFLAGS) -o mavupdate ./cmd/mavupdate/ + run-web: build-web ./mavweb -addr :9200 -voice 127.0.0.1:9100 diff --git a/cmd/mavupdate/main.go b/cmd/mavupdate/main.go new file mode 100644 index 0000000..4dfe1b5 --- /dev/null +++ b/cmd/mavupdate/main.go @@ -0,0 +1,204 @@ +// Command mavupdate deploys a new build of Maven to the box she runs on, with +// an automatic rollback when the new build does not come up (Vikunja #249). +// +// It is a CLI on purpose, and it is the ONLY trigger for the update path. +// +// The obvious design — an IPC method plus a button on the web UI behind the +// step-up passkey gate, the way /tools works — was considered and refused. A +// step-up gate protects against the wrong person clicking; it does not change +// the fact that anything reachable over the network becomes, in the event of a +// mavweb bug, a remote arbitrary-code path with a build system attached. An +// update needs shell access on the host, which is a strictly higher bar than +// the gate that guards the tool allowlist. That is deliberate and it is the +// reason there is no MethodApplyUpdate anywhere in internal/ipc. +// +// Consequently: mavend does not import internal/update, nothing runs on a timer, +// nothing checks a release server, and no act, intent, tool or LLM output can +// reach any of this. She cannot update herself. She can be updated, by him. +// +// mavupdate -config deploy/mavend.json list # snapshots available to roll back to +// mavupdate -config deploy/mavend.json verify # make build + make test, deploys nothing +// mavupdate -config deploy/mavend.json apply -yes # the whole thing +// mavupdate -config deploy/mavend.json rollback [id] # restore + restart (default: newest) +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "os" + "os/signal" + "syscall" + "time" + + "github.com/kami/maven/internal/config" + "github.com/kami/maven/internal/update" +) + +func main() { + cfgPath := flag.String("config", "deploy/mavend.json", "path to mavend.json (the update block is read from it)") + yes := flag.Bool("yes", false, "required by `apply` and `rollback`: yes, restart the daemon") + flag.Usage = usage + flag.Parse() + + // The stdlib flag package stops parsing at the first non-flag argument, so a + // `-yes` written after the subcommand (which is how anyone would type it, and + // how the usage text shows it) lands in Args instead of the flag. Pick it out + // by hand rather than silently treating "apply -yes" as an unconfirmed apply. + var args []string + for _, a := range flag.Args() { + if a == "-yes" || a == "--yes" { + *yes = true + continue + } + args = append(args, a) + } + if len(args) == 0 { + usage() + os.Exit(2) + } + + cfg, err := config.Load(*cfgPath) + if err != nil { + die("config: %v", err) + } + if cfg.Update == nil { + die("no `update` block in %s — the update capability is off unless configured.\nSee the package comment in internal/update for what it does and does not do.", *cfgPath) + } + + logf := func(format string, a ...any) { + fmt.Fprintf(os.Stderr, "%s %s\n", time.Now().Format("15:04:05"), fmt.Sprintf(format, a...)) + } + u, err := update.New(*cfg.Update, update.WithLogger(logf)) + if err != nil { + die("%v", err) + } + + // Ctrl-C cancels the build or the health wait. It cannot cancel a rollback + // midway into leaving the box in an unknown state, because the rollback runs + // on its own context — see cmdApply. + ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) + defer stop() + + switch args[0] { + case "list": + cmdList(u) + case "verify": + cmdVerify(ctx, u) + case "apply": + if !*yes { + die("apply restarts mavend and can roll her back. Re-run with -yes if that is what you want.") + } + cmdApply(ctx, u) + case "rollback": + if !*yes { + die("rollback restores the previous artifacts and restarts mavend. Re-run with -yes.") + } + id := "" + if len(args) > 1 { + id = args[1] + } + cmdRollback(ctx, u, id) + default: + usage() + os.Exit(2) + } +} + +func cmdList(u *update.Updater) { + snaps, err := u.Snapshots() + if err != nil { + die("snapshots: %v", err) + } + if len(snaps) == 0 { + fmt.Println("no snapshots yet — the first `apply` takes one before it builds anything") + return + } + fmt.Printf("%-18s %-12s %s\n", "SNAPSHOT", "COMMIT", "FILES") + for _, s := range snaps { + commit := s.Commit + if len(commit) > 12 { + commit = commit[:12] + } + if commit == "" { + commit = "-" + } + fmt.Printf("%-18s %-12s %d\n", s.ID, commit, len(s.Files)) + } + fmt.Printf("\nrollback to the newest with: mavupdate rollback -yes\n") +} + +func cmdVerify(ctx context.Context, u *update.Updater) { + steps, err := u.Verify(ctx) + report(steps) + if err != nil { + die("%v", err) + } + fmt.Println("verified: the tree builds and passes its own tests. Nothing was deployed — run `apply -yes` for that.") +} + +func cmdApply(ctx context.Context, u *update.Updater) { + res, err := u.Apply(ctx) + report(res.Steps) + summarize(res) + switch { + case err == nil: + fmt.Println("\nupdate committed: she answers on the new build.") + case errors.Is(err, update.ErrRollbackFailed): + die("\n%v\n\nSHE IS PROBABLY DOWN. The previous artifacts are in the snapshot dir; copy them\nover the install dir and restart by hand.", err) + case errors.Is(err, update.ErrRolledBack): + die("\n%v\n\nShe is answering again on the previous build. Nothing was lost; fix the change and retry.", err) + default: + die("\n%v", err) + } +} + +func cmdRollback(ctx context.Context, u *update.Updater, id string) { + res, err := u.Rollback(ctx, id) + report(res.Steps) + summarize(res) + if err != nil && !errors.Is(err, update.ErrRolledBack) { + die("\n%v", err) + } + fmt.Printf("\nrolled back to %s; she answers on it.\n", res.SnapshotID) +} + +func report(steps []update.Step) { + for _, s := range steps { + status := "ok" + if s.Err != nil { + status = "FAILED: " + s.Err.Error() + } + fmt.Printf(" %-8s %-8s %s\n", s.Name, s.Took.Round(time.Second), status) + if s.Output != "" { + fmt.Printf("---- %s output ----\n%s\n-------------------\n", s.Name, s.Output) + } + } +} + +func summarize(res update.Result) { + fmt.Printf("\nverified=%v snapshot=%s installed=%d restarted=%v healthy=%v rolled_back=%v rollback_healthy=%v took=%s\n", + res.Verified, res.SnapshotID, len(res.Installed), res.Restarted, res.Healthy, res.RolledBack, res.RollbackHealthy, res.Took.Round(time.Second)) +} + +func usage() { + fmt.Fprint(os.Stderr, `mavupdate — deploy a new build of Maven, with rollback. + + mavupdate [-config path] list + mavupdate [-config path] verify + mavupdate [-config path] apply -yes + mavupdate [-config path] rollback [snapshot-id] -yes + +apply is: health-check the running daemon, snapshot the deployed artifacts, +make build, make test, install, restart, health-check — and restore the +snapshot if any of that fails. It never fetches code and never runs by itself. + +`) + flag.PrintDefaults() +} + +func die(format string, a ...any) { + fmt.Fprintf(os.Stderr, format+"\n", a...) + os.Exit(1) +} diff --git a/deploy/README.md b/deploy/README.md index 2f68637..790cc55 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -114,3 +114,49 @@ build on the target host, most likely in one of these: work fine over the core socket. - **netdata** — `mavpoll` reaches it via `host.docker.internal`; adjust if netdata runs elsewhere. + +## Updating her (`mavupdate`, Vikunja #249) + +Off unless configured, and there is deliberately no button for it. There is no +IPC method, no web route, no timer and no act that starts an update — the trigger +is a human running `mavupdate` on the host, which needs shell access, a strictly +higher bar than the step-up passkey gate that guards `/tools`. She cannot update +herself; she can be updated. Nothing here ever fetches code: the new version is +whatever you pulled into the working tree yourself. + +Add an `update` block to `mavend.json` (mavend ignores it — only the CLI reads +it), with paths as they exist **on the host**, not inside a container: + +```json +"update": { + "source_dir": "/home/kami/apps/Maven", + "install_dir": "/home/kami/apps/Maven", + "snapshot_dir": "/var/lib/maven-snapshots", + "binaries": ["mavend", "mavweb", "mavsttd", "mavttsd", "mavwaked", + "mavenclient", "mavpoll", "mavcaldav", "mavmaild"], + "config_files": ["deploy/mavend.json"], + "restart_cmd": ["docker", "compose", "up", "-d", "--build"], + "health_socket": "/var/lib/docker/volumes/maven_sockets/_data/mavend.sock", + "health_timeout_sec": 120 +} +``` + +`snapshot_dir` must be outside `install_dir` (a restore must not read from what +the install writes) and `health_socket` is required: an update that cannot check +its own result cannot roll itself back, so the config is refused without one. + +Then: + +```sh +mavupdate -config deploy/mavend.json verify # make build + make test, deploys nothing +mavupdate -config deploy/mavend.json apply -yes # snapshot, verify, install, restart, health-check +mavupdate -config deploy/mavend.json list # what you can roll back to +mavupdate -config deploy/mavend.json rollback -yes # restore the previous artifacts and restart +``` + +`apply` refuses to start if she is not already answering — otherwise a failed +update and a box that was already broken are indistinguishable afterwards. On any +failure after the install it restores the snapshot, restarts, and checks again; +if that also fails it says so loudly and names the directory to copy back by hand. +The database is never snapshotted or rolled back (see the package comment in +`internal/update`); schema compatibility is `store.Migrate`'s job. diff --git a/internal/config/config.go b/internal/config/config.go index 579dfcc..5d9a0dd 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -23,6 +23,7 @@ import ( "github.com/kami/maven/internal/delivery/ntfysink" "github.com/kami/maven/internal/delivery/telegramsink" "github.com/kami/maven/internal/morning" + "github.com/kami/maven/internal/update" "github.com/robfig/cron/v3" ) @@ -108,6 +109,16 @@ type Config struct { // calls its /v1/chat/completions endpoint to phrase nudges and reminders. Phraser *PhraserConfig `json:"phraser,omitempty"` + // Update — how THIS box deploys a new build of Maven (Vikunja #249). nil ⇒ + // the update capability does not exist, which is the state to leave it in + // unless the operator has read internal/update's package comment. + // + // mavend never reads this block: the daemon does not import internal/update + // and cannot update itself. It lives here because cmd/mavupdate — a CLI the + // owner runs on the host, the only trigger there is — reads the same config + // file to find the socket it health-checks. + Update *update.Config `json:"update,omitempty"` + // Voice — the client↔core surface + the stt/tts modules the daemon // wires. nil ⇒ the daemon doesn't wire voice: the TCP listener stays // down, the dispatcher's Voice slot stays nil (the routing table's @@ -844,6 +855,14 @@ func (c *Config) validate() error { } } } + // The update block is validated here even though mavend never acts on it: a + // half-written update config that is only noticed by cmd/mavupdate is noticed + // at the worst possible moment, halfway through deploying a new build. + if c.Update != nil { + if err := c.Update.Validate(); err != nil { + return err + } + } if c.Voice != nil && c.Voice.Enabled { if c.Voice.Bind == "" { return errors.New("voice.enabled set but voice.bind is empty — refusing to start a voice surface with no bind address") diff --git a/internal/config/config_test.go b/internal/config/config_test.go index 4786e58..501aa43 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -325,3 +325,44 @@ func TestSwapModelsParsedAndMustBeAbsolute(t *testing.T) { t.Error("Load accepted a relative swap_models entry; want a startup failure") } } + +// TestUpdateBlockAbsentMeansOff — mavend never updates itself; the block only +// exists so cmd/mavupdate can find the deployment it is asked to update +// (Vikunja #249). Absent is the normal state. +func TestUpdateBlockAbsentMeansOff(t *testing.T) { + c, err := Load(writeConfig(t, `{}`)) + if err != nil { + t.Fatalf("Load: %v", err) + } + if c.Update != nil { + t.Errorf("update = %+v; want nil when unconfigured", c.Update) + } +} + +func TestUpdateBlockValidatedAtStartup(t *testing.T) { + good := `{"update": { + "source_dir": "/srv/maven", + "install_dir": "/srv/maven", + "snapshot_dir": "/var/lib/maven/snapshots", + "binaries": ["mavend", "mavweb"], + "restart_cmd": ["docker", "compose", "up", "-d", "--build", "mavend"], + "health_socket": "/run/maven/mavend.sock" + }}` + c, err := Load(writeConfig(t, good)) + if err != nil { + t.Fatalf("Load: %v", err) + } + if c.Update == nil || len(c.Update.Binaries) != 2 { + t.Fatalf("update block = %+v; want it parsed", c.Update) + } + // A block with no health check cannot detect its own failure, so it cannot + // roll back — refused at load, not halfway through a deploy. + noHealth := `{"update": { + "source_dir": "/srv/maven", "install_dir": "/srv/maven", + "snapshot_dir": "/var/lib/maven/snapshots", + "binaries": ["mavend"], "restart_cmd": ["true"] + }}` + if _, err := Load(writeConfig(t, noHealth)); err == nil { + t.Error("Load accepted an update block with no health_socket") + } +} diff --git a/internal/update/apply.go b/internal/update/apply.go new file mode 100644 index 0000000..d11f22a --- /dev/null +++ b/internal/update/apply.go @@ -0,0 +1,208 @@ +package update + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "time" +) + +// Result — the full account of one Apply. Every field is filled in on the +// failure paths too, because "what state is my box in" is the only question that +// matters after a failed update. +type Result struct { + Verified bool + SnapshotID string // the rollback target; named even when the rollback failed + Installed []string + Restarted bool + Healthy bool + RolledBack bool + // RollbackHealthy — whether she answered again after the restore. False with + // RolledBack true is the manual-recovery case. + RollbackHealthy bool + Steps []Step + Took time.Duration +} + +// Apply is the whole update, in the only order that is safe. +// +// It is called by a human running cmd/mavupdate on the box. Nothing else calls +// it: no timer, no IPC method, no web route, no act. See the package comment. +func (u *Updater) Apply(ctx context.Context) (Result, error) { + start := u.now() + res := Result{} + defer func() { res.Took = u.now().Sub(start) }() + + // 0. She has to be answering before we start. Otherwise a failed update and + // a box that was already broken look identical afterwards, and the rollback + // has no baseline to prove itself against. + u.log("preflight: checking the running daemon") + if err := u.health(ctx, u.cfg.HealthSocket); err != nil { + return res, fmt.Errorf("%w: %v", ErrUnhealthyBefore, err) + } + + // 1. Snapshot what is deployed now, BEFORE the build. + // + // The order matters and it is not the obvious one. `make build` writes its + // binaries into the working tree, and on the docker deployment the working + // tree IS the install dir — so snapshotting after the build would snapshot + // the new artifacts and leave nothing to roll back to. The snapshot is the + // only thing standing between a bad build and a box that needs a screwdriver, + // so it is taken first, while the deployed bytes are still the old ones. + names := append(append([]string{}, u.cfg.Binaries...), u.cfg.ConfigFiles...) + snap, err := u.store.Save(u.cfg.InstallDir, names, u.gitHead(ctx), "pre-update") + if err != nil { + return res, err + } + res.SnapshotID = snap.ID + u.log("snapshot: %s (%d files) in %s", snap.ID, len(snap.Files), snap.Dir()) + + // 2. Build and test before anything is deployed. A broken tree costs time + // and nothing else — but `make build` has already overwritten the binaries in + // the tree, so restore them: otherwise a later restart by hand would deploy + // code that failed its own tests. Nothing has been restarted, so this is a + // file restore with no restart and no health check. + steps, err := u.Verify(ctx) + res.Steps = append(res.Steps, steps...) + if err != nil { + if rerr := snap.Restore(u.cfg.InstallDir); rerr != nil { + u.log("verify failed and the artifacts could not be put back: %v — the previous ones are in %s", rerr, snap.Dir()) + } else { + res.RolledBack = true + u.log("verify failed; the previously deployed artifacts are back in place, she was never restarted") + } + return res, err + } + res.Verified = true + + // 3. Install. Per-file temp+rename, so an interruption leaves whole files. + // Config is snapshotted but never overwritten — an update does not get to + // replace the operator's config. + installed, err := u.install() + res.Installed = installed + if err != nil { + // Files may be half-swapped across the set, so restore before returning + // even though nothing has been restarted yet. + u.log("install failed: %v — restoring", err) + return u.rollback(ctx, snap, res, err) + } + u.log("install: %d artifact(s) into %s", len(installed), u.cfg.InstallDir) + + // 4. Restart, then 5. prove she answers. + if err := u.restart(ctx, &res); err != nil { + return u.rollback(ctx, snap, res, err) + } + u.log("restart: ok, waiting for her to answer (up to %s)", u.cfg.healthTimeout()) + if err := u.waitHealthy(ctx, u.cfg.healthTimeout()); err != nil { + return u.rollback(ctx, snap, res, err) + } + res.Healthy = true + u.log("health: she answers on %s — update committed", u.cfg.HealthSocket) + + if err := u.store.Prune(u.cfg.KeepSnapshots); err != nil { + u.log("prune: %v (harmless)", err) + } + return res, nil +} + +// Rollback restores a snapshot by id (empty = the newest) and restarts. Exposed +// separately so the operator can undo an update that verified, restarted and +// answered a Presence call but is wrong in a way no health check can see. +func (u *Updater) Rollback(ctx context.Context, id string) (Result, error) { + var snap Snapshot + var err error + if id == "" { + snaps, lerr := u.store.List() + if lerr != nil { + return Result{}, lerr + } + if len(snaps) == 0 { + return Result{}, errors.New("update: no snapshots to roll back to") + } + snap = snaps[0] + } else if snap, err = u.store.Load(id); err != nil { + return Result{}, err + } + res := Result{SnapshotID: snap.ID} + return u.rollback(ctx, snap, res, errors.New("operator asked for a rollback")) +} + +// rollback restores the snapshot and restarts, then reports whether that worked. +// It depends on nothing that the update changed: file copies out of the snapshot +// dir and the same restart command. No build, no migration, no cooperation from +// the code being replaced. +func (u *Updater) rollback(ctx context.Context, snap Snapshot, res Result, cause error) (Result, error) { + // A rollback interrupted halfway is the one outcome worse than the failure + // that triggered it, so it does not inherit the caller's cancellation: a + // Ctrl-C during the health wait must not abandon the restore mid-restart. + ctx = context.WithoutCancel(ctx) + res.RolledBack = true + u.log("rollback: restoring snapshot %s over %s", snap.ID, u.cfg.InstallDir) + if err := snap.Restore(u.cfg.InstallDir); err != nil { + u.log("rollback: RESTORE FAILED: %v", err) + return res, fmt.Errorf("%w: %v (after %v); the previous artifacts are in %s — copy them back by hand", ErrRollbackFailed, err, cause, snap.Dir()) + } + // A restore with no restart leaves the failed process running, so a failed + // restart here is still the manual-recovery case. + if err := u.restart(ctx, &res); err != nil { + u.log("rollback: RESTART FAILED: %v", err) + return res, fmt.Errorf("%w: restored %s but the restart failed: %v (after %v)", ErrRollbackFailed, snap.ID, err, cause) + } + if err := u.waitHealthy(ctx, u.cfg.healthTimeout()); err != nil { + u.log("rollback: she still does not answer: %v", err) + return res, fmt.Errorf("%w: restored %s and restarted but she does not answer: %v (after %v)", ErrRollbackFailed, snap.ID, err, cause) + } + res.RollbackHealthy = true + u.log("rollback: she answers again on the previous build (%s)", snap.ID) + return res, fmt.Errorf("%w to %s: %v", ErrRolledBack, snap.ID, cause) +} + +// install copies the freshly built binaries from SourceDir into InstallDir. +// +// When the two are the same directory — the docker deployment builds the image +// from the working tree — this is a no-op by design rather than by accident: the +// artifacts are already where they belong and the restart command rebuilds the +// image from them. +func (u *Updater) install() ([]string, error) { + if filepath.Clean(u.cfg.SourceDir) == filepath.Clean(u.cfg.InstallDir) { + return u.cfg.Binaries, nil + } + var done []string + for _, name := range u.cfg.Binaries { + src := filepath.Join(u.cfg.SourceDir, name) + fi, err := os.Stat(src) + if err != nil { + return done, fmt.Errorf("update: install %s: %w (did `make build` produce it?)", name, err) + } + if _, err := copyFile(src, filepath.Join(u.cfg.InstallDir, name), fi.Mode().Perm()); err != nil { + return done, fmt.Errorf("update: install %s: %w", name, err) + } + done = append(done, name) + } + return done, nil +} + +func (u *Updater) restart(ctx context.Context, res *Result) error { + u.log("restart: %v", u.cfg.RestartCmd) + out, err := u.run(ctx, u.cfg.SourceDir, u.cfg.RestartCmd) + if err != nil { + res.Steps = append(res.Steps, Step{Name: "restart", Argv: u.cfg.RestartCmd, Err: err, Output: tail(out, 4000)}) + return fmt.Errorf("update: restart %v: %w", u.cfg.RestartCmd, err) + } + res.Restarted = true + res.Steps = append(res.Steps, Step{Name: "restart", Argv: u.cfg.RestartCmd}) + return nil +} + +// gitHead records which commit produced a snapshot, for the operator's benefit. +// Best-effort: a tree without git is not a reason to refuse to snapshot. +func (u *Updater) gitHead(ctx context.Context) string { + out, err := u.run(ctx, u.cfg.SourceDir, []string{"git", "rev-parse", "HEAD"}) + if err != nil { + return "" + } + return strings.TrimSpace(out) +} diff --git a/internal/update/health.go b/internal/update/health.go new file mode 100644 index 0000000..06b21bc --- /dev/null +++ b/internal/update/health.go @@ -0,0 +1,61 @@ +package update + +import ( + "context" + "fmt" + "time" + + "github.com/kami/maven/internal/ipc" +) + +// The health check is the whole basis for rolling back, so it has to mean +// something. "The process is running" does not: mavend can be up with a dead +// store, a socket it never bound, or a config it failed to parse. What is +// checked instead is that she answers a real read over the real IPC socket — +// which exercises the socket, the dispatch table and the store in one call. +// +// Presence is the method used because it is read-only (safe to retry), needs no +// arguments, and touches the store. It cannot write anything, so a health check +// never leaves a trace in her memory. + +// DialHealth connects to the mavend socket and performs one read. +func DialHealth(ctx context.Context, socket string) error { + c, err := ipc.Dial(socket) + if err != nil { + return fmt.Errorf("update: health dial: %w", err) + } + defer c.Close() + if _, err := c.Presence(ctx); err != nil { + return fmt.Errorf("update: health read: %w", err) + } + return nil +} + +// waitHealthy retries the health check until it passes or the timeout elapses. +// A restart is not instantaneous — she loads a 1.7B on boot — so the first few +// failures are expected and are not a reason to roll back. +func (u *Updater) waitHealthy(ctx context.Context, timeout time.Duration) error { + deadline := u.now().Add(timeout) + delay := 500 * time.Millisecond + var last error + for { + attemptCtx, cancel := context.WithTimeout(ctx, 10*time.Second) + err := u.health(attemptCtx, u.cfg.HealthSocket) + cancel() + if err == nil { + return nil + } + last = err + if u.now().After(deadline) { + return fmt.Errorf("update: not healthy after %s: %w", timeout, last) + } + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(delay): + } + if delay < 5*time.Second { + delay *= 2 + } + } +} diff --git a/internal/update/snapshot.go b/internal/update/snapshot.go new file mode 100644 index 0000000..42af6c6 --- /dev/null +++ b/internal/update/snapshot.go @@ -0,0 +1,267 @@ +package update + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "path/filepath" + "sort" + "time" +) + +// A snapshot is a byte-for-byte copy of the deployed artifacts plus a manifest +// of their sha256 sums, taken before an install. +// +// It is copies, not hardlinks and not a git stash, for one reason: the restore +// path must work when everything else is broken. A hardlink into the install dir +// would be clobbered by the very install it exists to undo, and a git-based +// undo needs a toolchain, a clean tree, and a rebuild — three things a failed +// update is likely to have taken away. Copying two dozen megabytes of Go +// binaries costs a second and needs nothing but the filesystem. +// +// The sums are what make a restore verifiable rather than hopeful: Restore +// re-hashes every file it writes, so "the old bytes are back" is checked, not +// assumed. + +// FileRec — one file in a snapshot. +type FileRec struct { + Name string `json:"name"` // relative name inside the install dir + SHA256 string `json:"sha256"` // of the snapshotted bytes + Mode os.FileMode `json:"mode"` + Size int64 `json:"size"` +} + +// Snapshot — the manifest. Written last, so a directory without a readable +// manifest.json is an aborted snapshot and is never offered as a rollback target. +type Snapshot struct { + ID string `json:"id"` // sortable timestamp, also the directory name + CreatedAt time.Time `json:"created_at"` + Commit string `json:"commit,omitempty"` // git HEAD of the tree that produced it, when known + Note string `json:"note,omitempty"` + Files []FileRec `json:"files"` + + dir string // absolute path, filled in by List/Load +} + +// Dir — where this snapshot's file copies live. +func (s Snapshot) Dir() string { return s.dir } + +const manifestName = "manifest.json" + +// Store is a directory of snapshots. +type Store struct { + Dir string + now func() time.Time +} + +func (st *Store) clock() time.Time { + if st.now != nil { + return st.now() + } + return time.Now() +} + +// Save copies names (relative to srcDir) into a new snapshot and writes the +// manifest. A name that does not exist is skipped rather than fatal: the first +// ever run happens on a box where some artifact may legitimately be missing, and +// refusing to snapshot then would mean refusing to update. +func (st *Store) Save(srcDir string, names []string, commit, note string) (Snapshot, error) { + ts := st.clock().UTC() + snap := Snapshot{ + ID: ts.Format("20060102-150405"), + CreatedAt: ts, + Commit: commit, + Note: note, + } + snap.dir = filepath.Join(st.Dir, snap.ID) + if err := os.MkdirAll(snap.dir, 0o700); err != nil { + return Snapshot{}, fmt.Errorf("update: snapshot dir: %w", err) + } + for _, name := range names { + src := filepath.Join(srcDir, name) + fi, err := os.Stat(src) + if err != nil { + if errors.Is(err, os.ErrNotExist) { + continue + } + return Snapshot{}, fmt.Errorf("update: snapshot %s: %w", name, err) + } + if fi.IsDir() { + return Snapshot{}, fmt.Errorf("update: snapshot %s: is a directory (only files are deployable artifacts)", name) + } + dst := filepath.Join(snap.dir, name) + if err := os.MkdirAll(filepath.Dir(dst), 0o700); err != nil { + return Snapshot{}, err + } + sum, err := copyFile(src, dst, fi.Mode().Perm()) + if err != nil { + return Snapshot{}, fmt.Errorf("update: snapshot %s: %w", name, err) + } + snap.Files = append(snap.Files, FileRec{Name: name, SHA256: sum, Mode: fi.Mode().Perm(), Size: fi.Size()}) + } + if len(snap.Files) == 0 { + os.RemoveAll(snap.dir) + return Snapshot{}, fmt.Errorf("update: snapshot of %s is empty — none of the listed artifacts exist", srcDir) + } + // Manifest last: its presence is what makes the snapshot usable. + blob, err := json.MarshalIndent(snap, "", " ") + if err != nil { + return Snapshot{}, err + } + if err := os.WriteFile(filepath.Join(snap.dir, manifestName), blob, 0o600); err != nil { + return Snapshot{}, fmt.Errorf("update: snapshot manifest: %w", err) + } + return snap, nil +} + +// List returns the complete snapshots, newest first. +func (st *Store) List() ([]Snapshot, error) { + ents, err := os.ReadDir(st.Dir) + if err != nil { + if errors.Is(err, os.ErrNotExist) { + return nil, nil + } + return nil, err + } + var out []Snapshot + for _, e := range ents { + if !e.IsDir() { + continue + } + s, err := st.Load(e.Name()) + if err != nil { + continue // aborted or hand-mangled: not a rollback target + } + out = append(out, s) + } + sort.Slice(out, func(i, j int) bool { return out[i].ID > out[j].ID }) + return out, nil +} + +// Load reads one snapshot's manifest. +func (st *Store) Load(id string) (Snapshot, error) { + dir := filepath.Join(st.Dir, id) + blob, err := os.ReadFile(filepath.Join(dir, manifestName)) + if err != nil { + return Snapshot{}, err + } + var s Snapshot + if err := json.Unmarshal(blob, &s); err != nil { + return Snapshot{}, fmt.Errorf("update: manifest %s: %w", id, err) + } + s.dir = dir + return s, nil +} + +// Restore copies a snapshot's files back over dstDir and verifies every write +// against the manifest sum. Only the named files are touched; anything else in +// dstDir is left alone. +// +// This is the function the whole package exists to be able to run. It uses the +// filesystem and nothing else — no toolchain, no build, no cooperation from the +// code being replaced. +func (s Snapshot) Restore(dstDir string) error { + if s.dir == "" { + return errors.New("update: snapshot has no directory (load it through the store)") + } + for _, f := range s.Files { + src := filepath.Join(s.dir, f.Name) + sum, err := hashFile(src) + if err != nil { + return fmt.Errorf("update: restore %s: %w", f.Name, err) + } + if sum != f.SHA256 { + return fmt.Errorf("update: restore %s: snapshot is corrupt (sha256 %s, manifest says %s)", f.Name, sum, f.SHA256) + } + dst := filepath.Join(dstDir, f.Name) + if err := os.MkdirAll(filepath.Dir(dst), 0o755); err != nil { + return err + } + got, err := copyFile(src, dst, f.Mode) + if err != nil { + return fmt.Errorf("update: restore %s: %w", f.Name, err) + } + if got != f.SHA256 { + return fmt.Errorf("update: restore %s: wrote the wrong bytes (sha256 %s)", f.Name, got) + } + } + return nil +} + +// Prune keeps the newest keep snapshots and removes the rest. The newest is +// never pruned regardless of keep — it is the rollback target. +func (st *Store) Prune(keep int) error { + if keep < 1 { + keep = 1 + } + snaps, err := st.List() + if err != nil { + return err + } + for _, s := range snaps[min(keep, len(snaps)):] { + if err := os.RemoveAll(s.dir); err != nil { + return err + } + } + return nil +} + +// copyFile writes src to dst atomically (temp + rename, so a reader never sees a +// half file and an interrupted copy leaves the old one intact) and returns the +// sha256 of what was written. +func copyFile(src, dst string, mode os.FileMode) (string, error) { + in, err := os.Open(src) + if err != nil { + return "", err + } + defer in.Close() + if mode == 0 { + mode = 0o644 + } + tmp, err := os.CreateTemp(filepath.Dir(dst), ".update-*") + if err != nil { + return "", err + } + tmpName := tmp.Name() + defer os.Remove(tmpName) // no-op once the rename succeeds + h := sha256.New() + if _, err := io.Copy(io.MultiWriter(tmp, h), in); err != nil { + tmp.Close() + return "", err + } + // fsync before the rename: a binary that is renamed into place but whose + // bytes are still in the page cache is exactly the file a power cut turns + // into an unbootable daemon. + if err := tmp.Sync(); err != nil { + tmp.Close() + return "", err + } + if err := tmp.Chmod(mode); err != nil { + tmp.Close() + return "", err + } + if err := tmp.Close(); err != nil { + return "", err + } + if err := os.Rename(tmpName, dst); err != nil { + return "", err + } + return hex.EncodeToString(h.Sum(nil)), nil +} + +func hashFile(p string) (string, error) { + f, err := os.Open(p) + if err != nil { + return "", err + } + defer f.Close() + h := sha256.New() + if _, err := io.Copy(h, f); err != nil { + return "", err + } + return hex.EncodeToString(h.Sum(nil)), nil +} diff --git a/internal/update/update.go b/internal/update/update.go new file mode 100644 index 0000000..e030335 --- /dev/null +++ b/internal/update/update.go @@ -0,0 +1,270 @@ +// Package update applies a new build of Maven to the box she runs on, with a +// verified-before-committed install and an automatic rollback (Vikunja #249). +// +// # What this package refuses to be +// +// This is the highest-risk capability in the backlog — code that changes the +// running system — so the refusals are as much of the design as the features, +// and they are enforced here rather than described in a doc: +// +// - It is never automatic and never on a timer. There is no checker, no +// channel, no "check for updates" call and nothing that fires from the tick +// loop. Apply runs exactly when a human runs cmd/mavupdate on the box. +// - The daemon cannot update itself. mavend does not import this package and +// there is no IPC method and no web route that reaches it, so no act, no +// intent, no tool and no LLM output can start an update. The trigger needs +// shell access to the host, which is a strictly higher bar than the step-up +// passkey gate that guards /tools — an update is not a thing to expose to +// anything reachable over the network. +// - It does not fetch code. Nothing here talks to a release server, a +// registry, or GitHub. The new version is whatever is in the working tree +// the operator points it at, which he pulled himself. Downloading and +// running code on the strength of a checksum in the same download is not a +// property we can verify on one box. +// - It does not supervise its own death. The plan asked for an in-process +// crash-loop detector; a process cannot reliably notice that it keeps +// dying, and one that thinks it can is worse than nothing. Restart-on-crash +// belongs to whatever starts mavend (compose `restart: unless-stopped`, +// systemd `Restart=`). What this package guarantees instead is narrower and +// real: within one Apply, the new build is proven to answer before the old +// one is considered replaced, and if it does not answer the old bytes go +// back and are proven to answer again. +// +// # The order of operations, and why +// +// Apply is: health-check the CURRENT daemon → build → test → snapshot → install +// → restart → health-check → rollback on any failure. +// +// The first health check is not ceremony. If she is already not answering, a +// failed update and a broken box are indistinguishable afterwards, and the +// rollback has nothing to prove itself against — so Apply refuses to start. +// +// Build and test run BEFORE anything is written to the install dir, so a broken +// tree costs nothing but time. Install is per-file write-temp-then-rename, so a +// crash mid-install leaves whole files, not half ones. +// +// The rollback path deliberately depends on nothing that just changed: it copies +// byte-for-byte from a snapshot taken before the install and re-runs the same +// restart command. It does not ask the new binary to do anything, does not run +// a migration, and does not need the update to have gotten far enough to leave +// a working anything behind. +// +// # What is out of scope on purpose +// +// The database is not snapshotted or rolled back. It is encrypted, live, and +// often larger than the disk headroom; a store rolled back under a schema that +// already migrated forward loses writes silently, which is worse than a failed +// update. Schema compatibility is store.Migrate's job. A snapshot here is the +// deployable artifacts only: binaries and config. +package update + +import ( + "context" + "errors" + "fmt" + "os/exec" + "path/filepath" + "strings" + "time" +) + +var ( + // ErrNotConfigured — no update block in the config. The capability does not + // exist unless the operator described his own deployment. + ErrNotConfigured = errors.New("update: not configured") + + // ErrUnhealthyBefore — the daemon was already not answering when Apply + // started. Refused: see the package comment. + ErrUnhealthyBefore = errors.New("update: the running daemon is not healthy — refusing to update on top of a broken box") + + // ErrVerifyFailed — build or test failed. Nothing was installed. + ErrVerifyFailed = errors.New("update: verification failed") + + // ErrRolledBack — the new build was installed and did not come up healthy, + // so the previous snapshot was restored. Wraps the underlying failure. + ErrRolledBack = errors.New("update: rolled back") + + // ErrRollbackFailed — the worst case: the new build failed AND the restore + // did not bring her back. The operator has to fix the box by hand; the + // snapshot directory is named in the result so he knows what to copy. + ErrRollbackFailed = errors.New("update: ROLLBACK FAILED — manual recovery required") +) + +// Config — the operator's description of his own deployment. Every path is +// absolute and validated; nothing is guessed, because guessing wrong here means +// overwriting the wrong file. +type Config struct { + // SourceDir — the git working tree to build. The operator pulls it himself; + // this package never fetches. + SourceDir string `json:"source_dir"` + + // InstallDir — where the built binaries are copied to. On the docker + // deployment this is the tree the image is built from, so it is usually the + // same as SourceDir and Install is a no-op copy; on a bare-metal deployment + // it is /opt/maven/bin. + InstallDir string `json:"install_dir"` + + // SnapshotDir — where the pre-install copies live. Must not be inside + // InstallDir: a restore reading from a directory the install is writing to + // is not a restore. + SnapshotDir string `json:"snapshot_dir"` + + // Binaries — the artifact names to snapshot and install, relative to + // SourceDir (built) and InstallDir (deployed). Listed explicitly rather than + // globbed so a stray file in the tree never gets deployed. + Binaries []string `json:"binaries"` + + // ConfigFiles — extra files to snapshot alongside the binaries, relative to + // InstallDir. Snapshotted, never overwritten by an install: the operator's + // config is not something an update gets to replace. + ConfigFiles []string `json:"config_files,omitempty"` + + // RestartCmd — how this deployment restarts mavend, e.g. + // ["docker","compose","up","-d","--build","mavend"] or + // ["systemctl","restart","mavend"]. Run in SourceDir. Required: there is no + // portable default and picking one would mean restarting the wrong thing. + RestartCmd []string `json:"restart_cmd"` + + // HealthSocket — mavend's IPC socket, used to prove she answers after a + // restart. Required: without a health check there is no signal to roll back + // on, and an update that cannot detect its own failure is not what this + // package is for. + HealthSocket string `json:"health_socket"` + + // HealthTimeoutSec — how long to wait for the restarted daemon to answer. + // Default 90s; she loads a 1.7B on boot, so this is not a couple of seconds. + HealthTimeoutSec int `json:"health_timeout_sec,omitempty"` + + // VerifyTimeoutMin — cap on `make build` + `make test`. Default 20m. + VerifyTimeoutMin int `json:"verify_timeout_min,omitempty"` + + // KeepSnapshots — how many snapshots to retain. Default 5, minimum 1: the + // most recent one is the rollback target and is never pruned. + KeepSnapshots int `json:"keep_snapshots,omitempty"` +} + +// Validate — fail at startup, not halfway through an install. +func (c Config) Validate() error { + if c.SourceDir == "" || c.InstallDir == "" || c.SnapshotDir == "" { + return errors.New("update: source_dir, install_dir and snapshot_dir are all required") + } + for _, p := range []string{c.SourceDir, c.InstallDir, c.SnapshotDir} { + if !filepath.IsAbs(p) { + return fmt.Errorf("update: %q must be an absolute path", p) + } + } + if within(c.SnapshotDir, c.InstallDir) { + return fmt.Errorf("update: snapshot_dir %q is inside install_dir %q — a restore must not read from what the install writes", c.SnapshotDir, c.InstallDir) + } + if len(c.Binaries) == 0 { + return errors.New("update: binaries is empty — nothing to install") + } + for _, b := range append(append([]string{}, c.Binaries...), c.ConfigFiles...) { + if filepath.IsAbs(b) || strings.Contains(b, "..") { + return fmt.Errorf("update: %q must be a plain relative name", b) + } + } + if len(c.RestartCmd) == 0 { + return errors.New("update: restart_cmd is required — there is no safe default for restarting someone else's deployment") + } + if c.HealthSocket == "" { + return errors.New("update: health_socket is required — an update that cannot check its own result cannot roll back on failure") + } + return nil +} + +func (c Config) withDefaults() Config { + if c.HealthTimeoutSec <= 0 { + c.HealthTimeoutSec = 90 + } + if c.VerifyTimeoutMin <= 0 { + c.VerifyTimeoutMin = 20 + } + if c.KeepSnapshots < 1 { + c.KeepSnapshots = 5 + } + return c +} + +func (c Config) healthTimeout() time.Duration { + return time.Duration(c.HealthTimeoutSec) * time.Second +} + +func (c Config) verifyTimeout() time.Duration { + return time.Duration(c.VerifyTimeoutMin) * time.Minute +} + +// within reports whether p is dir or lives under it. +func within(p, dir string) bool { + p, dir = filepath.Clean(p), filepath.Clean(dir) + if p == dir { + return true + } + rel, err := filepath.Rel(dir, p) + return err == nil && rel != ".." && !strings.HasPrefix(rel, ".."+string(filepath.Separator)) +} + +// Runner runs one command and returns its combined output. Injected so the +// tests can drive build/test/restart failures without a toolchain, a container +// or a real daemon to break. +type Runner func(ctx context.Context, dir string, argv []string) (string, error) + +// ExecRunner is the real one. +func ExecRunner(ctx context.Context, dir string, argv []string) (string, error) { + cmd := exec.CommandContext(ctx, argv[0], argv[1:]...) + cmd.Dir = dir + out, err := cmd.CombinedOutput() + return string(out), err +} + +// HealthCheck proves the daemon at socket answers. Injected for the same reason +// as Runner. +type HealthCheck func(ctx context.Context, socket string) error + +// Logger receives one line per step. The CLI prints these as they happen: an +// update that goes quiet for four minutes during `make test` reads as a hang. +type Logger func(format string, args ...any) + +// Updater is the whole capability. Construct with New and call Apply or +// Rollback; there is no background goroutine and nothing starts on its own. +type Updater struct { + cfg Config + store *Store + run Runner + health HealthCheck + log Logger + now func() time.Time +} + +// New builds an Updater. Every seam has a real default; the tests replace them. +func New(cfg Config, opts ...Option) (*Updater, error) { + if err := cfg.Validate(); err != nil { + return nil, err + } + u := &Updater{ + cfg: cfg.withDefaults(), + store: &Store{Dir: cfg.SnapshotDir}, + run: ExecRunner, + health: DialHealth, + log: func(string, ...any) {}, + now: time.Now, + } + for _, o := range opts { + o(u) + } + u.store.now = u.now + return u, nil +} + +// Option — a constructor seam. +type Option func(*Updater) + +func WithRunner(r Runner) Option { return func(u *Updater) { u.run = r } } +func WithHealth(h HealthCheck) Option { return func(u *Updater) { u.health = h } } +func WithLogger(l Logger) Option { return func(u *Updater) { u.log = l } } +func WithClock(f func() time.Time) Option { + return func(u *Updater) { u.now = f } +} + +// Snapshots lists what is available to roll back to, newest first. +func (u *Updater) Snapshots() ([]Snapshot, error) { return u.store.List() } diff --git a/internal/update/update_test.go b/internal/update/update_test.go new file mode 100644 index 0000000..6f72f68 --- /dev/null +++ b/internal/update/update_test.go @@ -0,0 +1,437 @@ +package update + +import ( + "context" + "errors" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// The tests drive the whole orchestration against a fake box: a directory tree +// standing in for the install dir, an injected Runner standing in for +// make/git/docker, and an injected HealthCheck standing in for mavend. That is +// what makes the failure paths — the ones that matter — testable at all: you +// cannot ask a real deployment to fail its health check on demand, and the +// rollback path is exactly the path nobody exercises by hand. + +type fakeBox struct { + t *testing.T + root string + + // what the fake `make build` writes into the source tree + newBytes string + // scripted failures + buildErr error + testErr error + restartErr error + + // health: fails until the Nth call, then follows healthy + healthErrs int // remaining failures to serve + healthy bool + healthChecks int + // deployedAtRestart records the installed bytes each time restart runs, so a + // test can prove the rollback put the old bytes back BEFORE restarting. + deployedAtRestart []string + + ran []string +} + +func newFakeBox(t *testing.T) *fakeBox { + t.Helper() + root := t.TempDir() + for _, d := range []string{"src", "install", "snapshots"} { + if err := os.MkdirAll(filepath.Join(root, d), 0o755); err != nil { + t.Fatal(err) + } + } + // The currently deployed build, and a config file next to it. + write(t, filepath.Join(root, "install", "mavend"), "OLD-BUILD") + write(t, filepath.Join(root, "install", "mavend.json"), `{"tick_interval":"60s"}`) + // The source tree already contains a stale binary; `make build` overwrites it. + write(t, filepath.Join(root, "src", "mavend"), "STALE") + return &fakeBox{t: t, root: root, newBytes: "NEW-BUILD", healthy: true} +} + +func (b *fakeBox) cfg() Config { + return Config{ + SourceDir: filepath.Join(b.root, "src"), + InstallDir: filepath.Join(b.root, "install"), + SnapshotDir: filepath.Join(b.root, "snapshots"), + Binaries: []string{"mavend"}, + ConfigFiles: []string{"mavend.json"}, + RestartCmd: []string{"restart-the-thing"}, + HealthSocket: filepath.Join(b.root, "mavend.sock"), + HealthTimeoutSec: 1, + KeepSnapshots: 3, + } +} + +func (b *fakeBox) run(ctx context.Context, dir string, argv []string) (string, error) { + b.ran = append(b.ran, strings.Join(argv, " ")) + switch strings.Join(argv, " ") { + case "make build": + if b.buildErr != nil { + return "ld: undefined reference to everything", b.buildErr + } + // A real build writes its artifacts into the working tree — the behaviour + // the snapshot-before-build ordering exists to survive. + write(b.t, filepath.Join(b.root, "src", "mavend"), b.newBytes) + return "built", nil + case "make test": + if b.testErr != nil { + return "--- FAIL: TestSomething", b.testErr + } + return "ok", nil + case "git rev-parse HEAD": + return "cafebabecafebabecafebabecafebabecafebabe\n", nil + case "restart-the-thing": + b.deployedAtRestart = append(b.deployedAtRestart, read(b.t, filepath.Join(b.root, "install", "mavend"))) + if b.restartErr != nil { + return "no such container", b.restartErr + } + return "restarted", nil + } + return "", errors.New("unexpected command: " + strings.Join(argv, " ")) +} + +func (b *fakeBox) health(ctx context.Context, socket string) error { + b.healthChecks++ + if b.healthErrs > 0 { + b.healthErrs-- + return errors.New("connection refused") + } + if !b.healthy { + return errors.New("she does not answer") + } + return nil +} + +func (b *fakeBox) updater(t *testing.T, extra ...Option) *Updater { + t.Helper() + opts := append([]Option{WithRunner(b.run), WithHealth(b.health)}, extra...) + u, err := New(b.cfg(), opts...) + if err != nil { + t.Fatal(err) + } + return u +} + +func (b *fakeBox) deployed() string { return read(b.t, filepath.Join(b.root, "install", "mavend")) } + +func write(t *testing.T, path, content string) { + t.Helper() + if err := os.WriteFile(path, []byte(content), 0o755); err != nil { + t.Fatal(err) + } +} + +func read(t *testing.T, path string) string { + t.Helper() + b, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + return string(b) +} + +func TestApply_HappyPath(t *testing.T) { + b := newFakeBox(t) + res, err := b.updater(t).Apply(context.Background()) + if err != nil { + t.Fatalf("Apply: %v", err) + } + if !res.Verified || !res.Restarted || !res.Healthy || res.RolledBack { + t.Fatalf("result = %+v; want verified+restarted+healthy and no rollback", res) + } + if got := b.deployed(); got != "NEW-BUILD" { + t.Errorf("deployed binary = %q; want the new build", got) + } + // The order is the property: health, snapshot, build, test, install, restart. + want := []string{"git rev-parse HEAD", "make build", "make test", "restart-the-thing"} + if strings.Join(b.ran, "|") != strings.Join(want, "|") { + t.Errorf("commands ran = %v; want %v", b.ran, want) + } + if res.SnapshotID == "" { + t.Error("no snapshot was taken") + } +} + +func TestApply_RefusesWhenSheIsAlreadyDown(t *testing.T) { + // A box that is already broken has no baseline for the rollback to prove + // itself against, so the update never starts. + b := newFakeBox(t) + b.healthy = false + res, err := b.updater(t).Apply(context.Background()) + if !errors.Is(err, ErrUnhealthyBefore) { + t.Fatalf("Apply on an unhealthy box = %v; want ErrUnhealthyBefore", err) + } + if len(b.ran) != 0 { + t.Errorf("a refused update still ran %v", b.ran) + } + if res.SnapshotID != "" { + t.Error("a refused update still took a snapshot") + } +} + +func TestApply_TestFailureDeploysNothingAndPutsTheTreeBack(t *testing.T) { + b := newFakeBox(t) + b.testErr = errors.New("exit status 1") + res, err := b.updater(t).Apply(context.Background()) + if !errors.Is(err, ErrVerifyFailed) { + t.Fatalf("Apply with failing tests = %v; want ErrVerifyFailed", err) + } + if res.Verified { + t.Error("result claims verified after a failing test suite") + } + for _, c := range b.ran { + if c == "restart-the-thing" { + t.Fatal("a failed verification restarted the daemon") + } + } + if got := b.deployed(); got != "OLD-BUILD" { + t.Errorf("deployed binary = %q; want the old build untouched", got) + } + // The failing output is kept so the operator can see why. + var found bool + for _, s := range res.Steps { + if s.Name == "test" && strings.Contains(s.Output, "FAIL") { + found = true + } + } + if !found { + t.Error("the failing test output was not retained") + } +} + +func TestApply_BuildFailureIsCaughtBeforeTheTests(t *testing.T) { + b := newFakeBox(t) + b.buildErr = errors.New("exit status 2") + if _, err := b.updater(t).Apply(context.Background()); !errors.Is(err, ErrVerifyFailed) { + t.Fatalf("Apply with a failing build = %v; want ErrVerifyFailed", err) + } + for _, c := range b.ran { + if c == "make test" { + t.Error("ran the test suite after the build failed") + } + } +} + +func TestApply_UnhealthyAfterRestartRollsBackToTheOldBytes(t *testing.T) { + // The case the package exists for: everything verifies, the new build + // installs, and then she does not come up. + b := newFakeBox(t) + b.healthErrs = 1 // the preflight check passes, then she stops answering + b.healthy = false + u := b.updater(t) + // Once the rollback restores the old build, she answers again. + restored := false + u.health = func(ctx context.Context, socket string) error { + b.healthChecks++ + if b.deployed() == "OLD-BUILD" && restored { + return nil + } + if b.healthChecks == 1 { + return nil // preflight: the old build is up + } + if b.deployed() == "OLD-BUILD" { + restored = true + return nil + } + return errors.New("she does not answer on the new build") + } + res, err := u.Apply(context.Background()) + if !errors.Is(err, ErrRolledBack) { + t.Fatalf("Apply with a dead new build = %v; want ErrRolledBack", err) + } + if !res.RolledBack || !res.RollbackHealthy || res.Healthy { + t.Fatalf("result = %+v; want rolled back and healthy again on the old build", res) + } + if got := b.deployed(); got != "OLD-BUILD" { + t.Errorf("deployed binary after the rollback = %q; want OLD-BUILD", got) + } + // And the restore happened BEFORE the second restart, not after it. + if len(b.deployedAtRestart) != 2 { + t.Fatalf("restarts = %v; want two (the update and the rollback)", b.deployedAtRestart) + } + if b.deployedAtRestart[0] != "NEW-BUILD" || b.deployedAtRestart[1] != "OLD-BUILD" { + t.Errorf("bytes in place at each restart = %v; want [NEW-BUILD OLD-BUILD]", b.deployedAtRestart) + } +} + +func TestApply_RestartFailureRollsBack(t *testing.T) { + b := newFakeBox(t) + b.restartErr = errors.New("exit status 1") + res, err := b.updater(t).Apply(context.Background()) + // The rollback's own restart fails too, so this is the manual-recovery case — + // and it says so instead of reporting a tidy rollback. + if !errors.Is(err, ErrRollbackFailed) { + t.Fatalf("Apply with a broken restart command = %v; want ErrRollbackFailed", err) + } + if !res.RolledBack || res.RollbackHealthy { + t.Fatalf("result = %+v; want rolled back but not healthy", res) + } + if got := b.deployed(); got != "OLD-BUILD" { + t.Errorf("deployed binary = %q; want the old bytes restored even so", got) + } +} + +func TestApply_RollbackNeedsNoBuildAndNoNewCode(t *testing.T) { + // The rollback must not depend on the toolchain, the source tree, or the + // code it is replacing. Prove it: delete the source tree's binary and make + // every command except the restart fail, then roll back. + b := newFakeBox(t) + if _, err := b.updater(t).Apply(context.Background()); err != nil { + t.Fatalf("setup Apply: %v", err) + } + if b.deployed() != "NEW-BUILD" { + t.Fatal("setup did not deploy") + } + os.RemoveAll(filepath.Join(b.root, "src")) + if err := os.MkdirAll(filepath.Join(b.root, "src"), 0o755); err != nil { + t.Fatal(err) + } + b.buildErr = errors.New("no toolchain here") + b.testErr = errors.New("no toolchain here") + b.ran = nil + + res, err := b.updater(t).Rollback(context.Background(), "") + if err != nil && !errors.Is(err, ErrRolledBack) { + t.Fatalf("Rollback: %v", err) + } + if !res.RollbackHealthy { + t.Fatalf("result = %+v; want a healthy rollback", res) + } + if got := b.deployed(); got != "OLD-BUILD" { + t.Errorf("deployed binary = %q; want OLD-BUILD", got) + } + for _, c := range b.ran { + if strings.HasPrefix(c, "make") { + t.Errorf("the rollback ran %q — it must not need a build", c) + } + } +} + +func TestApply_ConfigIsSnapshottedButNeverOverwritten(t *testing.T) { + b := newFakeBox(t) + // A config in the source tree must not be deployed over the operator's. + write(t, filepath.Join(b.root, "src", "mavend.json"), `{"tick_interval":"1s"}`) + if _, err := b.updater(t).Apply(context.Background()); err != nil { + t.Fatalf("Apply: %v", err) + } + if got := read(t, filepath.Join(b.root, "install", "mavend.json")); !strings.Contains(got, "60s") { + t.Errorf("installed config = %q; an update must not replace his config", got) + } + snaps, err := b.updater(t).Snapshots() + if err != nil || len(snaps) == 0 { + t.Fatalf("Snapshots: %v %v", snaps, err) + } + var names []string + for _, f := range snaps[0].Files { + names = append(names, f.Name) + } + if len(names) != 2 { + t.Errorf("snapshot files = %v; want the binary and the config", names) + } + if snaps[0].Commit == "" { + t.Error("the snapshot did not record which commit produced it") + } +} + +func TestRollback_CorruptSnapshotIsRefusedNotRestored(t *testing.T) { + b := newFakeBox(t) + if _, err := b.updater(t).Apply(context.Background()); err != nil { + t.Fatalf("setup Apply: %v", err) + } + snaps, _ := b.updater(t).Snapshots() + // Something ate the snapshot. Restoring it would deploy garbage. + write(t, filepath.Join(snaps[0].Dir(), "mavend"), "CORRUPT") + _, err := b.updater(t).Rollback(context.Background(), snaps[0].ID) + if !errors.Is(err, ErrRollbackFailed) || !strings.Contains(err.Error(), "corrupt") { + t.Fatalf("Rollback of a corrupt snapshot = %v; want a refusal naming the corruption", err) + } + if got := b.deployed(); got != "NEW-BUILD" { + t.Errorf("deployed binary = %q; a refused restore must change nothing", got) + } +} + +func TestRollback_NoSnapshots(t *testing.T) { + b := newFakeBox(t) + if _, err := b.updater(t).Rollback(context.Background(), ""); err == nil { + t.Error("Rollback with no snapshots succeeded; want an error") + } +} + +func TestPrune_KeepsTheNewestAsTheRollbackTarget(t *testing.T) { + b := newFakeBox(t) + st := &Store{Dir: filepath.Join(b.root, "snapshots")} + base := time.Date(2026, 8, 1, 3, 0, 0, 0, time.UTC) + for i := 0; i < 4; i++ { + i := i + st.now = func() time.Time { return base.Add(time.Duration(i) * time.Minute) } + if _, err := st.Save(filepath.Join(b.root, "install"), []string{"mavend"}, "", ""); err != nil { + t.Fatal(err) + } + } + if err := st.Prune(0); err != nil { // 0 is clamped to 1, never to zero + t.Fatal(err) + } + snaps, err := st.List() + if err != nil { + t.Fatal(err) + } + if len(snaps) != 1 { + t.Fatalf("kept %d snapshots; want 1", len(snaps)) + } + if snaps[0].ID != "20260801-030300" { + t.Errorf("kept %s; want the newest", snaps[0].ID) + } +} + +func TestList_IgnoresSnapshotsWithNoManifest(t *testing.T) { + // An interrupted snapshot has files but no manifest. It must never be offered + // as a rollback target — restoring a half-copied binary is the worst outcome + // in the package. + b := newFakeBox(t) + dir := filepath.Join(b.root, "snapshots", "20260801-000000") + if err := os.MkdirAll(dir, 0o700); err != nil { + t.Fatal(err) + } + write(t, filepath.Join(dir, "mavend"), "HALF") + snaps, err := (&Store{Dir: filepath.Join(b.root, "snapshots")}).List() + if err != nil { + t.Fatal(err) + } + if len(snaps) != 0 { + t.Errorf("List returned %d snapshots; want none (no manifest)", len(snaps)) + } +} + +func TestConfigValidate(t *testing.T) { + ok := (&fakeBox{root: t.TempDir()}).cfg() + if err := ok.Validate(); err != nil { + t.Fatalf("valid config rejected: %v", err) + } + bad := map[string]func(c Config) Config{ + "relative source": func(c Config) Config { c.SourceDir = "src"; return c }, + "no restart command": func(c Config) Config { c.RestartCmd = nil; return c }, + "no health socket": func(c Config) Config { c.HealthSocket = ""; return c }, + "no binaries": func(c Config) Config { c.Binaries = nil; return c }, + "escaping artifact name": func(c Config) Config { c.Binaries = []string{"../../etc/passwd"}; return c }, + "absolute artifact name": func(c Config) Config { c.Binaries = []string{"/usr/bin/mavend"}; return c }, + "snapshots inside install": func(c Config) Config { c.SnapshotDir = filepath.Join(c.InstallDir, "snaps"); return c }, + } + for name, mutate := range bad { + if err := mutate(ok).Validate(); err == nil { + t.Errorf("%s was accepted; want a startup failure", name) + } + } + // And New refuses an invalid config outright rather than half-configuring. + if _, err := New(mutate(ok, "no health socket", bad)); err == nil { + t.Error("New accepted a config with no health socket") + } +} + +func mutate(c Config, key string, m map[string]func(Config) Config) Config { return m[key](c) } diff --git a/internal/update/verify.go b/internal/update/verify.go new file mode 100644 index 0000000..26b0511 --- /dev/null +++ b/internal/update/verify.go @@ -0,0 +1,65 @@ +package update + +import ( + "context" + "fmt" + "time" +) + +// Verification is "does this tree build and does it pass its own tests", run +// before a single byte is written to the install dir. +// +// It is `make build` and `make test`, not `go build`: the CGO daemons need the +// vendored toolchain and the whisper/piper include and library paths wired +// through the Makefile, and a bare `go build` on them fails in a way that has +// nothing to do with the change being deployed. `make test` is the -race suite +// with the CGO env set, and it is the only evidence available on a single box +// that the new code does what the old code did. +// +// This is not a substitute for a second environment. A test suite that passes +// says the code is self-consistent; it does not say the new build will start +// against this machine's actual models, sockets and encrypted store. That is +// what the post-restart health check is for, and it is why the install is +// reversible rather than merely careful. + +// Step — one verification or orchestration step and how it went. Kept so the CLI +// can print a truthful account of what was done, including on the failure path. +type Step struct { + Name string + Argv []string + Took time.Duration + Err error + Output string // combined output, only retained for failures +} + +// Verify runs the build and the test suite in SourceDir. +func (u *Updater) Verify(ctx context.Context) ([]Step, error) { + ctx, cancel := context.WithTimeout(ctx, u.cfg.verifyTimeout()) + defer cancel() + var steps []Step + for _, argv := range [][]string{{"make", "build"}, {"make", "test"}} { + u.log("verify: %v (this takes a while)", argv) + start := u.now() + out, err := u.run(ctx, u.cfg.SourceDir, argv) + st := Step{Name: argv[len(argv)-1], Argv: argv, Took: u.now().Sub(start), Err: err} + if err != nil { + st.Output = tail(out, 4000) + } + steps = append(steps, st) + if err != nil { + u.log("verify: %v FAILED after %s", argv, st.Took.Round(time.Second)) + return steps, fmt.Errorf("%w: %v: %v", ErrVerifyFailed, argv, err) + } + u.log("verify: %v ok in %s", argv, st.Took.Round(time.Second)) + } + return steps, nil +} + +// tail keeps the last n bytes — a failing `make test` prints far more than is +// useful, and the failure is always at the end. +func tail(s string, n int) string { + if len(s) <= n { + return s + } + return "…" + s[len(s)-n:] +}