diff --git a/README.md b/README.md index 13c2126..6c5cc4c 100644 --- a/README.md +++ b/README.md @@ -44,3 +44,16 @@ The daemon has no TCP listener. Restic passwords are passed through root-only te ## Current scope The code implements the MVP control plane and adapters. Docker and VM operations require the standard Unraid `docker` and `virsh` commands. Managed SMB/NFS mounts require the corresponding Unraid mount helpers. Hardware and end-to-end compatibility must be validated on supported Unraid 7 releases before a stable publication. + + +### Backup safety (2026.09.21.r001) + +Repository checks report success/failure through configured notification targets. The +standard check verifies repository structure; it is not a full data-read or restore test. +Docker jobs discover bind mounts and named volumes. Live backups do not guarantee +application/database consistency. Use stop mode for coordinated offline backups. +Stop mode restarts only workloads that were running before preparation; ambiguous +paused/restarting states fail safely. Each restart has a 120-second deadline and +recovery continues after individual failures. Shutdown waits for cleanup; the rc +script refuses a forced kill or overlapping restart if cleanup exceeds five minutes. +Hard crashes or power loss still require checking workload state and backup results. diff --git a/cmd/urbmd/main.go b/cmd/urbmd/main.go index 79113e1..5197afe 100644 --- a/cmd/urbmd/main.go +++ b/cmd/urbmd/main.go @@ -87,7 +87,7 @@ func main() { shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 15*time.Second) defer shutdownCancel() _ = server.Shutdown(shutdownCtx) - if err := svc.Wait(shutdownCtx); err != nil { + if err := svc.Wait(context.Background()); err != nil { log.Warn("queue cleanup did not finish before shutdown timeout", "error", err) } log.Info("urbm daemon stopped") diff --git a/dist/urbm-2026.09.21.r001-x86_64-1.txz b/dist/urbm-2026.09.21.r001-x86_64-1.txz new file mode 100644 index 0000000..361d9d2 Binary files /dev/null and b/dist/urbm-2026.09.21.r001-x86_64-1.txz differ diff --git a/dist/urbm-2026.09.21.r001-x86_64-1.txz.sha256 b/dist/urbm-2026.09.21.r001-x86_64-1.txz.sha256 new file mode 100644 index 0000000..583c384 --- /dev/null +++ b/dist/urbm-2026.09.21.r001-x86_64-1.txz.sha256 @@ -0,0 +1 @@ +55d48398ecd446793f51c0fcc987994ba77e52ac950dbc7f5a32f983449e5b83 urbm-2026.09.21.r001-x86_64-1.txz diff --git a/dist/urbm.plg b/dist/urbm.plg index 170c107..9d5f16b 100644 --- a/dist/urbm.plg +++ b/dist/urbm.plg @@ -2,13 +2,21 @@ - + - + ]> +### 2026.09.21.r001 +- Notify on successful and failed repository checks; complete notification attempts before finishing a task. +- Preserve stopped Docker/VM workloads and refuse ambiguous paused or restarting states. +- Recover each previously running workload with its own timeout, including cancellation and partial preparation failures. +- Include named Docker volumes and reject persistent mounts without a usable source path. +- Avoid force-killing the daemon during workload recovery and prevent overlapping daemon restarts. +- Add regression coverage for check notifications, volume discovery, state preservation and recovery failures. + ### 2026.07.13.r012 - Draw a separate color-coded line and legend entry for every backup job when the dashboard filter is set to all jobs. - Base dashboard health on successful backups, failed backup attempts, active work, and timezone-aware overdue schedules instead of unrelated maintenance or restore successes. @@ -402,7 +410,7 @@ chmod 0700 /boot/config/plugins/&name; -/etc/rc.d/rc.urbm stop || true +/etc/rc.d/rc.urbm stop || exit 1 removepkg &name; rm -rf /usr/local/emhttp/plugins/&name; /usr/local/libexec/&name; /usr/local/sbin/urbmd /etc/rc.d/rc.urbm /run/urbm /var/lib/urbm diff --git a/internal/platform/recovery_test.go b/internal/platform/recovery_test.go new file mode 100644 index 0000000..30ca189 --- /dev/null +++ b/internal/platform/recovery_test.go @@ -0,0 +1,120 @@ +package platform + +import ( + "context" + "git.casaderoll.de/michael/urbm/internal/model" + "os" + "path/filepath" + "strings" + "testing" +) + +func fakeDocker(t *testing.T, body string) string { + t.Helper() + dir := t.TempDir() + path := filepath.Join(dir, "docker") + if err := os.WriteFile(path, []byte("#!/bin/sh\n"+body), 0700); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", dir+":"+os.Getenv("PATH")) + return dir +} + +func TestPreparePreservesStoppedContainersAndIncludesVolumes(t *testing.T) { + dir := fakeDocker(t, `case "$1" in + inspect) + if [ "$2" = --format ]; then + if [ "$4" = off ]; then echo '{"Running":false}'; else echo '{"Running":true}'; fi + else echo '[{"Mounts":[{"Type":"bind","Source":"/appdata"},{"Type":"volume","Source":"/docker/volumes/db/_data"},{"Type":"tmpfs","Source":""}]}]'; fi;; + stop|start) echo "$1 $2" >> "$ACTIONS";; + esac +`) + actions := filepath.Join(dir, "actions") + t.Setenv("ACTIONS", actions) + w := &WorkloadManager{RuntimeDir: dir} + job := model.Job{ID: "test", Type: model.JobDocker, Sources: []model.Source{{WorkloadID: "off"}, {WorkloadID: "on"}}} + job.Consistency.Mode = "stop" + p, err := w.Prepare(context.Background(), job) + if err != nil { + t.Fatal(err) + } + if len(p.Stopped) != 1 || p.Stopped[0].WorkloadID != "on" { + t.Fatalf("stopped: %+v", p.Stopped) + } + if !strings.Contains(strings.Join(p.Sources, "\n"), "/docker/volumes/db/_data") { + t.Fatal("named volume omitted") + } + if err = w.Cleanup(context.Background(), p); err != nil { + t.Fatal(err) + } + data, _ := os.ReadFile(actions) + if strings.Contains(string(data), "start off") || !strings.Contains(string(data), "start on") { + t.Fatalf("actions: %s", data) + } +} + +func TestRecoveryContinuesAfterFailureAndCancellation(t *testing.T) { + dir := fakeDocker(t, `echo "$1 $2" >> "$ACTIONS" + if [ "$2" = broken ]; then exit 1; fi +`) + actions := filepath.Join(dir, "actions") + t.Setenv("ACTIONS", actions) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + w := &WorkloadManager{} + err := w.Cleanup(ctx, Prepared{Kind: model.JobDocker, Stopped: []model.Source{{WorkloadID: "healthy"}, {WorkloadID: "broken"}}}) + if err == nil || !strings.Contains(err.Error(), "broken") { + t.Fatalf("error: %v", err) + } + data, _ := os.ReadFile(actions) + if !strings.Contains(string(data), "start healthy") { + t.Fatalf("recovery aborted: %s", data) + } +} + +func TestPrepareRollsBackPreviouslyStoppedContainers(t *testing.T) { + dir := fakeDocker(t, `case "$1" in + inspect) + if [ "$2" = bad ]; then exit 1; fi + if [ "$2" = --format ]; then echo '{"Running":true}'; else echo '[{"Mounts":[]}]'; fi;; + stop|start) echo "$1 $2" >> "$ACTIONS";; + esac +`) + actions := filepath.Join(dir, "actions") + t.Setenv("ACTIONS", actions) + w := &WorkloadManager{RuntimeDir: dir} + job := model.Job{ID: "test", Type: model.JobDocker, Sources: []model.Source{{WorkloadID: "first"}, {WorkloadID: "bad"}}} + job.Consistency.Mode = "stop" + if _, err := w.Prepare(context.Background(), job); err == nil { + t.Fatal("expected failure") + } + data, _ := os.ReadFile(actions) + if !strings.Contains(string(data), "start first") { + t.Fatalf("missing rollback: %s", data) + } +} + +func TestSlowRecoveryDoesNotExhaustOtherStarts(t *testing.T) { + dir := fakeDocker(t, `if [ "$2" = slow ]; then sleep 6; fi + echo "$2" >> "$ACTIONS" +`) + actions := filepath.Join(dir, "actions") + t.Setenv("ACTIONS", actions) + err := (&WorkloadManager{}).Cleanup(context.Background(), Prepared{Kind: model.JobDocker, Stopped: []model.Source{{WorkloadID: "next"}, {WorkloadID: "slow"}}}) + if err != nil { + t.Fatal(err) + } + data, _ := os.ReadFile(actions) + if string(data) != "slow\nnext\n" { + t.Fatalf("incomplete recovery: %s", data) + } +} + +func TestPersistentMountWithoutPathFails(t *testing.T) { + dir := fakeDocker(t, `echo '[{"Mounts":[{"Type":"volume","Source":""}]}]' +`) + w := &WorkloadManager{} + if _, err := w.captureMetadata(context.Background(), model.JobDocker, "db", dir); err == nil { + t.Fatal("missing volume path silently omitted") + } +} diff --git a/internal/platform/workloads.go b/internal/platform/workloads.go index 40064fb..87a1511 100644 --- a/internal/platform/workloads.go +++ b/internal/platform/workloads.go @@ -4,6 +4,7 @@ import ( "context" "encoding/json" "encoding/xml" + "errors" "fmt" "os" "os/exec" @@ -138,16 +139,21 @@ func (w *WorkloadManager) Prepare(ctx context.Context, job model.Job) (Prepared, } discovered, err := w.captureMetadata(ctx, job.Type, source.WorkloadID, metadataDir) if err != nil { - w.Cleanup(context.Background(), prepared) - return Prepared{}, err + return Prepared{}, errors.Join(err, w.Cleanup(context.Background(), prepared)) } prepared.Sources = append(prepared.Sources, discovered...) if job.Consistency.Mode == "stop" { - if err := w.stop(ctx, job.Type, source.WorkloadID, time.Duration(job.ShutdownSecs)*time.Second); err != nil { - w.Cleanup(context.Background(), prepared) - return Prepared{}, err + running, err := w.running(ctx, job.Type, source.WorkloadID) + if err != nil { + return Prepared{}, errors.Join(err, w.Cleanup(context.Background(), prepared)) + } + if running { + // Record before stopping: cancellation can happen after the daemon stopped it. + prepared.Stopped = append(prepared.Stopped, source) + if err := w.stop(ctx, job.Type, source.WorkloadID, time.Duration(job.ShutdownSecs)*time.Second); err != nil { + return Prepared{}, errors.Join(err, w.Cleanup(context.Background(), prepared)) + } } - prepared.Stopped = append(prepared.Stopped, source) } } prepared.Sources = uniqueStrings(prepared.Sources) @@ -203,22 +209,56 @@ func (w *WorkloadManager) FlashDevice(ctx context.Context) (string, error) { } func (w *WorkloadManager) Cleanup(ctx context.Context, prepared Prepared) error { - if _, hasDeadline := ctx.Deadline(); !hasDeadline { - var cancel context.CancelFunc - ctx, cancel = context.WithTimeout(ctx, 5*time.Second) - defer cancel() - } - var first error + var failures []error + // Recovery must continue even when the backup was cancelled. Each workload + // receives its own deadline so one slow start cannot starve the others. for i := len(prepared.Stopped) - 1; i >= 0; i-- { source := prepared.Stopped[i] - if err := w.start(ctx, prepared.Kind, source.WorkloadID); err != nil && first == nil { - first = err + startCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 120*time.Second) + if err := w.start(startCtx, prepared.Kind, source.WorkloadID); err != nil { + failures = append(failures, fmt.Errorf("restart %s: %w", source.WorkloadID, err)) } + cancel() } if prepared.MetadataDir != "" { - _ = os.RemoveAll(prepared.MetadataDir) + if err := os.RemoveAll(prepared.MetadataDir); err != nil { + failures = append(failures, err) + } + } + return errors.Join(failures...) +} + +func (w *WorkloadManager) running(ctx context.Context, kind model.JobType, id string) (bool, error) { + if kind == model.JobDocker { + output, err := exec.CommandContext(ctx, "docker", "inspect", "--format", "{{json .State}}", id).Output() + if err != nil { + return false, fmt.Errorf("source: inspect state of %s: %w", id, err) + } + var state struct { + Running bool + Paused bool + Restarting bool + } + if err := json.Unmarshal(output, &state); err != nil { + return false, err + } + if state.Paused || state.Restarting { + return false, fmt.Errorf("source: %s is paused or restarting; refusing to change its state", id) + } + return state.Running, nil + } + output, err := exec.CommandContext(ctx, "virsh", "domstate", id).Output() + if err != nil { + return false, fmt.Errorf("source: inspect VM state %s: %w", id, err) + } + switch strings.TrimSpace(string(output)) { + case "running": + return true, nil + case "shut off": + return false, nil + default: + return false, fmt.Errorf("source: VM %s is in an unsupported state", id) } - return first } func (w *WorkloadManager) captureMetadata(ctx context.Context, kind model.JobType, id, dir string) ([]string, error) { @@ -249,7 +289,10 @@ func (w *WorkloadManager) captureMetadata(ctx context.Context, kind model.JobTyp var paths []string if len(inspected) > 0 { for _, mount := range inspected[0].Mounts { - if mount.Type == "bind" && filepath.IsAbs(mount.Source) { + if mount.Type == "bind" || mount.Type == "volume" { + if !filepath.IsAbs(mount.Source) { + return nil, fmt.Errorf("source: container %s has a %s mount without an absolute backup path", id, mount.Type) + } paths = append(paths, mount.Source) } } diff --git a/internal/service/notification_test.go b/internal/service/notification_test.go index 143259a..93a527e 100644 --- a/internal/service/notification_test.go +++ b/internal/service/notification_test.go @@ -1,6 +1,15 @@ package service import ( + "context" + "fmt" + "git.casaderoll.de/michael/urbm/internal/notify" + "git.casaderoll.de/michael/urbm/internal/platform" + "git.casaderoll.de/michael/urbm/internal/queue" + "git.casaderoll.de/michael/urbm/internal/restic" + "log/slog" + "os" + "path/filepath" "strings" "testing" "time" @@ -42,3 +51,41 @@ func TestNotificationMessageIncludesPruneSummary(t *testing.T) { t.Fatalf("message contains redundant prune completion details: %q", message) } } + +type checkSecrets struct{} + +func (checkSecrets) Get(string) (string, error) { return "test", nil } + +func TestCheckSendsSuccessAndFailureNotifications(t *testing.T) { + for _, code := range []int{0, 1} { + t.Run(fmt.Sprint(code), func(t *testing.T) { + dir := t.TempDir() + binary := filepath.Join(dir, "restic") + if err := os.WriteFile(binary, []byte(fmt.Sprintf("#!/bin/sh\necho check-result >&2\nexit %d\n", code)), 0700); err != nil { + t.Fatal(err) + } + var sent []string + s := &Service{ + config: model.Config{Repositories: []model.Repository{{ID: "repo", Name: "test", Type: model.RepositoryLocal, Location: dir}}, Notifications: []model.NotificationTarget{{Type: "unraid", Enabled: true, Events: []string{"success", "failed"}}}}, + repositoryLocks: map[string]chan struct{}{}, + restic: &restic.Runner{Binary: binary, RuntimeDir: dir, Secrets: checkSecrets{}}, + mounts: &platform.MountManager{}, queue: queue.New(nil, nil), log: slog.Default(), + notifier: ¬ify.Sender{RunCommand: func(_ context.Context, _ string, args ...string) error { + sent = append(sent, strings.Join(args, " ")) + return nil + }}, + } + result := s.execute(context.Background(), model.Run{ID: "run", JobID: "repo", TaskType: "check"}) + want := "success" + if code != 0 { + want = "failed" + } + if result.Status != want || len(sent) != 1 { + t.Fatalf("status=%s messages=%v", result.Status, sent) + } + if !strings.Contains(sent[0], "URBM") { + t.Fatal(sent) + } + }) + } +} diff --git a/internal/service/service.go b/internal/service/service.go index a586b84..771655a 100644 --- a/internal/service/service.go +++ b/internal/service/service.go @@ -1052,8 +1052,12 @@ func (s *Service) execute(ctx context.Context, run model.Run) (result model.Run) ctx = restic.WithRunID(ctx, run.ID) maintenanceRepoName := "" defer func() { - if result.TaskType == "prune" && (result.Status == "success" || result.Status == "warning" || result.Status == "failed") { - go s.notify("Repository-Bereinigung", maintenanceRepoName, result) + if (result.TaskType == "prune" || result.TaskType == "check") && (result.Status == "success" || result.Status == "warning" || result.Status == "failed") { + name := "Repository-Bereinigung" + if result.TaskType == "check" { + name = "Repository-Prüfung" + } + s.notify(name, maintenanceRepoName, result) } }() if run.TaskType == "backup" { @@ -1543,18 +1547,20 @@ func (s *Service) finishJob(run model.Run, job model.Job, result model.Run) mode repoName = repo.Name } } - go s.notify(job.Name, repoName, result) + s.notify(job.Name, repoName, result) } return result } func (s *Service) notify(name, repository string, run model.Run) { + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() title := notificationTitle(name, repository, run) message := notificationMessage(name, repository, run, time.Now().UTC()) for _, target := range s.Config().Notifications { for _, event := range target.Events { if event == run.Status { - if err := s.notifier.Send(context.Background(), target, title, message, run.Status); err != nil { + if err := s.notifier.Send(ctx, target, title, message, run.Status); err != nil { s.log.Warn("send notification failed", "target", target.ID, "type", target.Type, "error", err) } break diff --git a/packaging/build-package.sh b/packaging/build-package.sh index 2f40fea..742f73b 100755 --- a/packaging/build-package.sh +++ b/packaging/build-package.sh @@ -39,7 +39,12 @@ cp "$ROOT/plugin/rc.urbm" "$STAGE/etc/rc.d/rc.urbm" cp "$ROOT/plugin/doinst.sh" "$ROOT/plugin/slack-desc" "$STAGE/install/" chmod 0755 "$STAGE/etc/rc.d/rc.urbm" "$STAGE/install/doinst.sh" -(cd "$STAGE" && tar --owner=0 --group=0 -cJf "$DIST/urbm-${VERSION}-x86_64-1.txz" .) +# Suppress macOS metadata and normalize ownership with GNU or BSD tar. +if tar --version | grep -q 'GNU tar'; then + (cd "$STAGE" && COPYFILE_DISABLE=1 tar --owner=0 --group=0 -cJf "$DIST/urbm-${VERSION}-x86_64-1.txz" .) +else + (cd "$STAGE" && COPYFILE_DISABLE=1 tar --uid 0 --gid 0 --uname root --gname root -cJf "$DIST/urbm-${VERSION}-x86_64-1.txz" .) +fi (cd "$DIST" && sha256sum "urbm-${VERSION}-x86_64-1.txz" > "urbm-${VERSION}-x86_64-1.txz.sha256") PACKAGE_SHA256=$(sha256sum "$DIST/urbm-${VERSION}-x86_64-1.txz" | awk '{print $1}') sed -e "s///" -e "s/REPLACE_DURING_RELEASE/$PACKAGE_SHA256/" "$ROOT/plugin/urbm.plg" > "$DIST/urbm.plg" diff --git a/plugin/doinst.sh b/plugin/doinst.sh index 8f512f3..a3f82f7 100755 --- a/plugin/doinst.sh +++ b/plugin/doinst.sh @@ -48,5 +48,5 @@ rm -f \ "$ROOT_PREFIX/var/log/$OLD_NAME.log" if [ -x "$ROOT_PREFIX/etc/rc.d/rc.urbm" ]; then - "$ROOT_PREFIX/etc/rc.d/rc.urbm" restart || true + "$ROOT_PREFIX/etc/rc.d/rc.urbm" restart fi diff --git a/plugin/rc.urbm b/plugin/rc.urbm index 6b8d549..2debed2 100755 --- a/plugin/rc.urbm +++ b/plugin/rc.urbm @@ -6,7 +6,9 @@ LOGFILE=/var/log/urbm.log is_daemon_pid() { PID=${1:-} - [ -n "$PID" ] && [ "$(readlink -f "/proc/$PID/exe" 2>/dev/null)" = "$DAEMON" ] + [ -n "$PID" ] || return 1 + EXE=$(readlink "/proc/$PID/exe" 2>/dev/null) + [ "$EXE" = "$DAEMON" ] || [ "$EXE" = "$DAEMON (deleted)" ] } start() { @@ -35,11 +37,14 @@ stop() { PID=$(cat "$PIDFILE") if is_daemon_pid "$PID"; then kill "$PID" 2>/dev/null || true - for _ in $(seq 1 30); do + for _ in $(seq 1 300); do is_daemon_pid "$PID" || break sleep 1 done - is_daemon_pid "$PID" && kill -9 "$PID" 2>/dev/null || true + if is_daemon_pid "$PID"; then + echo "urbmd is still recovering workloads; refusing to force-kill it" >&2 + return 1 + fi STOPPED=1 fi rm -f "$PIDFILE" @@ -49,13 +54,14 @@ stop() { for PID in $PIDS; do kill "$PID" 2>/dev/null || true done - for _ in $(seq 1 30); do + for _ in $(seq 1 300); do [ -z "$(pidof urbmd 2>/dev/null)" ] && break sleep 1 done - for PID in $(pidof urbmd 2>/dev/null); do - kill -9 "$PID" 2>/dev/null || true - done + if [ -n "$(pidof urbmd 2>/dev/null)" ]; then + echo "urbmd is still recovering workloads; stop did not complete" >&2 + return 1 + fi fi rm -f /run/urbm/urbm.sock } @@ -63,7 +69,7 @@ stop() { case "${1:-}" in start) start ;; stop) stop ;; - restart) stop; start ;; + restart) stop && start ;; status) if [ -s "$PIDFILE" ] && is_daemon_pid "$(cat "$PIDFILE")"; then echo "urbmd is running"; else echo "urbmd is stopped"; exit 1; fi ;; diff --git a/plugin/urbm.plg b/plugin/urbm.plg index cde83f8..083c1c5 100644 --- a/plugin/urbm.plg +++ b/plugin/urbm.plg @@ -2,13 +2,21 @@ - + ]> +### 2026.09.21.r001 +- Notify on successful and failed repository checks; complete notification attempts before finishing a task. +- Preserve stopped Docker/VM workloads and refuse ambiguous paused or restarting states. +- Recover each previously running workload with its own timeout, including cancellation and partial preparation failures. +- Include named Docker volumes and reject persistent mounts without a usable source path. +- Avoid force-killing the daemon during workload recovery and prevent overlapping daemon restarts. +- Add regression coverage for check notifications, volume discovery, state preservation and recovery failures. + ### 2026.07.13.r012 - Draw a separate color-coded line and legend entry for every backup job when the dashboard filter is set to all jobs. - Base dashboard health on successful backups, failed backup attempts, active work, and timezone-aware overdue schedules instead of unrelated maintenance or restore successes. @@ -402,7 +410,7 @@ chmod 0700 /boot/config/plugins/&name; -/etc/rc.d/rc.urbm stop || true +/etc/rc.d/rc.urbm stop || exit 1 removepkg &name; rm -rf /usr/local/emhttp/plugins/&name; /usr/local/libexec/&name; /usr/local/sbin/urbmd /etc/rc.d/rc.urbm /run/urbm /var/lib/urbm diff --git a/webgui/URBM.page b/webgui/URBM.page index 1e89ed7..60cdd16 100644 --- a/webgui/URBM.page +++ b/webgui/URBM.page @@ -8,7 +8,7 @@ Tag="URBM Unraid Restic Backup Manager backup snapshots restore" ---