diff --git a/ansible/roles/podman/tasks/containers/home/zomboid.yml b/ansible/roles/podman/tasks/containers/home/zomboid.yml index 65ecfb0..9487b9e 100644 --- a/ansible/roles/podman/tasks/containers/home/zomboid.yml +++ b/ansible/roles/podman/tasks/containers/home/zomboid.yml @@ -34,6 +34,10 @@ loop: - config-template - config-backup + # Holds the world each restore displaces, so a restore is always undoable. + # Deliberately outside data/ -- that is the container's bind mount, and a + # stray world directory inside Saves/Multiplayer/ is something PZ would see. + - restore-backup - name: create zomboid host-side log directory become: true @@ -82,6 +86,35 @@ mode: '0644' notify: reload zomboid systemd +- name: deploy zomboid restore script + become: true + ansible.builtin.template: + src: zomboid/zomboid-restore.sh.j2 + dest: "{{ podman_home }}/bin/zomboid-restore.sh" + owner: "{{ podman_user }}" + group: "{{ podman_user }}" + mode: '0755' + +- name: deploy zomboid restore path unit + become: true + ansible.builtin.template: + src: zomboid/zomboid-restore.path.j2 + dest: "{{ podman_home }}/.config/systemd/user/zomboid-restore.path" + owner: "{{ podman_user }}" + group: "{{ podman_user }}" + mode: '0644' + notify: reload zomboid systemd + +- name: deploy zomboid restore service unit + become: true + ansible.builtin.template: + src: zomboid/zomboid-restore.service.j2 + dest: "{{ podman_home }}/.config/systemd/user/zomboid-restore.service" + owner: "{{ podman_user }}" + group: "{{ podman_user }}" + mode: '0644' + notify: reload zomboid systemd + - name: deploy zomboid stats script become: true ansible.builtin.template: @@ -761,3 +794,15 @@ enabled: true state: started daemon_reload: true + +# Restore is triggered the same way -- Discord bot -> trigger file -> path unit. +# See zomboid-restore.path and zomboid-restore.service. +- name: enable zomboid restore path unit + become: true + become_user: "{{ podman_user }}" + ansible.builtin.systemd: + name: zomboid-restore.path + scope: user + enabled: true + state: started + daemon_reload: true diff --git a/ansible/roles/podman/tasks/main.yml b/ansible/roles/podman/tasks/main.yml index edae065..b43c0c3 100644 --- a/ansible/roles/podman/tasks/main.yml +++ b/ansible/roles/podman/tasks/main.yml @@ -123,7 +123,7 @@ - import_tasks: containers/home/gregtime.yml vars: - image: localhost/greg-time-bot:3.16.5 + image: localhost/greg-time-bot:3.17.1 tags: gregtime # Gated on zomboid_enabled (roles/podman/defaults/main.yml) so it can be taken diff --git a/ansible/roles/podman/templates/zomboid/zomboid-restore.path.j2 b/ansible/roles/podman/templates/zomboid/zomboid-restore.path.j2 new file mode 100644 index 0000000..9adc0f1 --- /dev/null +++ b/ansible/roles/podman/templates/zomboid/zomboid-restore.path.j2 @@ -0,0 +1,9 @@ +[Unit] +Description=Watch for Zomboid backup restore trigger + +[Path] +PathExists={{ podman_home }}/.local/share/volumes/gregtime/data/zomboid-restore.trigger +Unit=zomboid-restore.service + +[Install] +WantedBy=default.target diff --git a/ansible/roles/podman/templates/zomboid/zomboid-restore.service.j2 b/ansible/roles/podman/templates/zomboid/zomboid-restore.service.j2 new file mode 100644 index 0000000..205cc1b --- /dev/null +++ b/ansible/roles/podman/templates/zomboid/zomboid-restore.service.j2 @@ -0,0 +1,8 @@ +[Unit] +Description=Zomboid Backup Restore Service + +[Service] +Type=oneshot +ExecStart={{ podman_home }}/bin/zomboid-restore.sh +StandardOutput=journal +StandardError=journal diff --git a/ansible/roles/podman/templates/zomboid/zomboid-restore.sh.j2 b/ansible/roles/podman/templates/zomboid/zomboid-restore.sh.j2 new file mode 100644 index 0000000..b3a2bcc --- /dev/null +++ b/ansible/roles/podman/templates/zomboid/zomboid-restore.sh.j2 @@ -0,0 +1,222 @@ +#!/bin/bash +# Zomboid Backup Restore Script +# Triggered by systemd path unit when the discord bot requests a restore. +# +# Sibling of world-reset.sh: same trigger-file -> path-unit -> oneshot shape, same +# podman unshare discipline. The difference is that a reset throws the world away +# and this puts an older one back, so it is a good deal more careful: +# +# - it resolves the target itself and never accepts a path from the trigger +# - it refuses and exits BEFORE stopping the server if anything is wrong +# - it moves the live world aside instead of deleting it, so every restore +# is undoable + +set -e + +VOL="{{ podman_home }}/.local/share/volumes" +LOGFILE="${VOL}/zomboid/logs/restore.log" +TRIGGER_FILE="${VOL}/gregtime/data/zomboid-restore.trigger" +RESULT_FILE="${VOL}/gregtime/data/zomboid-restore.result" +SERVER_NAME="{{ zomboid_server_name }}" +DATA="${VOL}/zomboid/data" +SAVES_PATH="${DATA}/Saves/Multiplayer/${SERVER_NAME}" +DB_PATH="${DATA}/db/${SERVER_NAME}.db" +BACKUPS="${DATA}/backups" +SNAP_ROOT="${VOL}/zomboid/restore-backup" +STAGING="${VOL}/zomboid/restore-staging" +KEEP_SNAPSHOTS=3 + +log() { + local msg="[$(date '+%Y-%m-%d %H:%M:%S')] $1" + echo "$msg" + # Same reasoning as world-reset.sh: stdout is already in the journal, and an + # unwritable log file must never be what aborts a restore under set -e. + echo "$msg" >> "$LOGFILE" 2>/dev/null || true +} + +# The bot reads this back to report into #zomboid. Written as the container uid +# so the bot (uid 1000 inside its own container) can actually open it. +result() { + local ok="$1" detail="$2" + printf '{"ok":%s,"detail":%s,"at":"%s"}\n' \ + "$ok" "$(printf '%s' "$detail" | sed 's/\\/\\\\/g; s/"/\\"/g; s/^/"/; s/$/"/')" \ + "$(date -Is)" > /tmp/zomboid-restore.result.$$ 2>/dev/null || return 0 + podman unshare cp /tmp/zomboid-restore.result.$$ "$RESULT_FILE" 2>/dev/null || true + podman unshare chown 1000:1000 "$RESULT_FILE" 2>/dev/null || true + rm -f /tmp/zomboid-restore.result.$$ 2>/dev/null || true +} + +die() { + log "ABORT: $1" + result false "$1" + exit 1 +} + +export XDG_RUNTIME_DIR="/run/user/$(id -u)" + +log "Restore triggered" + +# --------------------------------------------------------------------------- +# 1. Read and validate the trigger +# --------------------------------------------------------------------------- +podman unshare test -f "$TRIGGER_FILE" || die "no trigger file" +TRIGGER_BODY="$(podman unshare cat "$TRIGGER_FILE")" +podman unshare rm -f "$TRIGGER_FILE" + +field() { printf '%s\n' "$TRIGGER_BODY" | sed -n "s/^$1=//p" | head -1 | tr -d '\r'; } + +ACTION="$(field action)" +SET="$(field set)" +INDEX="$(field index)" +MTIME="$(field mtime)" +REQUESTER="$(field requester)" + +log "Requested by: ${REQUESTER:-unknown} (action=${ACTION:-restore})" + +[[ "$ACTION" == "restore" || "$ACTION" == "undo" ]] || die "bad action '${ACTION}'" + +# --------------------------------------------------------------------------- +# 2. Resolve the source -- entirely from our own filesystem, never from the +# trigger. The trigger only ever gets to *describe* a target. +# --------------------------------------------------------------------------- +SOURCE_DESC="" +SRC_ZIP="" +UNDO_DIR="" + +if [[ "$ACTION" == "restore" ]]; then + [[ "$SET" == "period" || "$SET" == "startup" || "$SET" == "version" ]] \ + || die "bad backup set '${SET}'" + [[ "$MTIME" =~ ^[0-9]+$ ]] || die "bad mtime '${MTIME}'" + + # Resolve by mtime, NOT by index. PZ rotates these every BackupsPeriod + # minutes -- backup_7.zip becomes backup_8.zip and so on -- so the index the + # bot showed a human 90 seconds ago may already point at a different world. + # The mtime is the only stable identity a PZ backup has. + for f in "${BACKUPS}/period"/backup_*.zip "${BACKUPS}/startup"/backup_*.zip "${BACKUPS}/version"/backup_*.zip; do + podman unshare test -f "$f" || continue + base="$(basename "$f")" + [[ "$base" =~ ^backup_[0-9]+\.zip$ ]] || continue + m="$(podman unshare stat -c %Y "$f" 2>/dev/null || echo 0)" + if [[ "$m" == "$MTIME" ]]; then + SRC_ZIP="$f" + break + fi + done + + [[ -n "$SRC_ZIP" ]] || die "backup from $(date -d "@${MTIME}" '+%Y-%m-%d %H:%M:%S' 2>/dev/null || echo "$MTIME") has rotated out -- run the list again" + + podman unshare unzip -l "$SRC_ZIP" "Saves/Multiplayer/${SERVER_NAME}/*" >/dev/null 2>&1 \ + || die "archive $(basename "$SRC_ZIP") has no ${SERVER_NAME} world in it" + + SOURCE_DESC="$(basename "$SRC_ZIP") (${SET}, $(date -d "@${MTIME}" '+%Y-%m-%d %H:%M:%S' 2>/dev/null || echo "$MTIME"))" + log "Resolved target: $SRC_ZIP -> $SOURCE_DESC" +else + # Newest snapshot that actually holds a world. + for d in $(ls -1dt "${SNAP_ROOT}"/*/ 2>/dev/null); do + if podman unshare test -d "${d}${SERVER_NAME}"; then + UNDO_DIR="${d%/}" + break + fi + done + [[ -n "$UNDO_DIR" ]] || die "no snapshot to undo to" + SOURCE_DESC="snapshot $(basename "$UNDO_DIR")" + log "Resolved undo target: $UNDO_DIR" +fi + +# --------------------------------------------------------------------------- +# 3. Stop the server. Nothing above this line touches it, so every failure mode +# up to here leaves a running world completely alone. +# --------------------------------------------------------------------------- +# Disarm the wipe trigger for the duration. A reset landing midway through a +# restore would delete the half-restored world and leave nothing coherent. +log "Disarming world-reset path unit" +systemctl --user stop zomboid-world-reset.path || true + +log "Stopping zomboid service..." +systemctl --user stop zomboid.service || true +sleep 5 + +# --------------------------------------------------------------------------- +# 4. Move the live world aside. Never delete -- this is the undo point. +# --------------------------------------------------------------------------- +SNAP_DIR="${SNAP_ROOT}/$(date '+%Y-%m-%d_%H-%M-%S')" +mkdir -p "$SNAP_DIR" 2>/dev/null || die "could not create $SNAP_DIR" + +if podman unshare test -d "$SAVES_PATH"; then + podman unshare mv "$SAVES_PATH" "${SNAP_DIR}/${SERVER_NAME}" + log "Live world moved to $(basename "$SNAP_DIR")" +fi +if podman unshare test -f "$DB_PATH"; then + podman unshare cp "$DB_PATH" "${SNAP_DIR}/${SERVER_NAME}.db" +fi + +# --------------------------------------------------------------------------- +# 5. Put the target in place +# --------------------------------------------------------------------------- +podman unshare rm -rf "$STAGING" +podman unshare mkdir -p "$STAGING" + +if [[ "$ACTION" == "restore" ]]; then + log "Extracting ${SOURCE_DESC}..." + podman unshare unzip -q "$SRC_ZIP" \ + "Saves/Multiplayer/${SERVER_NAME}/*" "db/${SERVER_NAME}.db" -d "$STAGING" \ + || die "extract failed" + podman unshare mv "${STAGING}/Saves/Multiplayer/${SERVER_NAME}" "$SAVES_PATH" + if podman unshare test -f "${STAGING}/db/${SERVER_NAME}.db"; then + podman unshare mv "${STAGING}/db/${SERVER_NAME}.db" "$DB_PATH" + fi + + # PZ's own backups do not capture everything under the save directory -- + # mod state such as blam/ is excluded from every archive it writes. Those + # files were equally present at the moment we are restoring to, so dropping + # them would land the world slightly *behind* the target rather than on it. + # Carry across anything the snapshot had that the archive did not. + for entry in $(podman unshare ls -1 "${SNAP_DIR}/${SERVER_NAME}" 2>/dev/null); do + if ! podman unshare test -e "${SAVES_PATH}/${entry}"; then + podman unshare cp -a "${SNAP_DIR}/${SERVER_NAME}/${entry}" "${SAVES_PATH}/${entry}" 2>/dev/null || true + log "Carried forward un-backed-up entry: ${entry}" + fi + done +else + log "Restoring ${SOURCE_DESC}..." + podman unshare cp -a "${UNDO_DIR}/${SERVER_NAME}" "$SAVES_PATH" + if podman unshare test -f "${UNDO_DIR}/${SERVER_NAME}.db"; then + podman unshare cp -a "${UNDO_DIR}/${SERVER_NAME}.db" "$DB_PATH" + fi +fi + +podman unshare rm -rf "$STAGING" + +# --------------------------------------------------------------------------- +# 6. Ownership, permissions, labels +# --------------------------------------------------------------------------- +# Container uid 1000; on the host that lands on the subuid base + 999. Same call +# zomboid.yml makes after seeding config. +log "Fixing ownership and permissions..." +podman unshare chown -R 1000:1000 "$SAVES_PATH" "$DB_PATH" +podman unshare chmod -R u+rwX,g+rwX "$SAVES_PATH" +podman unshare chmod 0664 "$DB_PATH" + +# Targeted, and -x so it can never wander off this filesystem. The role-wide +# restorecon handler exists in -x form for a reason; see roles/podman/handlers. +if command -v restorecon >/dev/null 2>&1; then + restorecon -Frx "$SAVES_PATH" "$DB_PATH" 2>/dev/null || true +fi + +# --------------------------------------------------------------------------- +# 7. Prune old snapshots +# --------------------------------------------------------------------------- +for old in $(ls -1dt "${SNAP_ROOT}"/*/ 2>/dev/null | tail -n +$((KEEP_SNAPSHOTS + 1))); do + podman unshare rm -rf "$old" || true + log "Pruned old snapshot $(basename "${old%/}")" +done + +# --------------------------------------------------------------------------- +# 8. Back up +# --------------------------------------------------------------------------- +log "Starting zomboid service..." +systemctl --user start zomboid.service +systemctl --user start zomboid-world-reset.path || true + +log "Restore complete: ${SOURCE_DESC}" +result true "Restored ${SOURCE_DESC}. Previous world kept as $(basename "$SNAP_DIR")."