Rolling a world back meant hand-work over ssh: stop the service, move the live save aside, unzip the right archive, chown into the container's subuid range, relabel, start. That is the wrong shape of task to do by hand, and it is always done under time pressure -- by construction, because the archive you want is being deleted while you work. PZ keeps BackupsCount=10 per set and writes one every BackupsPeriod=30 minutes, so a periodic backup is reachable for about five hours and then gone. On 2026-09-05 the snapshot the admins asked for (05:16, four minutes before the incident) had about 90 minutes of life left when the request came in. Same shape as the wipe: the Discord bot writes a trigger file into its own rw volume, zomboid-restore.path notices it, and zomboid-restore.service runs the script as the podman user. The bot gets no ssh, no systemd, and keeps only its existing read-only mount of the Zomboid volume. Two details carry most of the correctness. Resolution is by mtime, not by index. The rotation renames the files -- today's backup_7.zip is backup_8.zip half an hour from now, and a new backup_7.zip holds a different world -- so an index is valid only while the listing is fresh, which is not long enough to survive a human reading a confirmation prompt. The trigger names a set and an mtime; the script resolves the path itself, whitelists the filename, and refuses if nothing matches. It never accepts a path. Everything that can fail is checked before the server is touched. A rotated-out target, an archive with no debbzoid world in it, a bad action, a traversal attempt in the set name: each aborts with the server still running and writes a result file the bot reports back. The live world is moved aside rather than deleted, so a restore is undoable and the last three are kept. One thing PZ does not advertise: its backups do not cover the whole save directory. blam/, a mod's own state, is in none of them -- not the 05:16 archive and not the newest one. Restoring only what the archive holds therefore lands the world slightly *behind* the target rather than on it, so anything present in the displaced world and absent from the archive is carried across. The gregtime tag moves to 3.17.0 for the bot half of this -- `backups`, `restore <n>`, `restore confirm`, `restore undo`, gated to the same two admins as the wipe. That image is built and running on the host; its source is not committed yet. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016QdWYhwUtwM2NQGukiRh12
260 lines
9.6 KiB
YAML
260 lines
9.6 KiB
YAML
---
|
|
- import_tasks: firewall.yml
|
|
- import_tasks: podman/podman.yml
|
|
|
|
- import_tasks: podman/podman-prune.yml
|
|
tags: podman-prune
|
|
|
|
# Heals the TrueNAS CIFS mounts and bounces immich when they return.
|
|
# Must be deployable independently of the photos containers.
|
|
- import_tasks: podman/cifs-watchdog.yml
|
|
tags: cifs-watchdog, photos
|
|
|
|
# WEB SERVER: Caddy is the default and only web server
|
|
# nginx has been completely replaced and removed
|
|
|
|
# ===== WEB SERVER CONFIGURATION =====
|
|
# Caddy is the default web server
|
|
- import_tasks: containers/base/conf-caddy.yml
|
|
tags:
|
|
- caddy
|
|
- web
|
|
|
|
- import_tasks: containers/base/caddy.yml
|
|
vars:
|
|
image: docker.io/library/caddy:2.11.2
|
|
tags:
|
|
- caddy
|
|
- web
|
|
|
|
# nginx cleanup completed - infrastructure removed
|
|
|
|
|
|
- import_tasks: containers/base/awsddns.yml
|
|
vars:
|
|
image: docker.io/bdebyl/awsddns:1.0.34
|
|
tags: ddns
|
|
|
|
# Drone CI infrastructure completely removed
|
|
|
|
- import_tasks: containers/home/hass.yml
|
|
vars:
|
|
image: ghcr.io/home-assistant/home-assistant:2026.8.3
|
|
tags: hass
|
|
|
|
- import_tasks: containers/home/partsy.yml
|
|
vars:
|
|
image: "git.debyl.io/debyltech/partsy:latest"
|
|
tags: partsy
|
|
|
|
- import_tasks: containers/skudak/partsy.yml
|
|
vars:
|
|
image: "git.debyl.io/debyltech/partsy:latest"
|
|
tags: skudak, partsy-skudak
|
|
|
|
- import_tasks: containers/skudak/wiki.yml
|
|
vars:
|
|
db_image: docker.io/library/mysql:5.7.21
|
|
image: docker.io/solidnerd/bookstack:26.5.4
|
|
tags: skudak, skudak-wiki
|
|
|
|
- import_tasks: containers/home/photos.yml
|
|
vars:
|
|
db_image: ghcr.io/immich-app/postgres:14-vectorchord0.4.3-pgvectors0.2.0@sha256:bcf63357191b76a916ae5eb93464d65c07511da41e3bf7a8416db519b40b1c23
|
|
ml_image: ghcr.io/immich-app/immich-machine-learning:v3.1.0
|
|
redis_image: docker.io/redis:6.2-alpine@sha256:eaba718fecd1196d88533de7ba49bf903ad33664a92debb24660a922ecd9cac8
|
|
image: ghcr.io/immich-app/immich-server:v3.1.0
|
|
tags: photos
|
|
|
|
- import_tasks: containers/home/cloud.yml
|
|
vars:
|
|
db_image: docker.io/library/mariadb:10.6
|
|
image: docker.io/library/nextcloud:34.0.3-apache
|
|
tags: cloud
|
|
|
|
- import_tasks: containers/skudak/cloud.yml
|
|
vars:
|
|
db_image: docker.io/library/mariadb:10.6
|
|
redis_image: docker.io/redis:8.2-alpine
|
|
image: docker.io/library/nextcloud:34.0.3-apache
|
|
tags: skudak, skudak-cloud
|
|
|
|
- import_tasks: containers/debyltech/fulfillr.yml
|
|
vars:
|
|
image: git.debyl.io/debyltech/fulfillr:20260827.2009
|
|
tags: debyltech, fulfillr
|
|
|
|
# Staging back-office (fulfillr-dev.debyltech.com) — same image, staging Turso config.
|
|
- import_tasks: containers/debyltech/fulfillr-dev.yml
|
|
vars:
|
|
image: git.debyl.io/debyltech/fulfillr:20260827.2009
|
|
tags: debyltech, fulfillr-dev
|
|
|
|
- import_tasks: containers/debyltech/uptime-kuma.yml
|
|
vars:
|
|
image: docker.io/louislam/uptime-kuma:2.3.2
|
|
tags: debyltech, uptime-debyltech
|
|
|
|
- import_tasks: containers/home/uptime-kuma.yml
|
|
vars:
|
|
image: docker.io/louislam/uptime-kuma:2.3.2
|
|
tags: home, uptime
|
|
|
|
# GeoIP is only consumed by Graylog's enrichment pipelines, so it follows the
|
|
# same switch -- see graylog_enabled in inventories/home/hosts.yml.
|
|
- import_tasks: data/geoip.yml
|
|
when: graylog_enabled | bool
|
|
tags: graylog, geoip
|
|
|
|
- import_tasks: containers/debyltech/graylog.yml
|
|
vars:
|
|
mongo_image: docker.io/mongo:7.0
|
|
opensearch_image: docker.io/opensearchproject/opensearch:2
|
|
image: docker.io/graylog/graylog:7.0.1
|
|
when: graylog_enabled | bool
|
|
tags: debyltech, graylog
|
|
|
|
# The disabled path is an active teardown, not just a skipped create: without it
|
|
# the containers already on the host keep running and their systemd user units
|
|
# keep restarting them at boot.
|
|
- import_tasks: containers/debyltech/graylog-teardown.yml
|
|
when: not (graylog_enabled | bool)
|
|
tags: debyltech, graylog
|
|
|
|
- import_tasks: containers/home/gregtime.yml
|
|
vars:
|
|
image: localhost/greg-time-bot:3.17.1
|
|
tags: gregtime
|
|
|
|
# Gated on zomboid_enabled (roles/podman/defaults/main.yml) so it can be taken
|
|
# down for a CI-heavy stretch without losing the world.
|
|
- import_tasks: containers/home/zomboid.yml
|
|
vars:
|
|
image: docker.io/cm2network/steamcmd:root
|
|
when: zomboid_enabled | bool
|
|
tags: zomboid
|
|
|
|
# ---------------------------------------------------------- Gitea backups
|
|
# The Gitea pods themselves are owned by roles/git, but the backup machinery
|
|
# (containers/cloud-backup.yml plus templates/nextcloud/*) lives here, and an
|
|
# include_tasks reaching across roles would need a path outside the role. So
|
|
# the two Gitea backup instances are wired here alongside the container ones.
|
|
#
|
|
# Both run as ROOTLESS podman under "{{ git_user }}", not "{{ podman_user }}"
|
|
# -- hence backup_podman_user -- and use PostgreSQL rather than MariaDB.
|
|
#
|
|
# Scheduled ahead of the 04:00/04:30 Nextcloud runs so everything lands before
|
|
# the 05:00 TrueNAS snapshot.
|
|
# `apply` is required: tags on a dynamic include_tasks select whether the
|
|
# include runs, but do NOT propagate to the tasks inside it, so without this
|
|
# `make deploy TAGS=gitea-backup` includes the file and then filters out every
|
|
# task in it. The Nextcloud instances avoid this only because they are reached
|
|
# through a static import_tasks chain that tags their children at parse time.
|
|
- include_tasks:
|
|
file: containers/cloud-backup.yml
|
|
apply:
|
|
tags: gitea-backup
|
|
vars:
|
|
backup_name: gitea-debyl
|
|
backup_product: Gitea
|
|
backup_podman_user: "{{ git_user }}"
|
|
data_path: "{{ git_home }}/volumes/gitea/data"
|
|
db_container: gitea-debyl-postgres
|
|
backup_db_type: postgres
|
|
ssh_key_path: /etc/ssh/backup_keys/gitea
|
|
ssh_key_content: "{{ gitea_backup_ssh_key }}"
|
|
ssh_user: gitea
|
|
remote_path: /mnt/glacier/gitea
|
|
script_path: /usr/local/bin/gitea-backup.sh
|
|
# actions_log/artifacts are CI churn (510 MB and growing) and rebuildable;
|
|
# indexers/queues/tmp are derived state Gitea recreates on start. Note the
|
|
# default excludes are Nextcloud-specific, so this must be set explicitly.
|
|
backup_rsync_excludes: >-
|
|
--exclude '/gitea/actions_log' --exclude '/gitea/actions_artifacts'
|
|
--exclude '/gitea/tmp' --exclude '/gitea/indexers' --exclude '/gitea/queues'
|
|
backup_oncalendar: "*-*-* 03:30:00"
|
|
tags: gitea-backup
|
|
|
|
# ------------------------------------------------- Skudak app-data backups
|
|
# BookStack (wiki.skudak.com) and partsy-skudak are BUSINESS data. Both share
|
|
# one TrueNAS dataset (/mnt/glacier/skudakapps) so they need only one backup
|
|
# user, key and cloud-sync task between them; the personal "iDrive E2 Backup"
|
|
# task excludes /skudakapps/** and Skudak's own task pushes it to backup-all.
|
|
#
|
|
# Scheduled ahead of the 03:30+ Gitea/Nextcloud jobs and the 05:00 snapshot.
|
|
- include_tasks:
|
|
file: containers/cloud-backup.yml
|
|
apply:
|
|
tags: [skudak, skudak-apps-backup]
|
|
vars:
|
|
backup_name: bookstack
|
|
backup_product: BookStack
|
|
data_path: "{{ bookstack_path }}"
|
|
db_container: bookstack-db
|
|
# mysql:5.7 predates the mariadb-dump alias -- see cloud-backup.sh.j2.
|
|
backup_db_type: mysql
|
|
ssh_key_path: /etc/ssh/backup_keys/skudakapps
|
|
ssh_key_content: "{{ skudakapps_backup_ssh_key }}"
|
|
ssh_user: skudakapps
|
|
remote_path: /mnt/glacier/skudakapps/bookstack
|
|
script_path: /usr/local/bin/bookstack-backup.sh
|
|
# The wiki content is the DATABASE; /mysql is its raw datadir, which must
|
|
# not be rsynced live -- the dump above is the consistent copy. public/
|
|
# and storage/ hold the uploads and are the only file trees worth shipping.
|
|
backup_rsync_excludes: "--exclude '/mysql'"
|
|
backup_oncalendar: "*-*-* 03:00:00"
|
|
tags: skudak, skudak-apps-backup
|
|
|
|
- include_tasks:
|
|
file: containers/cloud-backup.yml
|
|
apply:
|
|
tags: [skudak, skudak-apps-backup]
|
|
vars:
|
|
backup_name: partsy-skudak
|
|
backup_product: Partsy
|
|
data_path: "{{ partsy_skudak_path }}"
|
|
# Live WAL-mode SQLite: snapshotted via `sqlite3 .backup` rather than
|
|
# rsynced, so the -wal/-shm sidecars are deliberately excluded from the
|
|
# file tree -- shipping them alongside a separately-taken snapshot would
|
|
# only invite a confusing restore.
|
|
backup_sqlite_dbs:
|
|
- "{{ partsy_skudak_path }}/data/partsy.db"
|
|
backup_rsync_excludes: "--exclude '*-wal' --exclude '*-shm'"
|
|
ssh_key_path: /etc/ssh/backup_keys/skudakapps
|
|
ssh_key_content: "{{ skudakapps_backup_ssh_key }}"
|
|
ssh_user: skudakapps
|
|
remote_path: /mnt/glacier/skudakapps/partsy-skudak
|
|
script_path: /usr/local/bin/partsy-skudak-backup.sh
|
|
backup_oncalendar: "*-*-* 03:10:00"
|
|
tags: skudak, skudak-apps-backup
|
|
|
|
# BUSINESS data. Rsynced to TrueNAS here, then pushed offsite to SKUDAK'S OWN
|
|
# iDrive e2 account (bucket `backup-all`) by TrueNAS cloud-sync task "Skudak
|
|
# iDrive - Gitea" (id 9, /mnt/glacier/skudakgit -> /skudakgit, daily 06:30).
|
|
#
|
|
# The personal "iDrive E2 Backup" task's /skudakgit/** exclude is PERMANENT:
|
|
# it is what keeps business data out of personal storage. Do not remove it --
|
|
# Skudak has its own task and bucket instead.
|
|
- include_tasks:
|
|
file: containers/cloud-backup.yml
|
|
apply:
|
|
tags: [skudak, gitea-backup-skudak]
|
|
vars:
|
|
backup_name: skudak-gitea
|
|
backup_product: Gitea
|
|
backup_podman_user: "{{ git_user }}"
|
|
data_path: "{{ git_home }}/volumes/gitea-skudak/data"
|
|
db_container: gitea-skudak-postgres
|
|
backup_db_type: postgres
|
|
ssh_key_path: /etc/ssh/backup_keys/skudak-gitea
|
|
ssh_key_content: "{{ skudakgit_backup_ssh_key }}"
|
|
ssh_user: skudakgit
|
|
remote_path: /mnt/glacier/skudakgit
|
|
script_path: /usr/local/bin/skudak-gitea-backup.sh
|
|
backup_rsync_excludes: >-
|
|
--exclude '/gitea/actions_log' --exclude '/gitea/actions_artifacts'
|
|
--exclude '/gitea/tmp' --exclude '/gitea/indexers' --exclude '/gitea/queues'
|
|
backup_oncalendar: "*-*-* 03:45:00"
|
|
tags: skudak, gitea-backup-skudak
|
|
|