#!/bin/bash # {{ ansible_managed }} # {{ backup_product | default('Nextcloud') }} "{{ backup_name }}" -> truenas.localdomain. # # Ordering is deliberate: the database is dumped BEFORE the file tree is # synced. A DB snapshot slightly OLDER than the files degrades to "files # the app has not indexed yet" and is repaired with a rescan. A DB snapshot # NEWER than the files references blobs that never made it into the backup, # which surfaces as broken shares and dead file entries on restore. set -euo pipefail # Tag stays "nextcloud-backup" for EVERY instance, Gitea included: an external # Graylog rule matches status=failed on this tag, and renaming it here would # silently stop alerting for all of them. Misnomer retained deliberately. TAG=nextcloud-backup INSTANCE={{ backup_name }} STAGE={{ backup_stage_path | default('/var/backups/nextcloud/' ~ backup_name) }} KEEP={{ backup_db_keep | default(7) }} DUMP="$STAGE/db/${INSTANCE}-$(date +%Y%m%d).sql.gz" log() { logger -t "$TAG" -p daemon.info -- "instance=$INSTANCE $*"; echo "$TAG: $*"; } fail() { logger -t "$TAG" -p daemon.err -- "instance=$INSTANCE status=failed $*" echo "$TAG: FAILED: $*" >&2; exit 1; } SSH="ssh -i {{ ssh_key_path }} -o StrictHostKeyChecking=accept-new -o ServerAliveInterval=30 -o ServerAliveCountMax=6" DEST={{ ssh_user }}@truenas.localdomain log "status=start" {% if db_container | default('') %} # ------------------------------------------------------------ 1. database # The containers are ROOTLESS podman owned by # "{{ backup_podman_user | default(podman_user) }}" -- Nextcloud runs under # {{ podman_user }}, Gitea under its own git user -- but this script runs as # root under systemd. Every podman call therefore goes through sudo: # -H HOME becomes the podman user's home, so podman finds its rootless # graph root under ~/.local/share/containers # cd; required preamble (see CLAUDE.md) so the shell starts in that home # XDG_RUNTIME_DIR the podman user's runtime dir. Lingering is enabled by # roles/podman/tasks/podman/podman.yml so /run/user/ exists; # guarded anyway so podman falls back cleanly if it ever does not. # # No credential is stored in this file or placed on a host command line: # $MYSQL_ROOT_PASSWORD / $MYSQL_DATABASE (or $POSTGRES_* for postgres) are # expanded by the shell INSIDE the database container, which already carries # them in its environment. {% if backup_db_type | default('mariadb') != 'postgres' %} # # MariaDB-specific history, for the flag set below: # --routines and --events initially aborted the dump here: mysql.proc read as # corrupted (error 1728) and the event scheduler reported disabled (1577), # both artefacts of an image bump without mariadb-upgrade. `mariadb-upgrade # --force` has since been run against both instances and repaired the system # tables, so the full flag set works and is kept for completeness. # # Note that upgrade still exits non-zero on these containers: it cannot # install the `sys` schema because the datadir root (/var/lib/mysql) is owned # by daemon rather than mysql, so mysqld may not create new top-level # databases. `sys` is purely diagnostic and unused by Nextcloud, so this is # cosmetic -- but it does mean creating a NEW database would fail too. {% else %} # # --clean --if-exists makes the dump replayable into an existing database # (`psql -U gitea gitea < dump.sql`) rather than only into an empty one, and # --no-owner keeps it restorable when the target role names differ. {% endif %} pexec() { sudo -H -u {{ backup_podman_user | default(podman_user) }} bash -c \ 'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d" exec podman "$@"' _ "$@" } install -d -m 0700 "$STAGE" "$STAGE/db" tmp="$DUMP.tmp" rm -f "$tmp" log "dumping {{ db_container }}" set +e {% if backup_db_type | default('mariadb') == 'postgres' %} pexec exec {{ db_container }} sh -c ' exec env PGPASSWORD="$POSTGRES_PASSWORD" pg_dump -U "$POSTGRES_USER" \ --no-owner --clean --if-exists "$POSTGRES_DB" ' | gzip -6 > "$tmp" {% elif backup_db_type | default('mariadb') == 'mysql' %} {# Upstream mysql:5.7 ships mysqldump, NOT the mariadb-dump alias that only appears in MariaDB 10.5+. Flags and completion trailer are identical to the MariaDB branch; 5.7.21 was verified to accept --no-tablespaces. #} pexec exec {{ db_container }} sh -c ' exec env MYSQL_PWD="$MYSQL_ROOT_PASSWORD" mysqldump -u root \ --single-transaction --quick --routines --events --triggers \ --no-tablespaces --default-character-set=utf8mb4 "$MYSQL_DATABASE" ' | gzip -6 > "$tmp" {% else %} pexec exec {{ db_container }} sh -c ' exec env MYSQL_PWD="$MYSQL_ROOT_PASSWORD" mariadb-dump -u root \ --single-transaction --quick --routines --events --triggers \ --no-tablespaces --default-character-set=utf8mb4 "$MYSQL_DATABASE" ' | gzip -6 > "$tmp" {% endif %} dump_rc=${PIPESTATUS[0]} set -e [ "$dump_rc" -eq 0 ] || fail "{% if backup_db_type | default('mariadb') == 'postgres' %}pg_dump{% elif backup_db_type | default('mariadb') == 'mysql' %}mysqldump{% else %}mariadb-dump{% endif %} {{ db_container }} exited $dump_rc" # A new dump is promoted over yesterday's only after it proves complete: a # valid gzip stream AND the trailer the dump tool writes only on a clean # finish. The two engines word it differently, so the string is per-type -- # grepping for the MariaDB one against a pg_dump would fail every run. # `mv` is atomic within the staging filesystem, so a failed or truncated run # can never replace a good dump. gzip -t "$tmp" || fail "dump is not a valid gzip stream" gunzip -c "$tmp" | tail -c 512 \ | grep -q '{{ "PostgreSQL database dump complete" if backup_db_type | default("mariadb") == "postgres" else "Dump completed" }}' \ || fail "dump is truncated (no completion trailer)" mv -f "$tmp" "$DUMP" log "db_dump=ok bytes=$(stat -c %s "$DUMP")" # Local retention. The staging sync below mirrors with --delete, so remote # retention follows the same window; deeper history comes from the TrueNAS # periodic ZFS snapshots (see roles/podman/README.md). ls -1t "$STAGE"/db/"$INSTANCE"-*.sql.gz | tail -n +$((KEEP + 1)) | xargs -r rm -f {% endif %} {% if backup_sqlite_dbs | default([]) %} # ----------------------------------------------------- 1b. sqlite snapshots # These databases are WAL-mode SQLite written by a LIVE process, so neither # rsync option is correct: skipping the -wal loses committed transactions, # and copying .db + -wal together can catch a checkpoint mid-write. Only # `.backup` takes a consistent snapshot of a database that is being written. # # Run on the HOST, not in the container: the files live on the host # filesystem and sqlite3 is installed there, so this needs no podman at all # and works even when the owning container is stopped. install -d -m 0700 "$STAGE" "$STAGE/db" {% for db in backup_sqlite_dbs %} sqlite_out="$STAGE/db/{{ db | basename | regex_replace('\\.db$', '') }}.sqlite" if [ ! -f "{{ db }}" ]; then fail "sqlite source missing: {{ db }}" fi sqlite3 "{{ db }}" ".backup '$sqlite_out.tmp'" \ || fail "sqlite3 .backup failed for {{ db }}" # Prove the snapshot is a loadable database before it replaces yesterday's, # mirroring the gzip/trailer gate the SQL dumps get above. sqlite3 "$sqlite_out.tmp" 'pragma integrity_check;' | grep -qx ok \ || fail "sqlite snapshot failed integrity_check: {{ db }}" mv -f "$sqlite_out.tmp" "$sqlite_out" log "sqlite_dump=ok db={{ db | basename }} bytes=$(stat -c %s "$sqlite_out")" {% endfor %} {% endif %} # ----------------------------------------------------------- 2. file tree # --exclude .ssh is load-bearing: {{ remote_path }} IS {{ ssh_user }}'s home # on TrueNAS and its authorized_keys lives there, so a `.ssh` directory # appearing in the data tree must never be shipped. No --delete here: the # data tree is append-mostly and a source-side mishap must not propagate. log "syncing data" rsync -az --timeout=1800 --exclude .ssh \ {{ backup_rsync_excludes | default("--exclude '/nextcloud.log*' --exclude '/updater.log'") }} \ {{ backup_rsync_extra_args | default('') }} \ -e "$SSH" {{ data_path }}/ "$DEST:{{ remote_path }}/" # ------------------------------------------- 3. config/ and database dumps {# --mkpath creates the nested _backup// destination; rsync will not build more than one missing level on its own. Requires rsync >= 3.2.3 on both ends (galactica 3.4.1, truenas 3.2.7). #} {% if config_path | default('') %} log "syncing config" rsync -az --timeout=600 --delete --mkpath {{ backup_rsync_extra_args | default('') }} \ -e "$SSH" {{ config_path }}/ "$DEST:{{ remote_path }}/_backup/config/" {% endif %} {% if db_container | default('') or backup_sqlite_dbs | default([]) %} log "syncing db dumps" rsync -az --timeout=600 --delete --mkpath {{ backup_rsync_extra_args | default('') }} \ -e "$SSH" "$STAGE/db/" "$DEST:{{ remote_path }}/_backup/db/" {% endif %} log "status=ok"