6e99794d0f
mariadb-upgrade --force has now been run against both instances and repaired the system tables, so --routines and --events no longer abort the dump. Restore them for completeness. Both verified: rc=0 with a clean completion trailer, and both Nextcloud instances report installed with unchanged table counts afterwards. The upgrade still exits non-zero on these containers because it cannot create the `sys` schema: /var/lib/mysql is owned by daemon rather than mysql, so mysqld may not create top-level databases. `sys` is diagnostic only and unused by Nextcloud, but the same permission would block creating any new database, so it is recorded in the template comment. The alert handler claimed FAILED in its subject line regardless of what the unit actually reported, so starting it by hand mailed out a failure notice for a run that succeeded. Derive the subject and log line from the real Result, and emit status=spurious rather than status=failed so such triggers cannot match a Graylog alert rule keyed on genuine failures. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
118 lines
5.5 KiB
Django/Jinja
118 lines
5.5 KiB
Django/Jinja
#!/bin/bash
|
|
# {{ ansible_managed }}
|
|
# Nextcloud "{{ backup_name }}" -> truenas.localdomain.
|
|
#
|
|
# Ordering is deliberate: the database is dumped BEFORE the file tree is
|
|
# synced. A DB snapshot slightly OLDER than the files degrades to "files
|
|
# Nextcloud has not indexed yet" and is repaired with `occ files:scan`. A DB
|
|
# snapshot NEWER than the files references blobs that never made it into the
|
|
# backup, which surfaces as broken shares and dead file entries on restore.
|
|
set -euo pipefail
|
|
|
|
TAG=nextcloud-backup
|
|
INSTANCE={{ backup_name }}
|
|
STAGE={{ backup_stage_path | default('/var/backups/nextcloud/' ~ backup_name) }}
|
|
KEEP={{ backup_db_keep | default(7) }}
|
|
DUMP="$STAGE/db/${INSTANCE}-$(date +%Y%m%d).sql.gz"
|
|
|
|
log() { logger -t "$TAG" -p daemon.info -- "instance=$INSTANCE $*"; echo "$TAG: $*"; }
|
|
fail() { logger -t "$TAG" -p daemon.err -- "instance=$INSTANCE status=failed $*"
|
|
echo "$TAG: FAILED: $*" >&2; exit 1; }
|
|
|
|
SSH="ssh -i {{ ssh_key_path }} -o StrictHostKeyChecking=accept-new -o ServerAliveInterval=30 -o ServerAliveCountMax=6"
|
|
DEST={{ ssh_user }}@truenas.localdomain
|
|
|
|
log "status=start"
|
|
|
|
{% if db_container | default('') %}
|
|
# ------------------------------------------------------------ 1. database
|
|
# The Nextcloud containers are ROOTLESS podman owned by "{{ podman_user }}",
|
|
# but this script runs as root under systemd. Every podman call therefore
|
|
# goes through sudo:
|
|
# -H HOME becomes the podman user's home, so podman finds its rootless
|
|
# graph root under ~/.local/share/containers
|
|
# cd; required preamble (see CLAUDE.md) so the shell starts in that home
|
|
# XDG_RUNTIME_DIR the podman user's runtime dir. Lingering is enabled by
|
|
# roles/podman/tasks/podman/podman.yml so /run/user/<uid> exists;
|
|
# guarded anyway so podman falls back cleanly if it ever does not.
|
|
#
|
|
# No credential is stored in this file or placed on a host command line:
|
|
# $MYSQL_ROOT_PASSWORD and $MYSQL_DATABASE are expanded by the shell INSIDE
|
|
# the database container, which already carries them in its environment.
|
|
#
|
|
# --routines and --events initially aborted the dump here: mysql.proc read as
|
|
# corrupted (error 1728) and the event scheduler reported disabled (1577),
|
|
# both artefacts of an image bump without mariadb-upgrade. `mariadb-upgrade
|
|
# --force` has since been run against both instances and repaired the system
|
|
# tables, so the full flag set works and is kept for completeness.
|
|
#
|
|
# Note that upgrade still exits non-zero on these containers: it cannot
|
|
# install the `sys` schema because the datadir root (/var/lib/mysql) is owned
|
|
# by daemon rather than mysql, so mysqld may not create new top-level
|
|
# databases. `sys` is purely diagnostic and unused by Nextcloud, so this is
|
|
# cosmetic -- but it does mean creating a NEW database would fail too.
|
|
pexec() {
|
|
sudo -H -u {{ podman_user }} bash -c \
|
|
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
|
|
exec podman "$@"' _ "$@"
|
|
}
|
|
|
|
install -d -m 0700 "$STAGE" "$STAGE/db"
|
|
tmp="$DUMP.tmp"
|
|
rm -f "$tmp"
|
|
|
|
log "dumping {{ db_container }}"
|
|
set +e
|
|
pexec exec {{ db_container }} sh -c '
|
|
exec env MYSQL_PWD="$MYSQL_ROOT_PASSWORD" mariadb-dump -u root \
|
|
--single-transaction --quick --routines --events --triggers \
|
|
--no-tablespaces --default-character-set=utf8mb4 "$MYSQL_DATABASE"
|
|
' | gzip -6 > "$tmp"
|
|
dump_rc=${PIPESTATUS[0]}
|
|
set -e
|
|
[ "$dump_rc" -eq 0 ] || fail "mariadb-dump {{ db_container }} exited $dump_rc"
|
|
|
|
# A new dump is promoted over yesterday's only after it proves complete:
|
|
# a valid gzip stream AND the "-- Dump completed" trailer that mariadb-dump
|
|
# writes only on a clean finish. `mv` is atomic within the staging
|
|
# filesystem, so a failed or truncated run can never replace a good dump.
|
|
gzip -t "$tmp" || fail "dump is not a valid gzip stream"
|
|
gunzip -c "$tmp" | tail -c 512 | grep -q 'Dump completed' \
|
|
|| fail "dump is truncated (no completion trailer)"
|
|
mv -f "$tmp" "$DUMP"
|
|
log "db_dump=ok bytes=$(stat -c %s "$DUMP")"
|
|
|
|
# Local retention. The staging sync below mirrors with --delete, so remote
|
|
# retention follows the same window; deeper history comes from the TrueNAS
|
|
# periodic ZFS snapshots (see roles/podman/README.md).
|
|
ls -1t "$STAGE"/db/"$INSTANCE"-*.sql.gz | tail -n +$((KEEP + 1)) | xargs -r rm -f
|
|
{% endif %}
|
|
|
|
# ----------------------------------------------------------- 2. file tree
|
|
# --exclude .ssh is load-bearing: {{ remote_path }} IS {{ ssh_user }}'s home
|
|
# on TrueNAS and its authorized_keys lives there, so a `.ssh` directory
|
|
# appearing in the data tree must never be shipped. No --delete here: the
|
|
# data tree is append-mostly and a source-side mishap must not propagate.
|
|
log "syncing data"
|
|
rsync -az --timeout=1800 --exclude .ssh \
|
|
{{ backup_rsync_excludes | default("--exclude '/nextcloud.log*' --exclude '/updater.log'") }} \
|
|
{{ backup_rsync_extra_args | default('') }} \
|
|
-e "$SSH" {{ data_path }}/ "$DEST:{{ remote_path }}/"
|
|
|
|
# ------------------------------------------- 3. config/ and database dumps
|
|
{# --mkpath creates the nested _backup/<x>/ destination; rsync will not build
|
|
more than one missing level on its own. Requires rsync >= 3.2.3 on both
|
|
ends (galactica 3.4.1, truenas 3.2.7). #}
|
|
{% if config_path | default('') %}
|
|
log "syncing config"
|
|
rsync -az --timeout=600 --delete --mkpath {{ backup_rsync_extra_args | default('') }} \
|
|
-e "$SSH" {{ config_path }}/ "$DEST:{{ remote_path }}/_backup/config/"
|
|
{% endif %}
|
|
{% if db_container | default('') %}
|
|
log "syncing db dumps"
|
|
rsync -az --timeout=600 --delete --mkpath {{ backup_rsync_extra_args | default('') }} \
|
|
-e "$SSH" "$STAGE/db/" "$DEST:{{ remote_path }}/_backup/db/"
|
|
{% endif %}
|
|
|
|
log "status=ok"
|