Files
deploy_home/ansible/roles/podman/templates/nextcloud/nextcloud-backup-alert.sh.j2
T
Bastian de Byl 16145fb6bd only email genuine backup failures
A spurious invocation carries no action for a reader, so mailing it just
trains them to skip past the subject line -- which defeats the point of the
alert. Send mail only when the unit actually reports failure; spurious
triggers still leave their journald record, so they stay greppable and can
still feed a Graylog rule.

Verified both paths: a spurious trigger leaves /var/log/msmtp.log untouched
and logs alert_mail=skipped reason=spurious, while a genuinely failed unit
still composes mail with the FAILED subject.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-07-26 20:34:45 -04:00

66 lines
2.8 KiB
Django/Jinja

#!/bin/bash
# {{ ansible_managed }}
# OnFailure= handler for the Nextcloud backup units. Invoked as:
# nextcloud-backup-alert.sh <failed-unit-name>
#
# Deliberately NOT `set -e`: an alert handler that dies partway through
# reports nothing, which is worse than a partial report. Same reasoning as
# roles/ups/templates/ups-restore.sh.j2.
set -uo pipefail
TAG=nextcloud-backup
UNIT="${1:-unknown}"
TO="{{ backup_alert_email | default('root') }}"
HOST="$(hostname -f 2>/dev/null || hostname)"
result="$(systemctl show -p Result --value "$UNIT" 2>/dev/null)"
code="$(systemctl show -p ExecMainStatus --value "$UNIT" 2>/dev/null)"
# Only claim a failure when the unit actually reports one. Starting this
# handler by hand (or any other spurious trigger) would otherwise mail out a
# subject line saying FAILED about a run that succeeded. Keeping status=failed
# exact also stops such triggers matching the Graylog alert rule.
if [ "${result:-success}" = "success" ]; then
state=spurious
prio=daemon.warning
headline="$(printf 'Nextcloud backup alert handler was invoked on %s, but %s reports SUCCESS.\nThis is not a backup failure -- most likely the handler was started manually.' "$HOST" "$UNIT")"
subject="[$HOST] Nextcloud backup alert (spurious, unit OK): $UNIT"
else
state=failed
prio=daemon.err
headline="Nextcloud backup FAILED on $HOST"
subject="[$HOST] Nextcloud backup FAILED: $UNIT"
fi
# One machine-parseable line for Graylog, then the context.
logger -t "$TAG" -p "$prio" -- \
"status=$state unit=$UNIT result=${result:-unknown} exit=${code:-unknown}"
body="$(printf '%s\n\nunit: %s\nresult: %s\nexit: %s\n\n--- last 40 journal lines ---\n' \
"$headline" "$UNIT" "${result:-unknown}" "${code:-unknown}")
$(journalctl -u "$UNIT" -n 40 --no-pager -o cat 2>/dev/null)"
echo "$body" | logger -t "$TAG" -p "$prio"
# Only genuine failures are worth an email. A spurious invocation carries no
# action for a human, and mailing it trains the reader to ignore the subject
# line -- which defeats the point of having the alert at all. The journald
# record above is kept either way, so spurious triggers stay greppable.
if [ "$state" != "failed" ]; then
logger -t "$TAG" -p daemon.info -- "alert_mail=skipped reason=$state"
exit 0
fi
# Mail is best-effort: if the MTA is not configured the journald record above
# is still the authoritative signal, so never fail the handler on this.
if command -v sendmail >/dev/null 2>&1; then
printf 'To: %s\nSubject: %s\nContent-Type: text/plain; charset=UTF-8\n\n%s\n' \
"$TO" "$subject" "$body" | sendmail -t \
&& logger -t "$TAG" -p daemon.info -- "alert_mail=sent to=$TO" \
|| logger -t "$TAG" -p daemon.err -- "alert_mail=failed to=$TO"
else
logger -t "$TAG" -p daemon.err -- "alert_mail=skipped reason=no-sendmail"
fi
exit 0