Compare commits
6
Commits
87a0332a75
...
9954d774e7
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9954d774e7 | ||
|
|
fb09a01c88 | ||
|
|
ae99ba415e | ||
|
|
2a6390c0ad | ||
|
|
f21be79452 | ||
|
|
9e136ce903 |
@@ -21,3 +21,12 @@ services:
|
||||
- podman
|
||||
- fail2ban
|
||||
- systemd-timesyncd
|
||||
|
||||
# Storage tuning for the SATA SSD (see tasks/service.yml and the udev template).
|
||||
ssd_io_scheduler: mq-deadline
|
||||
ssd_nr_requests: "256"
|
||||
# 256 MB before background writeback starts, 1 GB before writers block. Absolute
|
||||
# bytes rather than the default percent-of-RAM ratios, which scale to multi-GB
|
||||
# stalls on a 31 GB host.
|
||||
vm_dirty_background_bytes: "268435456"
|
||||
vm_dirty_bytes: "1073741824"
|
||||
|
||||
@@ -26,3 +26,13 @@
|
||||
ansible.builtin.systemd:
|
||||
name: systemd-journald
|
||||
state: restarted
|
||||
|
||||
# --reload-rules alone only affects devices that appear later; the root disk is
|
||||
# already attached, so trigger a change event to apply the rule now rather than
|
||||
# at the next reboot.
|
||||
- name: reload_udev
|
||||
become: true
|
||||
ansible.builtin.shell: |
|
||||
udevadm control --reload-rules
|
||||
udevadm trigger --subsystem-match=block --action=change
|
||||
changed_when: true
|
||||
|
||||
@@ -23,6 +23,77 @@
|
||||
notify: restart_journald
|
||||
tags: security, service, journald
|
||||
|
||||
# rsyslog wrote a second full copy of the journal to /var/log/messages. It
|
||||
# loads imjournal, so it reads the journal directly and ForwardToSyslog=no
|
||||
# alone does not stop it -- the unit itself has to go. Nothing consumes those
|
||||
# files: fail2ban runs backend=systemd and matches on the journal
|
||||
# (_SYSTEMD_UNIT=sshd.service, "No file is currently monitored"), and lsof
|
||||
# showed only rsyslogd itself holding them open. The journal is capped and is
|
||||
# the system of record, so this was pure write amplification: 6 GB of rotated
|
||||
# copies, two weekly files of which were 2.8 GB and 2.5 GB.
|
||||
# Masked rather than merely disabled so a dependency cannot pull it back in.
|
||||
- name: disable rsyslog, which duplicated the journal to /var/log/messages
|
||||
become: true
|
||||
ansible.builtin.systemd:
|
||||
name: rsyslog
|
||||
state: stopped
|
||||
enabled: false
|
||||
masked: true
|
||||
tags: security, service, journald
|
||||
|
||||
- name: find the syslog copies rsyslog left behind
|
||||
become: true
|
||||
ansible.builtin.find:
|
||||
paths: /var/log
|
||||
patterns: "messages*,secure*,cron*,maillog*"
|
||||
register: syslog_leftovers
|
||||
tags: security, service, journald
|
||||
|
||||
- name: reclaim the syslog copies
|
||||
become: true
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.path }}"
|
||||
state: absent
|
||||
loop: "{{ syslog_leftovers.files }}"
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
tags: security, service, journald
|
||||
|
||||
# Storage tuning for the SATA SSD. See the template for why bfq is wrong here.
|
||||
- name: use an SSD-appropriate I/O scheduler and queue depth
|
||||
become: true
|
||||
ansible.builtin.template:
|
||||
src: 60-ssd-scheduler.rules.j2
|
||||
dest: /etc/udev/rules.d/60-ssd-scheduler.rules
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
notify: reload_udev
|
||||
tags: security, service, storage
|
||||
|
||||
# Writeback was left at the defaults, which are ratios of RAM: dirty_ratio=20
|
||||
# and dirty_background_ratio=10 on 31 GB means the kernel will sit on up to
|
||||
# 3.1 GB before it starts writing back and 6.2 GB before it blocks writers
|
||||
# outright. Flushing that much at once to a QLC SATA drive -- which falls to
|
||||
# roughly 80-160 MB/s once its SLC cache is spent -- takes tens of seconds, and
|
||||
# everything else stalls behind it. Capping the dirty set in absolute bytes
|
||||
# instead trades one long stall for frequent short ones, which is what keeps
|
||||
# CI and the databases responsive.
|
||||
- name: bound writeback so a flush cannot stall the box
|
||||
become: true
|
||||
ansible.posix.sysctl:
|
||||
name: "{{ item.name }}"
|
||||
value: "{{ item.value }}"
|
||||
sysctl_set: true
|
||||
state: present
|
||||
reload: true
|
||||
loop:
|
||||
- { name: vm.dirty_background_bytes, value: "{{ vm_dirty_background_bytes }}" }
|
||||
- { name: vm.dirty_bytes, value: "{{ vm_dirty_bytes }}" }
|
||||
loop_control:
|
||||
label: "{{ item.name }}={{ item.value }}"
|
||||
tags: security, service, storage
|
||||
|
||||
- name: ensure desired services are started and enabled
|
||||
become: true
|
||||
ansible.builtin.service:
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
# {{ ansible_managed }}
|
||||
# Fedora's /usr/lib/udev/rules.d/60-block-scheduler.rules picks bfq for every
|
||||
# rotational=0 SATA disk. bfq is a fairness scheduler built for spinning disks
|
||||
# and desktop interactivity: it costs real CPU per request and, with 26
|
||||
# containers plus two CI runners all writing at once, it lets reads queue
|
||||
# behind write bursts. That is the shape of the stalls seen here -- I/O
|
||||
# pressure spiking to 76% while the CPU sat nearly idle.
|
||||
#
|
||||
# mq-deadline instead of none: this is a SATA SSD with a 32-deep NCBQ queue,
|
||||
# not an NVMe device with its own deep queues, so the request merging and the
|
||||
# read-expiry deadline are both worth having. The deadline is what stops reads
|
||||
# starving behind a QLC write burst.
|
||||
#
|
||||
# bfq also hard-caps nr_requests at 64; mq-deadline allows a deeper queue,
|
||||
# which is what lets concurrent container and CI I/O actually overlap.
|
||||
ACTION=="add|change", KERNEL=="sd[a-z]", ATTR{queue/rotational}=="0", \
|
||||
ATTR{queue/scheduler}="{{ ssd_io_scheduler }}", \
|
||||
ATTR{queue/nr_requests}="{{ ssd_nr_requests }}"
|
||||
@@ -20,3 +20,9 @@
|
||||
{% endif %}
|
||||
[Journal]
|
||||
SystemMaxUse={{ journald_max_use | default('500M') }}
|
||||
|
||||
# Nothing should be forwarded to syslog: rsyslog is disabled (see
|
||||
# tasks/service.yml) because it duplicated the whole journal into
|
||||
# /var/log/messages. This also closes the imuxsock path so anything that
|
||||
# logs via logger(1) still lands in the journal and nowhere else.
|
||||
ForwardToSyslog=no
|
||||
|
||||
@@ -27,6 +27,14 @@ hass_path: "{{ podman_volumes }}/hass"
|
||||
partsy_path: "{{ podman_volumes }}/partsy"
|
||||
partsy_skudak_path: "{{ podman_volumes }}/partsy-skudak"
|
||||
photos_path: "{{ podman_volumes }}/photos"
|
||||
# Named rather than hardcoded so the ML service can be stood up beside a broken
|
||||
# one without editing tasks. On 2026-08-28 a worker thread wedged in
|
||||
# uninterruptible sleep (D state) in exit_mmap, which made the container
|
||||
# unremovable -- it survived SIGKILL and `podman rm -f` and kept its name and
|
||||
# network alias until the host was rebooted. The reboot cleared it, so this
|
||||
# stays on the canonical name; see MACHINE_LEARNING_MODEL_TTL in photos.yml for
|
||||
# the change that stops it recurring.
|
||||
immich_ml_container: immich-machine-learning
|
||||
uptime_kuma_path: "{{ podman_volumes }}/uptime-kuma"
|
||||
uptime_kuma_personal_path: "{{ podman_volumes }}/uptime-kuma-personal"
|
||||
zomboid_path: "{{ podman_volumes }}/zomboid"
|
||||
@@ -269,7 +277,19 @@ podman_prune_users:
|
||||
- "{{ git_user }}"
|
||||
# Keep 30 days of unused images so a rollback needs no rebuild or re-pull.
|
||||
podman_prune_until: 720h
|
||||
podman_prune_oncalendar: "Sun *-*-* 02:00:00"
|
||||
|
||||
# The CI runners were never pruned and had run away: gitea-runner reached 1205
|
||||
# images / 113 GB with 100% reclaimable, actions-runner 137 exited job
|
||||
# containers. Build layers carry no rollback value, so they keep a far shorter
|
||||
# window than the service stores and their exited containers are reaped too.
|
||||
podman_prune_ci_users:
|
||||
- gitea-runner
|
||||
- actions-runner
|
||||
podman_prune_ci_until: 48h
|
||||
|
||||
# Daily rather than weekly: CI turns over many images a day, and a week of that
|
||||
# is what let the store reach 113 GB between runs.
|
||||
podman_prune_oncalendar: "*-*-* 02:00:00"
|
||||
|
||||
# Hardened CIFS options for the TrueNAS shares (see containers/home/photos.yml).
|
||||
# x-systemd.automount is what makes a failed mount recoverable without a human.
|
||||
|
||||
@@ -9,6 +9,11 @@ http:
|
||||
use_x_forwarded_for: true
|
||||
trusted_proxies:
|
||||
- 127.0.0.1
|
||||
# Caddy runs on the host network and reaches us through the published
|
||||
# port, so pasta (rootless podman's default since v5) rewrites the source
|
||||
# to the container's own link-local tap0 address, not 10.0.2.x as
|
||||
# slirp4netns used to.
|
||||
- 169.254.0.0/16
|
||||
- 10.0.0.0/8
|
||||
- 192.168.1.0/24
|
||||
|
||||
|
||||
@@ -1,8 +1,20 @@
|
||||
---
|
||||
# -x keeps this on the local filesystem. Two TrueNAS CIFS shares are mounted
|
||||
# INSIDE this tree -- volumes/photos/immich and volumes/photos/storage -- and
|
||||
# without -x restorecon walked the entire remote photo library over SMB,
|
||||
# relabelling a filesystem that cannot even store SELinux xattrs. On 2026-08-28
|
||||
# that pinned CPU#0 at 100% system time and the kernel logged escalating soft
|
||||
# lockups ("BUG: soft lockup - CPU#0 stuck for 1423s! [restorecon]") until the
|
||||
# host could no longer create login sessions and needed a hard reboot. An
|
||||
# earlier run the same day had already been SIGKILLed at 49s, which was the
|
||||
# same problem surfacing quietly.
|
||||
#
|
||||
# The mounts are also x-systemd.automount, so merely walking into them triggers
|
||||
# a mount -- there is nothing to relabel there and never was.
|
||||
- name: restorecon podman
|
||||
become: true
|
||||
ansible.builtin.command: |
|
||||
restorecon -Frv {{ podman_home }}/.local/share/volumes
|
||||
restorecon -Frxv {{ podman_home }}/.local/share/volumes
|
||||
tags:
|
||||
- podman
|
||||
- selinux
|
||||
|
||||
@@ -94,26 +94,35 @@
|
||||
|
||||
- import_tasks: podman/podman-check.yml
|
||||
vars:
|
||||
container_name: immich-machine-learning
|
||||
container_name: "{{ immich_ml_container }}"
|
||||
container_image: "{{ ml_image }}"
|
||||
|
||||
# MACHINE_LEARNING_MODEL_TTL=0 keeps the models resident instead of unloading
|
||||
# them after 300s idle. The default made every search following an idle gap
|
||||
# reload four models (CLIP, buffalo_l detection + recognition, PP-OCRv5) on a
|
||||
# CPU-only 4-core box, then tear those mappings down again -- and it is that
|
||||
# teardown, in exit_mmap, that wedged a worker thread in uninterruptible sleep
|
||||
# on 2026-08-27 and took smart search, face detection and OCR down with it.
|
||||
# Costs ~1-2 GB resident to remove the code path entirely.
|
||||
- name: create immich-ml container
|
||||
become: true
|
||||
become_user: "{{ podman_user }}"
|
||||
containers.podman.podman_container:
|
||||
name: immich-machine-learning
|
||||
name: "{{ immich_ml_container }}"
|
||||
image: "{{ ml_image }}"
|
||||
restart_policy: on-failure:3
|
||||
log_driver: journald
|
||||
network:
|
||||
- shared
|
||||
env:
|
||||
MACHINE_LEARNING_MODEL_TTL: "0"
|
||||
volumes:
|
||||
- "{{ photos_path }}/mlcache:/cache"
|
||||
|
||||
- name: create systemd startup job for immich-machine-learning
|
||||
include_tasks: podman/systemd-generate.yml
|
||||
vars:
|
||||
container_name: immich-machine-learning
|
||||
container_name: "{{ immich_ml_container }}"
|
||||
|
||||
- import_tasks: podman/podman-check.yml
|
||||
vars:
|
||||
@@ -186,6 +195,10 @@
|
||||
DB_USERNAME: photos
|
||||
DB_PASSWORD: "{{ photos_db_pass }}"
|
||||
IMMICH_PORT: 8088
|
||||
# Set explicitly rather than relying on immich's default of
|
||||
# http://immich-machine-learning:3003, so the ML container can be renamed
|
||||
# without silently losing search, faces and OCR.
|
||||
IMMICH_MACHINE_LEARNING_URL: "http://{{ immich_ml_container }}:3003"
|
||||
volumes:
|
||||
- "{{ photos_path }}/storage:/mnt/media/originals"
|
||||
- "{{ photos_path }}/immich:/usr/src/app/upload"
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
|
||||
- import_tasks: containers/home/hass.yml
|
||||
vars:
|
||||
image: ghcr.io/home-assistant/home-assistant:2026.5.1
|
||||
image: ghcr.io/home-assistant/home-assistant:2026.8.3
|
||||
tags: hass
|
||||
|
||||
- import_tasks: containers/home/partsy.yml
|
||||
@@ -81,13 +81,13 @@
|
||||
|
||||
- import_tasks: containers/debyltech/fulfillr.yml
|
||||
vars:
|
||||
image: git.debyl.io/debyltech/fulfillr:20260827.1453
|
||||
image: git.debyl.io/debyltech/fulfillr:20260827.2009
|
||||
tags: debyltech, fulfillr
|
||||
|
||||
# Staging back-office (fulfillr-dev.debyltech.com) — same image, staging Turso config.
|
||||
- import_tasks: containers/debyltech/fulfillr-dev.yml
|
||||
vars:
|
||||
image: git.debyl.io/debyltech/fulfillr:20260825.1909
|
||||
image: git.debyl.io/debyltech/fulfillr:20260827.2009
|
||||
tags: debyltech, fulfillr-dev
|
||||
|
||||
- import_tasks: containers/debyltech/uptime-kuma.yml
|
||||
|
||||
@@ -1,13 +1,26 @@
|
||||
#!/bin/bash
|
||||
# {{ ansible_managed }}
|
||||
# Weekly reclaim of unused podman images and volumes.
|
||||
# Daily reclaim of unused podman images, volumes and exited job containers.
|
||||
#
|
||||
# Every image bump leaves the previous tag behind and nothing ever removed
|
||||
# them: this was written after finding 896 images totalling 59.6 GB, 75% of it
|
||||
# unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy.
|
||||
#
|
||||
# --filter until={{ podman_prune_until }} keeps recent images so a rollback
|
||||
# does not require a rebuild or re-pull. Anything older is unused AND stale.
|
||||
# Two policies, because the stores serve different purposes:
|
||||
#
|
||||
# service users ({{ podman_prune_users | join(', ') }})
|
||||
# until={{ podman_prune_until }} keeps recent images so a rollback does not
|
||||
# require a rebuild or re-pull. Containers are deliberately NOT pruned here:
|
||||
# they are the live services, and reaping one that merely happens to be
|
||||
# stopped would turn a transient crash into a unit that cannot start again
|
||||
# until the next deploy.
|
||||
#
|
||||
# CI users ({{ podman_prune_ci_users | join(', ') }})
|
||||
# Build layers are throwaway and there is no rollback to protect, so these
|
||||
# get a much shorter window ({{ podman_prune_ci_until }}) and their exited
|
||||
# job containers are reaped too. They were never covered before: gitea-
|
||||
# runner had reached 1205 images / 113 GB, 100% of it reclaimable, and it
|
||||
# is the layer count that makes overlayfs lookups -- and so CI itself -- slow.
|
||||
#
|
||||
# Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts
|
||||
# under {{ podman_volumes }} that hold real service data -- those are
|
||||
@@ -20,34 +33,54 @@ TAG=podman-prune
|
||||
|
||||
log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; }
|
||||
|
||||
total_before=0
|
||||
total_after=0
|
||||
# Rootless podman: -H so HOME points at the user's store, and the `cd;`
|
||||
# preamble is required (see CLAUDE.md) or podman cannot find its graph root.
|
||||
run() {
|
||||
local u=$1
|
||||
shift
|
||||
sudo -H -u "$u" bash -c \
|
||||
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
|
||||
exec podman "$@"' _ "$@"
|
||||
}
|
||||
|
||||
for u in {{ podman_prune_users | join(' ') }}; do
|
||||
# Rootless podman: -H so HOME points at the user's store, and the `cd;`
|
||||
# preamble is required (see CLAUDE.md) or podman cannot find its graph root.
|
||||
run() {
|
||||
sudo -H -u "$u" bash -c \
|
||||
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
|
||||
exec podman "$@"' _ "$@"
|
||||
}
|
||||
# prune_user <user> <until> <prune_containers: yes|no>
|
||||
prune_user() {
|
||||
local u=$1 keep=$2 do_containers=$3
|
||||
local before after img vol con
|
||||
|
||||
if ! id "$u" >/dev/null 2>&1; then
|
||||
log "user=$u status=skipped reason=no-such-user"
|
||||
continue
|
||||
return
|
||||
fi
|
||||
|
||||
before=$(run system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
||||
before=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
||||
|
||||
# Deliberately NOT `set -e`: a prune failing for one user must not stop the
|
||||
# other, and a busy image is a normal, non-fatal outcome.
|
||||
img=$(run image prune -af --filter "until={{ podman_prune_until }}" 2>&1 | tail -1)
|
||||
vol=$(run volume prune -f 2>&1 | tail -1)
|
||||
# others, and a busy image is a normal, non-fatal outcome.
|
||||
#
|
||||
# Containers are reaped BEFORE images on purpose -- an exited container pins
|
||||
# the image it ran from, so pruning images first would leave those layers
|
||||
# behind for another day.
|
||||
con=none
|
||||
if [ "$do_containers" = yes ]; then
|
||||
con=$(run "$u" container prune -f --filter "until=$keep" 2>&1 | tail -1)
|
||||
fi
|
||||
|
||||
after=$(run system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
||||
img=$(run "$u" image prune -af --filter "until=$keep" 2>&1 | tail -1)
|
||||
vol=$(run "$u" volume prune -f 2>&1 | tail -1)
|
||||
|
||||
log "user=$u images_before=$before images_after=$after"
|
||||
log "user=$u image_prune=${img:-none} volume_prune=${vol:-none}"
|
||||
after=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
||||
|
||||
log "user=$u keep=$keep size_before=$before size_after=$after"
|
||||
log "user=$u image_prune=${img:-none} volume_prune=${vol:-none} container_prune=${con:-none}"
|
||||
}
|
||||
|
||||
for u in {{ podman_prune_users | join(' ') }}; do
|
||||
prune_user "$u" "{{ podman_prune_until }}" no
|
||||
done
|
||||
|
||||
for u in {{ podman_prune_ci_users | join(' ') }}; do
|
||||
prune_user "$u" "{{ podman_prune_ci_until }}" yes
|
||||
done
|
||||
|
||||
log "status=ok"
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
[Unit]
|
||||
Description=Weekly podman image and volume prune
|
||||
Description=Daily podman image, volume and CI container prune
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ podman_prune_oncalendar | default('Sun *-*-* 02:00:00') }}
|
||||
RandomizedDelaySec=15m
|
||||
# Weekly, and growth is one tag per deploy, so a missed run is worth catching
|
||||
# up on rather than skipping.
|
||||
# Growth is one tag per deploy plus every CI build, so a missed run is worth
|
||||
# catching up on rather than skipping.
|
||||
Persistent=true
|
||||
|
||||
[Install]
|
||||
|
||||
@@ -34,3 +34,8 @@ ups_deps:
|
||||
|
||||
# Secrets live in ansible/vars/vault.yml (no vault_ prefix, per repo
|
||||
# convention): nut_upsmon_password, nut_truenas_password, idrac_password
|
||||
|
||||
# Seconds between upsd restart attempts. Paired with StartLimitIntervalSec=0 in
|
||||
# the nut-server drop-in, so a late network address retries patiently instead of
|
||||
# exhausting the default five-tries-per-interval and giving up for good.
|
||||
ups_server_restart_sec: 10
|
||||
|
||||
@@ -16,11 +16,16 @@
|
||||
name: "nut-driver@{{ ups_name }}.service"
|
||||
state: restarted
|
||||
|
||||
# daemon_reload so the drop-in under nut-server.service.d is picked up. The
|
||||
# reload also clears the start-limit counter, which matters because a unit that
|
||||
# has latched into "start request repeated too quickly" refuses a plain restart
|
||||
# until that state is reset.
|
||||
- name: restart nut server
|
||||
become: true
|
||||
ansible.builtin.systemd:
|
||||
name: nut-server.service
|
||||
state: restarted
|
||||
daemon_reload: true
|
||||
|
||||
- name: restart nut monitor
|
||||
become: true
|
||||
|
||||
@@ -42,6 +42,29 @@
|
||||
notify: restart nut server
|
||||
tags: ups
|
||||
|
||||
# upsd binds an explicit address, so it must not start before that address
|
||||
# exists. See the template for the boot race this fixes.
|
||||
- name: ensure nut-server drop-in directory exists
|
||||
become: true
|
||||
ansible.builtin.file:
|
||||
path: /etc/systemd/system/nut-server.service.d
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0755'
|
||||
tags: ups
|
||||
|
||||
- name: order upsd after the network is actually online
|
||||
become: true
|
||||
ansible.builtin.template:
|
||||
src: nut-server-network-online.conf.j2
|
||||
dest: /etc/systemd/system/nut-server.service.d/10-network-online.conf
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
notify: restart nut server
|
||||
tags: ups
|
||||
|
||||
- name: deploy upsd users
|
||||
become: true
|
||||
ansible.builtin.template:
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
# {{ ansible_managed }}
|
||||
# upsd binds explicit addresses (see upsd.conf), but the packaged unit only
|
||||
# orders itself After=network.target -- which is satisfied when networking
|
||||
# STARTS, not when an address actually exists. The shipped unit even carries
|
||||
# these two lines commented out, because upstream knows the case.
|
||||
#
|
||||
# On 2026-08-28 the host came up and upsd tried to bind 11 seconds into boot,
|
||||
# before NetworkManager had assigned the address:
|
||||
# upsd: not listening on {{ ups_listen_addr }} port {{ ups_listen_port }}
|
||||
# upsd: Fatal error: some listening interfaces were not available
|
||||
# It then burned all five of the default restart attempts inside one second,
|
||||
# tripped the start limit, and stayed dead. truenas monitors this host as a
|
||||
# SLAVE (MONITOR cyberpower@{{ ups_listen_addr }}:{{ ups_listen_port }}), so
|
||||
# it alarmed NOCOMM continuously until someone noticed. The UPS driver itself
|
||||
# was fine throughout -- only the server that publishes its state was gone.
|
||||
#
|
||||
# NetworkManager-wait-online is enabled on this host, so network-online.target
|
||||
# genuinely waits for addresses rather than just for the service to start.
|
||||
#
|
||||
# The restart settings are belt and braces: StartLimitIntervalSec=0 removes the
|
||||
# rate limit so a slow address can never exhaust the attempts, and RestartSec
|
||||
# spaces the retries out instead of hammering five times in a second.
|
||||
[Unit]
|
||||
Wants=network-online.target
|
||||
After=network-online.target
|
||||
StartLimitIntervalSec=0
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec={{ ups_server_restart_sec }}
|
||||
Reference in New Issue
Block a user