diff --git a/ansible/roles/ups/defaults/main.yml b/ansible/roles/ups/defaults/main.yml index b8d8408..757b3d2 100644 --- a/ansible/roles/ups/defaults/main.yml +++ b/ansible/roles/ups/defaults/main.yml @@ -34,3 +34,8 @@ ups_deps: # Secrets live in ansible/vars/vault.yml (no vault_ prefix, per repo # convention): nut_upsmon_password, nut_truenas_password, idrac_password + +# Seconds between upsd restart attempts. Paired with StartLimitIntervalSec=0 in +# the nut-server drop-in, so a late network address retries patiently instead of +# exhausting the default five-tries-per-interval and giving up for good. +ups_server_restart_sec: 10 diff --git a/ansible/roles/ups/handlers/main.yml b/ansible/roles/ups/handlers/main.yml index e2af69c..d07685e 100644 --- a/ansible/roles/ups/handlers/main.yml +++ b/ansible/roles/ups/handlers/main.yml @@ -16,11 +16,16 @@ name: "nut-driver@{{ ups_name }}.service" state: restarted +# daemon_reload so the drop-in under nut-server.service.d is picked up. The +# reload also clears the start-limit counter, which matters because a unit that +# has latched into "start request repeated too quickly" refuses a plain restart +# until that state is reset. - name: restart nut server become: true ansible.builtin.systemd: name: nut-server.service state: restarted + daemon_reload: true - name: restart nut monitor become: true diff --git a/ansible/roles/ups/tasks/nut.yml b/ansible/roles/ups/tasks/nut.yml index 3f86819..ef6cbe4 100644 --- a/ansible/roles/ups/tasks/nut.yml +++ b/ansible/roles/ups/tasks/nut.yml @@ -42,6 +42,29 @@ notify: restart nut server tags: ups +# upsd binds an explicit address, so it must not start before that address +# exists. See the template for the boot race this fixes. +- name: ensure nut-server drop-in directory exists + become: true + ansible.builtin.file: + path: /etc/systemd/system/nut-server.service.d + state: directory + owner: root + group: root + mode: '0755' + tags: ups + +- name: order upsd after the network is actually online + become: true + ansible.builtin.template: + src: nut-server-network-online.conf.j2 + dest: /etc/systemd/system/nut-server.service.d/10-network-online.conf + owner: root + group: root + mode: '0644' + notify: restart nut server + tags: ups + - name: deploy upsd users become: true ansible.builtin.template: diff --git a/ansible/roles/ups/templates/nut-server-network-online.conf.j2 b/ansible/roles/ups/templates/nut-server-network-online.conf.j2 new file mode 100644 index 0000000..01555d6 --- /dev/null +++ b/ansible/roles/ups/templates/nut-server-network-online.conf.j2 @@ -0,0 +1,30 @@ +# {{ ansible_managed }} +# upsd binds explicit addresses (see upsd.conf), but the packaged unit only +# orders itself After=network.target -- which is satisfied when networking +# STARTS, not when an address actually exists. The shipped unit even carries +# these two lines commented out, because upstream knows the case. +# +# On 2026-08-28 the host came up and upsd tried to bind 11 seconds into boot, +# before NetworkManager had assigned the address: +# upsd: not listening on {{ ups_listen_addr }} port {{ ups_listen_port }} +# upsd: Fatal error: some listening interfaces were not available +# It then burned all five of the default restart attempts inside one second, +# tripped the start limit, and stayed dead. truenas monitors this host as a +# SLAVE (MONITOR cyberpower@{{ ups_listen_addr }}:{{ ups_listen_port }}), so +# it alarmed NOCOMM continuously until someone noticed. The UPS driver itself +# was fine throughout -- only the server that publishes its state was gone. +# +# NetworkManager-wait-online is enabled on this host, so network-online.target +# genuinely waits for addresses rather than just for the service to start. +# +# The restart settings are belt and braces: StartLimitIntervalSec=0 removes the +# rate limit so a slow address can never exhaust the attempts, and RestartSec +# spaces the retries out instead of hammering five times in a second. +[Unit] +Wants=network-online.target +After=network-online.target +StartLimitIntervalSec=0 + +[Service] +Restart=on-failure +RestartSec={{ ups_server_restart_sec }}