From 62c4015410e898582ce9eb74c1db11e6b64a0900 Mon Sep 17 00:00:00 2001 From: Bastian de Byl Date: Wed, 26 Aug 2026 13:43:53 -0400 Subject: [PATCH] fix: let systemd actually notice when the Zomboid container exits `@bot restart` saves, sends RCON quit, and relies on the unit's Restart=always to bring the server back. It never came back. The container sat at exited(0) while the unit reported active/running with NRestarts=0, so the bot waited for a startup that was never going to happen and the server stayed down until someone noticed. podman generate systemd emits Type=forking with ExecStart=podman start and a PIDFile pointing at conmon. podman start returns immediately, so the process systemd was told to supervise was never its child -- it warns about exactly this in the journal, once per poll, and then does not notice the exit: zomboid.service: Supervising process 2431312 which is not our child. We'll most likely not notice when it exits. That PIDFile is also stale by design: it embeds the container ID, so every deploy that recreates the container leaves it pointing at nothing. Type=simple with `podman start -a` keeps podman in the foreground as systemd's own child, so the exit is seen and Restart=always does what it always claimed to. stdout and stderr are discarded deliberately -- attaching re-emits the container's output for systemd to capture straight back into the journal, which is the flood the k8s-file driver exists to prevent. podman logs still has it. Co-Authored-By: Claude Opus 5 (1M context) --- .../podman/tasks/containers/home/zomboid.yml | 35 ++++++++++++++++--- 1 file changed, 31 insertions(+), 4 deletions(-) diff --git a/ansible/roles/podman/tasks/containers/home/zomboid.yml b/ansible/roles/podman/tasks/containers/home/zomboid.yml index bdb90aa..d6e9c01 100644 --- a/ansible/roles/podman/tasks/containers/home/zomboid.yml +++ b/ansible/roles/podman/tasks/containers/home/zomboid.yml @@ -271,14 +271,41 @@ vars: container_name: zomboid -# Ensure zomboid restarts on any exit (including admin-triggered restarts) -- name: configure zomboid systemd to always restart +# Make systemd actually supervise the container, so it restarts on any exit -- +# including the clean one that `@bot restart` produces via RCON quit. +# +# `podman generate systemd` emits Type=forking with ExecStart=podman start and +# a PIDFile pointing at conmon. podman start returns immediately, so the process +# systemd is told to watch was never its child; it says so itself in the journal +# ("Supervising process N which is not our child. We'll most likely not notice +# when it exits") and it does not notice. The unit sat at active/running with +# NRestarts=0 while the container was exited(0) and the server was down. The +# PIDFile is worse than useless here: it embeds the container ID, so it goes +# stale every time a deploy recreates the container. +# +# Type=simple with `podman start -a` keeps podman in the foreground as systemd's +# own child, so the exit is seen and Restart=always fires. +# +# stdout/stderr are discarded on purpose. Attaching re-emits the container's +# output on podman's stdout, which systemd would capture straight back into the +# journal -- exactly the flood the k8s-file log driver exists to avoid. Nothing +# is lost: `podman logs` still serves it from the container's own log. +- name: configure zomboid systemd supervision become: true become_user: "{{ podman_user }}" ansible.builtin.lineinfile: path: "{{ podman_home }}/.config/systemd/user/zomboid.service" - regexp: "^Restart=" - line: "Restart=always" + regexp: "{{ item.regexp }}" + line: "{{ item.line }}" + state: "{{ item.state | default('present') }}" + insertafter: '^\[Service\]' + loop: + - { regexp: '^Restart=', line: 'Restart=always' } + - { regexp: '^Type=', line: 'Type=simple' } + - { regexp: '^ExecStart=', line: 'ExecStart=/usr/bin/podman start -a zomboid' } + - { regexp: '^StandardOutput=', line: 'StandardOutput=null' } + - { regexp: '^StandardError=', line: 'StandardError=null' } + - { regexp: '^PIDFile=', line: '', state: absent } notify: reload zomboid systemd # Firewall logging for player IP correlation