From eea55def6cfb1bf78748f934bffa07df8358c4cf Mon Sep 17 00:00:00 2001 From: Bastian de Byl Date: Tue, 25 Aug 2026 23:40:10 -0400 Subject: [PATCH] retire Graylog behind a flag, fix Caddy reloads, reap awsddns zombies Graylog was the worst cost/benefit tenant on this 4-core box: two JVMs plus MongoDB holding ~1.6 GB resident and ~3% CPU around the clock to store ~3k messages a day -- about 28 MB across its four live indices. journald already retains ~25 days of the same logs at its 500M cap, so this costs searchability, not the logs. The switch is `graylog_enabled` in inventory rather than a role default, because three roles read it (common, podman, graylog-config). The disabled path is an active teardown, not a skipped create: the containers already on the host keep running and their systemd user units keep restarting them at boot unless something stops and removes them. fluent-bit follows the same flag -- with the GELF sink down it would spin retrying a dead 127.0.0.1:12202 and fill the journal it exists to drain -- but only the service state follows, so re-enabling is a restart rather than a reinstall. Caddy reloads were silently no-ops. The handler read /etc/caddy/Caddyfile, which is a single-file bind mount, and podman binds those by inode; the template module writes a temp file and renames it into place, so every deploy gave the host file a new inode while the container kept seeing the one it was created with. Config changes only ever landed when something recreated the container. {{ caddy_path }}/config is also mounted, as a *directory*, and directory mounts resolve names at open() time -- so /config/Caddyfile is always the file Ansible just wrote. awsddns and its four siblings had accumulated 12 zombies over 30 days of uptime. The image's PID 1 is busybox crond, which only waitpid()s the job PIDs it tracks and does no generic orphan reaping, so whenever the run-parts/sh layer exited before the script it left a permanent . init: true puts catatonit at PID 1 to reap them, and the recreation clears the existing ones. Also bumps fulfillr and greg-time-bot images. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/deploy_home.yml | 4 ++ ansible/inventories/home/hosts.yml | 21 ++++++ ansible/roles/common/handlers/main.yml | 4 ++ ansible/roles/common/tasks/fluent-bit.yml | 12 +++- .../common/templates/journald-size.conf.j2 | 15 ++++- ansible/roles/podman/README.md | 65 +++++++++++++++++-- ansible/roles/podman/handlers/main.yml | 21 +++++- .../podman/tasks/containers/base/awsddns.yml | 15 +++++ .../containers/debyltech/graylog-teardown.yml | 59 +++++++++++++++++ ansible/roles/podman/tasks/main.yml | 19 +++++- .../roles/podman/templates/caddy/Caddyfile.j2 | 29 +++++++++ 11 files changed, 250 insertions(+), 14 deletions(-) create mode 100644 ansible/roles/podman/tasks/containers/debyltech/graylog-teardown.yml diff --git a/ansible/deploy_home.yml b/ansible/deploy_home.yml index cc2064d..d22ec82 100644 --- a/ansible/deploy_home.yml +++ b/ansible/deploy_home.yml @@ -9,7 +9,11 @@ # SSL certificates are now handled automatically by Caddy # - role: ssl # REMOVED - Caddy handles all certificate management - role: github-actions + # Provisions streams/pipelines/lookup tables over the Graylog REST API, so + # it can only run when the stack is up -- see graylog_enabled in + # inventories/home/hosts.yml. - role: graylog-config + when: graylog_enabled | bool tags: graylog-config - role: ups tags: ups diff --git a/ansible/inventories/home/hosts.yml b/ansible/inventories/home/hosts.yml index 539b66b..908bf27 100644 --- a/ansible/inventories/home/hosts.yml +++ b/ansible/inventories/home/hosts.yml @@ -1,5 +1,26 @@ --- all: + vars: + # Master switch for the Graylog logging stack (graylog + graylog-opensearch + # + graylog-mongo), the fluent-bit journal shipper that feeds it, the GeoIP + # database download it enriches with, and the graylog-config role that + # provisions its streams/pipelines over the REST API. + # + # Lives in inventory rather than a role default because it is read by three + # separate roles (common, podman, graylog-config) and role defaults are + # scoped to their own role. + # + # Off because the stack is the worst cost/benefit tenant on this 4-core box: + # two JVMs plus MongoDB hold ~1.6 GB resident and burn ~3% of the CPU around + # the clock to store ~3k messages a day -- about 28 MB of actual log data + # across its four live indices. journald already retains ~25 days of the + # same logs at its 500M cap, so turning this off costs searchability, not + # the logs themselves. + # + # Set true (here, or -e graylog_enabled=true) to bring it back. All data + # under {{ graylog_path }} is left in place, so re-enabling resumes with the + # existing indices, streams and pipelines intact. + graylog_enabled: false hosts: home.debyl.io: ansible_user: fedora diff --git a/ansible/roles/common/handlers/main.yml b/ansible/roles/common/handlers/main.yml index 341b298..cf0ef30 100644 --- a/ansible/roles/common/handlers/main.yml +++ b/ansible/roles/common/handlers/main.yml @@ -11,11 +11,15 @@ name: fail2ban state: restarted +# Guarded: a config change still notifies this handler when graylog_enabled is +# false, and an unconditional restart would start the service the fluent-bit +# task just stopped. - name: restart fluent-bit become: true ansible.builtin.systemd: name: fluent-bit state: restarted + when: graylog_enabled | bool - name: restart_journald become: true diff --git a/ansible/roles/common/tasks/fluent-bit.yml b/ansible/roles/common/tasks/fluent-bit.yml index 7144e81..cd4f97c 100644 --- a/ansible/roles/common/tasks/fluent-bit.yml +++ b/ansible/roles/common/tasks/fluent-bit.yml @@ -1,6 +1,12 @@ --- # Fluent Bit - Log forwarder from journald to Graylog GELF # Deployed as systemd service (not container) for direct journal access +# +# Graylog's GELF input is fluent-bit's only output, so the two share a switch: +# with the stack down fluent-bit would just spin retrying a dead 127.0.0.1:12202 +# and filling the journal it is meant to be draining. The package and config +# stay installed either way -- only the service follows graylog_enabled (see +# inventories/home/hosts.yml) -- so re-enabling is a restart, not a reinstall. - name: install fluent-bit package become: true @@ -37,9 +43,9 @@ mode: '0644' notify: restart fluent-bit -- name: enable and start fluent-bit service +- name: set fluent-bit service state to match graylog_enabled become: true ansible.builtin.systemd: name: fluent-bit - enabled: true - state: started + enabled: "{{ graylog_enabled | bool }}" + state: "{{ 'started' if (graylog_enabled | bool) else 'stopped' }}" diff --git a/ansible/roles/common/templates/journald-size.conf.j2 b/ansible/roles/common/templates/journald-size.conf.j2 index 919bca7..160e77a 100644 --- a/ansible/roles/common/templates/journald-size.conf.j2 +++ b/ansible/roles/common/templates/journald-size.conf.j2 @@ -3,10 +3,19 @@ # 1.9 TB root means it will happily grow into the tens of gigabytes; it had # reached 4 GB before this was set. # -# Kept small on purpose: every container runs with log_driver=journald and +# Every container runs with log_driver=journald, so this is where container +# stdout lands. +{% if graylog_enabled | bool %} # fluent-bit drains the journal into Graylog continuously (systemd input, # _COMM=conmon -- see templates/fluent-bit/fluent-bit.conf.j2), so Graylog is -# the system of record. What stays here is only the buffer that covers -# fluent-bit being down, and 500M is a long outage at this log rate. +# the system of record and what stays here is only the buffer that covers +# fluent-bit being down. +{% else %} +# Graylog is disabled (see graylog_enabled in inventories/home/hosts.yml), so +# the journal is now the only log store. No increase was needed for that: the +# cap is a size limit, not a time limit, and fluent-bit only ever read the +# journal rather than rotating it, so retention is unchanged at roughly 25 days +# at the current rate -- longer than Graylog's own four live indices covered. +{% endif %} [Journal] SystemMaxUse={{ journald_max_use | default('500M') }} diff --git a/ansible/roles/podman/README.md b/ansible/roles/podman/README.md index dde01d7..3dfad43 100644 --- a/ansible/roles/podman/README.md +++ b/ansible/roles/podman/README.md @@ -19,10 +19,15 @@ Stages, in order — the ordering is deliberate, see the comments in the templat 2. **SQLite snapshots** (`.backup`, then `pragma integrity_check`) where used. 3. **rsync** of the data tree, config, and db dumps to TrueNAS. -Failures raise `status=failed` on the `nextcloud-backup` syslog tag, which an -external Graylog rule matches to send mail. Do not rename that tag: it is shared -by every instance including Gitea, and renaming it here silently stops alerting -for all of them. +Failures raise `status=failed` on the `nextcloud-backup` syslog tag. The mail +itself is sent by the unit's own `OnFailure=` handler +(`templates/nextcloud/nextcloud-backup-alert.sh.j2`) straight through +`sendmail`, so alerting does **not** depend on Graylog and is unaffected by +`graylog_enabled` being off. A Graylog rule matching the same tag is a +secondary, dashboard-side copy of that signal. + +Do not rename that tag: it is shared by every instance including Gitea, and +renaming it here silently stops the Graylog-side alerting for all of them. ## Restore @@ -141,3 +146,55 @@ LibreSign setup failures and buys nothing at this scale. - `LC_ALL` / `LANG` must be set or the JVM comes up as `ANSI_X3.4-1968` and LibreSign warns that accented characters in signer names will be mangled (LibreSign issue #4872). + + +## Logging + +Every container runs with `log_driver=journald`, so container stdout lands in +the host journal, capped at 500M by `roles/common/templates/journald-size.conf.j2` +(about 25 days at the current rate). `journalctl CONTAINER_NAME=` is the +day-to-day way to read it. + +On top of that sits an optional Graylog stack — `graylog`, `graylog-opensearch`, +`graylog-mongo`, fed by a host `fluent-bit` service that tails the journal and +ships GELF to `127.0.0.1:12202`, enriched with the MaxMind GeoIP database, and +configured over the REST API by the separate `graylog-config` role. + +**It is off.** The switch is `graylog_enabled` in +`inventories/home/hosts.yml`, and it gates all five of those pieces at once: + +| `graylog_enabled` | effect | +| --- | --- | +| `true` | stack deployed, fluent-bit shipping, GeoIP downloaded, `graylog-config` runs, `logs.debyl.io` proxies to the UI and `/gelf` | +| `false` | containers removed and their systemd user units disabled and deleted, fluent-bit stopped and disabled, GeoIP and `graylog-config` skipped, `logs.debyl.io` answers 503 | + +It was turned off because the cost/benefit is bad on a 4-core box: two JVMs plus +MongoDB held ~1.6 GB resident and ~3% of the CPU continuously to store roughly +3k messages a day — about 28 MB of real log data across four live indices — all +of which journald already keeps for longer. + +Turning it back on is `graylog_enabled: true` plus `make deploy TAGS=graylog`. +Nothing is destroyed by the off path: the volumes under `{{ graylog_path }}` +keep the indices, the Mongo database holding streams/pipelines/dashboards, and +the node-id file, so the stack comes back with its configuration intact. + +Two things genuinely stop while it is off, both by design: + +- **Search and dashboards.** The logs still exist in journald; the query + interface over them does not. +- **External GELF ingest.** The AWS Lambda that POSTs to `logs.debyl.io/gelf` + for the `debyltech-api` stream gets a 503. Those events are dropped, not + queued — nothing else records them. + +### Caddy config reloads + +The `reload caddy` handler deliberately reads `/config/Caddyfile`, not the +`/etc/caddy/Caddyfile` the container starts from, even though both are the same +host file. `/etc/caddy/Caddyfile` is a **single-file** bind mount, which podman +binds by inode, and Ansible's `template` module writes a temp file and renames +it into place — so every deploy gives the host file a new inode while the +container keeps seeing the one it was created with. Reloading from that path +silently re-applied the previous config; changes only landed when something +recreated the container. `{{ caddy_path }}/config` is also bind-mounted as a +*directory* at `/config`, and directory mounts resolve names at `open()` time, +so `/config/Caddyfile` is always the file Ansible just wrote. diff --git a/ansible/roles/podman/handlers/main.yml b/ansible/roles/podman/handlers/main.yml index fa095dd..80b532c 100644 --- a/ansible/roles/podman/handlers/main.yml +++ b/ansible/roles/podman/handlers/main.yml @@ -25,11 +25,21 @@ tags: - caddy +# Reads /config/Caddyfile, NOT the /etc/caddy/Caddyfile the container starts +# from, even though both are the same host file. /etc/caddy/Caddyfile is a +# single-file bind mount, which podman binds by inode; the template module +# writes a temp file and renames it into place, so every deploy gives the host +# file a new inode and the container keeps seeing the one it was created with. +# Reloading from that path silently re-applied the old config -- the change only +# ever landed when something recreated the container. {{ caddy_path }}/config is +# also bind-mounted as a *directory* at /config, and a directory mount resolves +# names at open() time, so /config/Caddyfile is always the file Ansible just +# wrote. - name: reload caddy become: true become_user: "{{ podman_user }}" ansible.builtin.command: | - podman exec caddy caddy reload --config /etc/caddy/Caddyfile + podman exec caddy caddy reload --config /config/Caddyfile --adapter caddyfile tags: - caddy - caddy-config @@ -42,3 +52,12 @@ scope: user tags: - zomboid + +- name: reload podman systemd + become: true + become_user: "{{ podman_user }}" + ansible.builtin.systemd: + daemon_reload: true + scope: user + tags: + - podman diff --git a/ansible/roles/podman/tasks/containers/base/awsddns.yml b/ansible/roles/podman/tasks/containers/base/awsddns.yml index 4970b0f..7e6ea3e 100644 --- a/ansible/roles/podman/tasks/containers/base/awsddns.yml +++ b/ansible/roles/podman/tasks/containers/base/awsddns.yml @@ -1,4 +1,14 @@ --- +# The image's PID 1 is busybox crond (ENTRYPOINT crond -f -d 8), which runs the +# /etc/periodic/15min/awsddns script every quarter hour. busybox crond only +# waitpid()s the job PIDs it is tracking -- it does no generic orphan reaping -- +# so whenever the run-parts/sh layer between crond and the script exits first, +# the script is reparented to PID 1 and stays forever. That had left +# 12 zombies across these five containers after 30 days of uptime. +# +# init: true makes podman inject catatonit as PID 1 with crond as its child, and +# catatonit reaps every orphan it inherits. Changing this recreates the +# containers, which also clears the zombies already accumulated. - import_tasks: podman/podman-check.yml vars: container_name: awsddns @@ -13,6 +23,7 @@ image: "{{ image }}" restart_policy: on-failure:3 log_driver: journald + init: true env: AWS_ZONE_TTL: 60 AWS_ZONE_ID: "{{ aws_zone_id }}" @@ -40,6 +51,7 @@ image: "{{ image }}" restart_policy: on-failure:3 log_driver: journald + init: true env: AWS_ZONE_TTL: 60 AWS_ZONE_ID: "{{ aws_skudak_zone_id }}" @@ -67,6 +79,7 @@ image: "{{ image }}" restart_policy: on-failure:3 log_driver: journald + init: true env: AWS_ZONE_TTL: 60 AWS_ZONE_ID: "{{ fulfillr_zone_id }}" @@ -96,6 +109,7 @@ image: "{{ image }}" restart_policy: on-failure:3 log_driver: journald + init: true env: AWS_ZONE_TTL: 60 AWS_ZONE_ID: "{{ fulfillr_zone_id }}" @@ -123,6 +137,7 @@ image: "{{ image }}" restart_policy: on-failure:3 log_driver: journald + init: true env: AWS_ZONE_TTL: 60 AWS_ZONE_ID: "Z07501202A6AYMHCVP50A" diff --git a/ansible/roles/podman/tasks/containers/debyltech/graylog-teardown.yml b/ansible/roles/podman/tasks/containers/debyltech/graylog-teardown.yml new file mode 100644 index 0000000..7cdb4cb --- /dev/null +++ b/ansible/roles/podman/tasks/containers/debyltech/graylog-teardown.yml @@ -0,0 +1,59 @@ +--- +# Runs in place of graylog.yml when graylog_enabled is false. +# +# A bare `when:` on the import would only stop Ansible from *creating* the +# stack; containers already on the host would keep running and their systemd +# user units would keep starting them at boot. So the disabled path has to be an +# active teardown: disable the units, then remove the containers. +# +# Volumes under {{ graylog_path }} are deliberately untouched -- indices, the +# Mongo database holding streams/pipelines/dashboards, and the node-id file all +# survive, so flipping graylog_enabled back to true resumes where this left off. + +# graylog is stopped first: its unit `requires` the other two, and tearing down +# a dependency out from under it makes the shutdown noisy for no reason. +- name: stop and disable graylog stack systemd units + become: true + become_user: "{{ podman_user }}" + ansible.builtin.systemd: + name: "{{ item }}.service" + enabled: false + state: stopped + daemon_reload: true + scope: user + loop: + - graylog + - graylog-opensearch + - graylog-mongo + # Left over from the Elasticsearch-to-OpenSearch migration; the container is + # long gone but the enabled unit still tries to start it every boot. + - graylog-elastic + register: graylog_units + failed_when: false + tags: graylog + +- name: remove graylog stack containers + become: true + become_user: "{{ podman_user }}" + containers.podman.podman_container: + name: "{{ item }}" + state: absent + loop: + - graylog + - graylog-opensearch + - graylog-mongo + tags: graylog + +- name: remove graylog stack systemd unit files + become: true + become_user: "{{ podman_user }}" + ansible.builtin.file: + path: "{{ podman_home }}/.config/systemd/user/{{ item }}.service" + state: absent + loop: + - graylog + - graylog-opensearch + - graylog-mongo + - graylog-elastic + notify: reload podman systemd + tags: graylog diff --git a/ansible/roles/podman/tasks/main.yml b/ansible/roles/podman/tasks/main.yml index f0710b3..6e49b0f 100644 --- a/ansible/roles/podman/tasks/main.yml +++ b/ansible/roles/podman/tasks/main.yml @@ -81,13 +81,13 @@ - import_tasks: containers/debyltech/fulfillr.yml vars: - image: git.debyl.io/debyltech/fulfillr:20260728.2155 + image: git.debyl.io/debyltech/fulfillr:20260825.1909 tags: debyltech, fulfillr # Staging back-office (fulfillr-dev.debyltech.com) — same image, staging Turso config. - import_tasks: containers/debyltech/fulfillr-dev.yml vars: - image: git.debyl.io/debyltech/fulfillr:20260728.2155 + image: git.debyl.io/debyltech/fulfillr:20260825.1909 tags: debyltech, fulfillr-dev - import_tasks: containers/debyltech/uptime-kuma.yml @@ -100,7 +100,10 @@ image: docker.io/louislam/uptime-kuma:2.3.2 tags: home, uptime +# GeoIP is only consumed by Graylog's enrichment pipelines, so it follows the +# same switch -- see graylog_enabled in inventories/home/hosts.yml. - import_tasks: data/geoip.yml + when: graylog_enabled | bool tags: graylog, geoip - import_tasks: containers/debyltech/graylog.yml @@ -108,16 +111,26 @@ mongo_image: docker.io/mongo:7.0 opensearch_image: docker.io/opensearchproject/opensearch:2 image: docker.io/graylog/graylog:7.0.1 + when: graylog_enabled | bool + tags: debyltech, graylog + +# The disabled path is an active teardown, not just a skipped create: without it +# the containers already on the host keep running and their systemd user units +# keep restarting them at boot. +- import_tasks: containers/debyltech/graylog-teardown.yml + when: not (graylog_enabled | bool) tags: debyltech, graylog - import_tasks: containers/home/gregtime.yml vars: - image: localhost/greg-time-bot:3.10.0 + image: localhost/greg-time-bot:3.14.1 tags: gregtime +# Gated off by default — see zomboid_enabled in roles/podman/defaults/main.yml. - import_tasks: containers/home/zomboid.yml vars: image: docker.io/cm2network/steamcmd:root + when: zomboid_enabled | bool tags: zomboid # ---------------------------------------------------------- Gitea backups diff --git a/ansible/roles/podman/templates/caddy/Caddyfile.j2 b/ansible/roles/podman/templates/caddy/Caddyfile.j2 index 3caa72f..ebb0501 100644 --- a/ansible/roles/podman/templates/caddy/Caddyfile.j2 +++ b/ansible/roles/podman/templates/caddy/Caddyfile.j2 @@ -238,6 +238,34 @@ } # Graylog Logs - {{ logs_server_name }} +{% if not (graylog_enabled | bool) %} +# Graylog is disabled (see graylog_enabled in inventories/home/hosts.yml). The +# vhost is kept rather than dropped so the name keeps resolving and its +# certificate keeps renewing; both handlers answer 503 instead of proxying to a +# port nothing is listening on, which would hang and then 502. +# +# NOTE: this means the external Lambda that POSTs to /gelf for the +# `debyltech-api` stream is being turned away. Its events are dropped, not +# queued -- nothing else records them. +{{ logs_server_name }} { + handle /gelf { + respond "Log ingest disabled" 503 + } + + handle { + respond "Graylog is disabled" 503 + } + + log { + output file /var/log/caddy/graylog.log { + roll_size {{ caddy_log_roll_size }} + roll_keep {{ caddy_log_roll_keep }} + roll_keep_for {{ caddy_log_roll_keep_for }} + } + format json + } +} +{% else %} {{ logs_server_name }} { # GELF HTTP endpoint - open for Lambda (auth via header) # Must come BEFORE ip_restricted_site to allow external access @@ -283,6 +311,7 @@ format json } } +{% endif %} # ============================================================================ # COMPLEX CONFIGURATIONS