From f21be79452fdff12349a771285d56f01e554a4b3 Mon Sep 17 00:00:00 2001 From: Bastian de Byl Date: Fri, 28 Aug 2026 11:39:05 -0400 Subject: [PATCH] perf(host): drop the duplicate syslog copy and fix the SSD I/O path Three findings from investigating sustained I/O pressure on the root SSD. Grouped because the logging and storage edits land in the same task file. rsyslog was writing a second complete copy of the journal to /var/log/messages: 6.7 GB of rotated copies, two weekly files of which were 2.8 GB and 2.5 GB. It loads imjournal, so it reads the journal directly and ForwardToSyslog=no alone does not stop it -- the unit itself has to go. Verified nothing consumes those files first: fail2ban runs backend=systemd and matches on the journal ("No file is currently monitored"), and lsof showed only rsyslogd holding them. Measured afterwards: writes 46 -> 23 GB/day, /var/log 6.7 GB -> 655 MB, journald still capturing container stdout. The SSD was on bfq, which fedora's stock 60-block-scheduler.rules picks for any rotational=0 disk. bfq is built for spinning disks and desktop interactivity: it costs CPU per request, lets reads queue behind write bursts, and hard-caps nr_requests at 64. With 26 containers and two CI runners writing at once that is the wrong trade. mq-deadline rather than none because this is SATA with a 32-deep NCQ queue, not NVMe -- the merging and the read-expiry deadline both earn their place. Writeback was at the stock percent-of-RAM ratios, so on 31 GB the kernel would sit on 3.1 GB before starting writeback and 6.2 GB before blocking writers. Flushing that to a QLC drive that falls to ~80-160 MB/s once its SLC cache is spent takes tens of seconds with everything stalled behind it. Capped in absolute bytes instead: one long stall traded for frequent short ones. Co-Authored-By: Claude Opus 5 --- ansible/roles/common/defaults/main.yml | 9 +++ ansible/roles/common/handlers/main.yml | 10 +++ ansible/roles/common/tasks/service.yml | 71 +++++++++++++++++++ .../templates/60-ssd-scheduler.rules.j2 | 18 +++++ .../common/templates/journald-size.conf.j2 | 6 ++ 5 files changed, 114 insertions(+) create mode 100644 ansible/roles/common/templates/60-ssd-scheduler.rules.j2 diff --git a/ansible/roles/common/defaults/main.yml b/ansible/roles/common/defaults/main.yml index 98d1b24..dd07db7 100644 --- a/ansible/roles/common/defaults/main.yml +++ b/ansible/roles/common/defaults/main.yml @@ -21,3 +21,12 @@ services: - podman - fail2ban - systemd-timesyncd + +# Storage tuning for the SATA SSD (see tasks/service.yml and the udev template). +ssd_io_scheduler: mq-deadline +ssd_nr_requests: "256" +# 256 MB before background writeback starts, 1 GB before writers block. Absolute +# bytes rather than the default percent-of-RAM ratios, which scale to multi-GB +# stalls on a 31 GB host. +vm_dirty_background_bytes: "268435456" +vm_dirty_bytes: "1073741824" diff --git a/ansible/roles/common/handlers/main.yml b/ansible/roles/common/handlers/main.yml index cf0ef30..2b9730e 100644 --- a/ansible/roles/common/handlers/main.yml +++ b/ansible/roles/common/handlers/main.yml @@ -26,3 +26,13 @@ ansible.builtin.systemd: name: systemd-journald state: restarted + +# --reload-rules alone only affects devices that appear later; the root disk is +# already attached, so trigger a change event to apply the rule now rather than +# at the next reboot. +- name: reload_udev + become: true + ansible.builtin.shell: | + udevadm control --reload-rules + udevadm trigger --subsystem-match=block --action=change + changed_when: true diff --git a/ansible/roles/common/tasks/service.yml b/ansible/roles/common/tasks/service.yml index cdeb176..76be757 100644 --- a/ansible/roles/common/tasks/service.yml +++ b/ansible/roles/common/tasks/service.yml @@ -23,6 +23,77 @@ notify: restart_journald tags: security, service, journald +# rsyslog wrote a second full copy of the journal to /var/log/messages. It +# loads imjournal, so it reads the journal directly and ForwardToSyslog=no +# alone does not stop it -- the unit itself has to go. Nothing consumes those +# files: fail2ban runs backend=systemd and matches on the journal +# (_SYSTEMD_UNIT=sshd.service, "No file is currently monitored"), and lsof +# showed only rsyslogd itself holding them open. The journal is capped and is +# the system of record, so this was pure write amplification: 6 GB of rotated +# copies, two weekly files of which were 2.8 GB and 2.5 GB. +# Masked rather than merely disabled so a dependency cannot pull it back in. +- name: disable rsyslog, which duplicated the journal to /var/log/messages + become: true + ansible.builtin.systemd: + name: rsyslog + state: stopped + enabled: false + masked: true + tags: security, service, journald + +- name: find the syslog copies rsyslog left behind + become: true + ansible.builtin.find: + paths: /var/log + patterns: "messages*,secure*,cron*,maillog*" + register: syslog_leftovers + tags: security, service, journald + +- name: reclaim the syslog copies + become: true + ansible.builtin.file: + path: "{{ item.path }}" + state: absent + loop: "{{ syslog_leftovers.files }}" + loop_control: + label: "{{ item.path }}" + tags: security, service, journald + +# Storage tuning for the SATA SSD. See the template for why bfq is wrong here. +- name: use an SSD-appropriate I/O scheduler and queue depth + become: true + ansible.builtin.template: + src: 60-ssd-scheduler.rules.j2 + dest: /etc/udev/rules.d/60-ssd-scheduler.rules + owner: root + group: root + mode: 0644 + notify: reload_udev + tags: security, service, storage + +# Writeback was left at the defaults, which are ratios of RAM: dirty_ratio=20 +# and dirty_background_ratio=10 on 31 GB means the kernel will sit on up to +# 3.1 GB before it starts writing back and 6.2 GB before it blocks writers +# outright. Flushing that much at once to a QLC SATA drive -- which falls to +# roughly 80-160 MB/s once its SLC cache is spent -- takes tens of seconds, and +# everything else stalls behind it. Capping the dirty set in absolute bytes +# instead trades one long stall for frequent short ones, which is what keeps +# CI and the databases responsive. +- name: bound writeback so a flush cannot stall the box + become: true + ansible.posix.sysctl: + name: "{{ item.name }}" + value: "{{ item.value }}" + sysctl_set: true + state: present + reload: true + loop: + - { name: vm.dirty_background_bytes, value: "{{ vm_dirty_background_bytes }}" } + - { name: vm.dirty_bytes, value: "{{ vm_dirty_bytes }}" } + loop_control: + label: "{{ item.name }}={{ item.value }}" + tags: security, service, storage + - name: ensure desired services are started and enabled become: true ansible.builtin.service: diff --git a/ansible/roles/common/templates/60-ssd-scheduler.rules.j2 b/ansible/roles/common/templates/60-ssd-scheduler.rules.j2 new file mode 100644 index 0000000..60c1118 --- /dev/null +++ b/ansible/roles/common/templates/60-ssd-scheduler.rules.j2 @@ -0,0 +1,18 @@ +# {{ ansible_managed }} +# Fedora's /usr/lib/udev/rules.d/60-block-scheduler.rules picks bfq for every +# rotational=0 SATA disk. bfq is a fairness scheduler built for spinning disks +# and desktop interactivity: it costs real CPU per request and, with 26 +# containers plus two CI runners all writing at once, it lets reads queue +# behind write bursts. That is the shape of the stalls seen here -- I/O +# pressure spiking to 76% while the CPU sat nearly idle. +# +# mq-deadline instead of none: this is a SATA SSD with a 32-deep NCBQ queue, +# not an NVMe device with its own deep queues, so the request merging and the +# read-expiry deadline are both worth having. The deadline is what stops reads +# starving behind a QLC write burst. +# +# bfq also hard-caps nr_requests at 64; mq-deadline allows a deeper queue, +# which is what lets concurrent container and CI I/O actually overlap. +ACTION=="add|change", KERNEL=="sd[a-z]", ATTR{queue/rotational}=="0", \ + ATTR{queue/scheduler}="{{ ssd_io_scheduler }}", \ + ATTR{queue/nr_requests}="{{ ssd_nr_requests }}" diff --git a/ansible/roles/common/templates/journald-size.conf.j2 b/ansible/roles/common/templates/journald-size.conf.j2 index 4ee4abb..ebdb78c 100644 --- a/ansible/roles/common/templates/journald-size.conf.j2 +++ b/ansible/roles/common/templates/journald-size.conf.j2 @@ -20,3 +20,9 @@ {% endif %} [Journal] SystemMaxUse={{ journald_max_use | default('500M') }} + +# Nothing should be forwarded to syslog: rsyslog is disabled (see +# tasks/service.yml) because it duplicated the whole journal into +# /var/log/messages. This also closes the imuxsock path so anything that +# logs via logger(1) still lands in the journal and nowhere else. +ForwardToSyslog=no