perf(host): drop the duplicate syslog copy and fix the SSD I/O path
Three findings from investigating sustained I/O pressure on the root SSD.
Grouped because the logging and storage edits land in the same task file.
rsyslog was writing a second complete copy of the journal to
/var/log/messages: 6.7 GB of rotated copies, two weekly files of which were
2.8 GB and 2.5 GB. It loads imjournal, so it reads the journal directly and
ForwardToSyslog=no alone does not stop it -- the unit itself has to go.
Verified nothing consumes those files first: fail2ban runs backend=systemd and
matches on the journal ("No file is currently monitored"), and lsof showed only
rsyslogd holding them. Measured afterwards: writes 46 -> 23 GB/day, /var/log
6.7 GB -> 655 MB, journald still capturing container stdout.
The SSD was on bfq, which fedora's stock 60-block-scheduler.rules picks for any
rotational=0 disk. bfq is built for spinning disks and desktop interactivity:
it costs CPU per request, lets reads queue behind write bursts, and hard-caps
nr_requests at 64. With 26 containers and two CI runners writing at once that
is the wrong trade. mq-deadline rather than none because this is SATA with a
32-deep NCQ queue, not NVMe -- the merging and the read-expiry deadline both
earn their place.
Writeback was at the stock percent-of-RAM ratios, so on 31 GB the kernel would
sit on 3.1 GB before starting writeback and 6.2 GB before blocking writers.
Flushing that to a QLC drive that falls to ~80-160 MB/s once its SLC cache is
spent takes tens of seconds with everything stalled behind it. Capped in
absolute bytes instead: one long stall traded for frequent short ones.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
9e136ce903
commit
f21be79452
@@ -21,3 +21,12 @@ services:
|
|||||||
- podman
|
- podman
|
||||||
- fail2ban
|
- fail2ban
|
||||||
- systemd-timesyncd
|
- systemd-timesyncd
|
||||||
|
|
||||||
|
# Storage tuning for the SATA SSD (see tasks/service.yml and the udev template).
|
||||||
|
ssd_io_scheduler: mq-deadline
|
||||||
|
ssd_nr_requests: "256"
|
||||||
|
# 256 MB before background writeback starts, 1 GB before writers block. Absolute
|
||||||
|
# bytes rather than the default percent-of-RAM ratios, which scale to multi-GB
|
||||||
|
# stalls on a 31 GB host.
|
||||||
|
vm_dirty_background_bytes: "268435456"
|
||||||
|
vm_dirty_bytes: "1073741824"
|
||||||
|
|||||||
@@ -26,3 +26,13 @@
|
|||||||
ansible.builtin.systemd:
|
ansible.builtin.systemd:
|
||||||
name: systemd-journald
|
name: systemd-journald
|
||||||
state: restarted
|
state: restarted
|
||||||
|
|
||||||
|
# --reload-rules alone only affects devices that appear later; the root disk is
|
||||||
|
# already attached, so trigger a change event to apply the rule now rather than
|
||||||
|
# at the next reboot.
|
||||||
|
- name: reload_udev
|
||||||
|
become: true
|
||||||
|
ansible.builtin.shell: |
|
||||||
|
udevadm control --reload-rules
|
||||||
|
udevadm trigger --subsystem-match=block --action=change
|
||||||
|
changed_when: true
|
||||||
|
|||||||
@@ -23,6 +23,77 @@
|
|||||||
notify: restart_journald
|
notify: restart_journald
|
||||||
tags: security, service, journald
|
tags: security, service, journald
|
||||||
|
|
||||||
|
# rsyslog wrote a second full copy of the journal to /var/log/messages. It
|
||||||
|
# loads imjournal, so it reads the journal directly and ForwardToSyslog=no
|
||||||
|
# alone does not stop it -- the unit itself has to go. Nothing consumes those
|
||||||
|
# files: fail2ban runs backend=systemd and matches on the journal
|
||||||
|
# (_SYSTEMD_UNIT=sshd.service, "No file is currently monitored"), and lsof
|
||||||
|
# showed only rsyslogd itself holding them open. The journal is capped and is
|
||||||
|
# the system of record, so this was pure write amplification: 6 GB of rotated
|
||||||
|
# copies, two weekly files of which were 2.8 GB and 2.5 GB.
|
||||||
|
# Masked rather than merely disabled so a dependency cannot pull it back in.
|
||||||
|
- name: disable rsyslog, which duplicated the journal to /var/log/messages
|
||||||
|
become: true
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: rsyslog
|
||||||
|
state: stopped
|
||||||
|
enabled: false
|
||||||
|
masked: true
|
||||||
|
tags: security, service, journald
|
||||||
|
|
||||||
|
- name: find the syslog copies rsyslog left behind
|
||||||
|
become: true
|
||||||
|
ansible.builtin.find:
|
||||||
|
paths: /var/log
|
||||||
|
patterns: "messages*,secure*,cron*,maillog*"
|
||||||
|
register: syslog_leftovers
|
||||||
|
tags: security, service, journald
|
||||||
|
|
||||||
|
- name: reclaim the syslog copies
|
||||||
|
become: true
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item.path }}"
|
||||||
|
state: absent
|
||||||
|
loop: "{{ syslog_leftovers.files }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.path }}"
|
||||||
|
tags: security, service, journald
|
||||||
|
|
||||||
|
# Storage tuning for the SATA SSD. See the template for why bfq is wrong here.
|
||||||
|
- name: use an SSD-appropriate I/O scheduler and queue depth
|
||||||
|
become: true
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: 60-ssd-scheduler.rules.j2
|
||||||
|
dest: /etc/udev/rules.d/60-ssd-scheduler.rules
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: 0644
|
||||||
|
notify: reload_udev
|
||||||
|
tags: security, service, storage
|
||||||
|
|
||||||
|
# Writeback was left at the defaults, which are ratios of RAM: dirty_ratio=20
|
||||||
|
# and dirty_background_ratio=10 on 31 GB means the kernel will sit on up to
|
||||||
|
# 3.1 GB before it starts writing back and 6.2 GB before it blocks writers
|
||||||
|
# outright. Flushing that much at once to a QLC SATA drive -- which falls to
|
||||||
|
# roughly 80-160 MB/s once its SLC cache is spent -- takes tens of seconds, and
|
||||||
|
# everything else stalls behind it. Capping the dirty set in absolute bytes
|
||||||
|
# instead trades one long stall for frequent short ones, which is what keeps
|
||||||
|
# CI and the databases responsive.
|
||||||
|
- name: bound writeback so a flush cannot stall the box
|
||||||
|
become: true
|
||||||
|
ansible.posix.sysctl:
|
||||||
|
name: "{{ item.name }}"
|
||||||
|
value: "{{ item.value }}"
|
||||||
|
sysctl_set: true
|
||||||
|
state: present
|
||||||
|
reload: true
|
||||||
|
loop:
|
||||||
|
- { name: vm.dirty_background_bytes, value: "{{ vm_dirty_background_bytes }}" }
|
||||||
|
- { name: vm.dirty_bytes, value: "{{ vm_dirty_bytes }}" }
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name }}={{ item.value }}"
|
||||||
|
tags: security, service, storage
|
||||||
|
|
||||||
- name: ensure desired services are started and enabled
|
- name: ensure desired services are started and enabled
|
||||||
become: true
|
become: true
|
||||||
ansible.builtin.service:
|
ansible.builtin.service:
|
||||||
|
|||||||
@@ -0,0 +1,18 @@
|
|||||||
|
# {{ ansible_managed }}
|
||||||
|
# Fedora's /usr/lib/udev/rules.d/60-block-scheduler.rules picks bfq for every
|
||||||
|
# rotational=0 SATA disk. bfq is a fairness scheduler built for spinning disks
|
||||||
|
# and desktop interactivity: it costs real CPU per request and, with 26
|
||||||
|
# containers plus two CI runners all writing at once, it lets reads queue
|
||||||
|
# behind write bursts. That is the shape of the stalls seen here -- I/O
|
||||||
|
# pressure spiking to 76% while the CPU sat nearly idle.
|
||||||
|
#
|
||||||
|
# mq-deadline instead of none: this is a SATA SSD with a 32-deep NCBQ queue,
|
||||||
|
# not an NVMe device with its own deep queues, so the request merging and the
|
||||||
|
# read-expiry deadline are both worth having. The deadline is what stops reads
|
||||||
|
# starving behind a QLC write burst.
|
||||||
|
#
|
||||||
|
# bfq also hard-caps nr_requests at 64; mq-deadline allows a deeper queue,
|
||||||
|
# which is what lets concurrent container and CI I/O actually overlap.
|
||||||
|
ACTION=="add|change", KERNEL=="sd[a-z]", ATTR{queue/rotational}=="0", \
|
||||||
|
ATTR{queue/scheduler}="{{ ssd_io_scheduler }}", \
|
||||||
|
ATTR{queue/nr_requests}="{{ ssd_nr_requests }}"
|
||||||
@@ -20,3 +20,9 @@
|
|||||||
{% endif %}
|
{% endif %}
|
||||||
[Journal]
|
[Journal]
|
||||||
SystemMaxUse={{ journald_max_use | default('500M') }}
|
SystemMaxUse={{ journald_max_use | default('500M') }}
|
||||||
|
|
||||||
|
# Nothing should be forwarded to syslog: rsyslog is disabled (see
|
||||||
|
# tasks/service.yml) because it duplicated the whole journal into
|
||||||
|
# /var/log/messages. This also closes the imuxsock path so anything that
|
||||||
|
# logs via logger(1) still lands in the journal and nowhere else.
|
||||||
|
ForwardToSyslog=no
|
||||||
|
|||||||
Reference in New Issue
Block a user