--- # Bound the journal. Container stdout all lands here (log_driver=journald) and # is shipped to Graylog by fluent-bit, so the local journal only needs to be a # buffer -- see the template for the reasoning behind the size. - name: ensure journald drop-in directory exists become: true ansible.builtin.file: path: /etc/systemd/journald.conf.d state: directory owner: root group: root mode: 0755 tags: security, service, journald - name: cap journald disk usage become: true ansible.builtin.template: src: journald-size.conf.j2 dest: /etc/systemd/journald.conf.d/99-size.conf owner: root group: root mode: 0644 notify: restart_journald tags: security, service, journald # rsyslog wrote a second full copy of the journal to /var/log/messages. It # loads imjournal, so it reads the journal directly and ForwardToSyslog=no # alone does not stop it -- the unit itself has to go. Nothing consumes those # files: fail2ban runs backend=systemd and matches on the journal # (_SYSTEMD_UNIT=sshd.service, "No file is currently monitored"), and lsof # showed only rsyslogd itself holding them open. The journal is capped and is # the system of record, so this was pure write amplification: 6 GB of rotated # copies, two weekly files of which were 2.8 GB and 2.5 GB. # Masked rather than merely disabled so a dependency cannot pull it back in. - name: disable rsyslog, which duplicated the journal to /var/log/messages become: true ansible.builtin.systemd: name: rsyslog state: stopped enabled: false masked: true tags: security, service, journald - name: find the syslog copies rsyslog left behind become: true ansible.builtin.find: paths: /var/log patterns: "messages*,secure*,cron*,maillog*" register: syslog_leftovers tags: security, service, journald - name: reclaim the syslog copies become: true ansible.builtin.file: path: "{{ item.path }}" state: absent loop: "{{ syslog_leftovers.files }}" loop_control: label: "{{ item.path }}" tags: security, service, journald # Storage tuning for the SATA SSD. See the template for why bfq is wrong here. - name: use an SSD-appropriate I/O scheduler and queue depth become: true ansible.builtin.template: src: 60-ssd-scheduler.rules.j2 dest: /etc/udev/rules.d/60-ssd-scheduler.rules owner: root group: root mode: 0644 notify: reload_udev tags: security, service, storage # Writeback was left at the defaults, which are ratios of RAM: dirty_ratio=20 # and dirty_background_ratio=10 on 31 GB means the kernel will sit on up to # 3.1 GB before it starts writing back and 6.2 GB before it blocks writers # outright. Flushing that much at once to a QLC SATA drive -- which falls to # roughly 80-160 MB/s once its SLC cache is spent -- takes tens of seconds, and # everything else stalls behind it. Capping the dirty set in absolute bytes # instead trades one long stall for frequent short ones, which is what keeps # CI and the databases responsive. - name: bound writeback so a flush cannot stall the box become: true ansible.posix.sysctl: name: "{{ item.name }}" value: "{{ item.value }}" sysctl_set: true state: present reload: true loop: - { name: vm.dirty_background_bytes, value: "{{ vm_dirty_background_bytes }}" } - { name: vm.dirty_bytes, value: "{{ vm_dirty_bytes }}" } loop_control: label: "{{ item.name }}={{ item.value }}" tags: security, service, storage - name: ensure desired services are started and enabled become: true ansible.builtin.service: name: "{{ item }}" state: started enabled: true loop: "{{ services }}" tags: security, service