diff --git a/ansible/roles/podman/defaults/main.yml b/ansible/roles/podman/defaults/main.yml index 0e369c2..77869c9 100644 --- a/ansible/roles/podman/defaults/main.yml +++ b/ansible/roles/podman/defaults/main.yml @@ -269,7 +269,19 @@ podman_prune_users: - "{{ git_user }}" # Keep 30 days of unused images so a rollback needs no rebuild or re-pull. podman_prune_until: 720h -podman_prune_oncalendar: "Sun *-*-* 02:00:00" + +# The CI runners were never pruned and had run away: gitea-runner reached 1205 +# images / 113 GB with 100% reclaimable, actions-runner 137 exited job +# containers. Build layers carry no rollback value, so they keep a far shorter +# window than the service stores and their exited containers are reaped too. +podman_prune_ci_users: + - gitea-runner + - actions-runner +podman_prune_ci_until: 48h + +# Daily rather than weekly: CI turns over many images a day, and a week of that +# is what let the store reach 113 GB between runs. +podman_prune_oncalendar: "*-*-* 02:00:00" # Hardened CIFS options for the TrueNAS shares (see containers/home/photos.yml). # x-systemd.automount is what makes a failed mount recoverable without a human. diff --git a/ansible/roles/podman/templates/podman-prune.sh.j2 b/ansible/roles/podman/templates/podman-prune.sh.j2 index 9ad1ca2..380b532 100644 --- a/ansible/roles/podman/templates/podman-prune.sh.j2 +++ b/ansible/roles/podman/templates/podman-prune.sh.j2 @@ -1,13 +1,26 @@ #!/bin/bash # {{ ansible_managed }} -# Weekly reclaim of unused podman images and volumes. +# Daily reclaim of unused podman images, volumes and exited job containers. # # Every image bump leaves the previous tag behind and nothing ever removed # them: this was written after finding 896 images totalling 59.6 GB, 75% of it # unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy. # -# --filter until={{ podman_prune_until }} keeps recent images so a rollback -# does not require a rebuild or re-pull. Anything older is unused AND stale. +# Two policies, because the stores serve different purposes: +# +# service users ({{ podman_prune_users | join(', ') }}) +# until={{ podman_prune_until }} keeps recent images so a rollback does not +# require a rebuild or re-pull. Containers are deliberately NOT pruned here: +# they are the live services, and reaping one that merely happens to be +# stopped would turn a transient crash into a unit that cannot start again +# until the next deploy. +# +# CI users ({{ podman_prune_ci_users | join(', ') }}) +# Build layers are throwaway and there is no rollback to protect, so these +# get a much shorter window ({{ podman_prune_ci_until }}) and their exited +# job containers are reaped too. They were never covered before: gitea- +# runner had reached 1205 images / 113 GB, 100% of it reclaimable, and it +# is the layer count that makes overlayfs lookups -- and so CI itself -- slow. # # Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts # under {{ podman_volumes }} that hold real service data -- those are @@ -20,34 +33,54 @@ TAG=podman-prune log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; } -total_before=0 -total_after=0 +# Rootless podman: -H so HOME points at the user's store, and the `cd;` +# preamble is required (see CLAUDE.md) or podman cannot find its graph root. +run() { + local u=$1 + shift + sudo -H -u "$u" bash -c \ + 'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d" + exec podman "$@"' _ "$@" +} -for u in {{ podman_prune_users | join(' ') }}; do - # Rootless podman: -H so HOME points at the user's store, and the `cd;` - # preamble is required (see CLAUDE.md) or podman cannot find its graph root. - run() { - sudo -H -u "$u" bash -c \ - 'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d" - exec podman "$@"' _ "$@" - } +# prune_user +prune_user() { + local u=$1 keep=$2 do_containers=$3 + local before after img vol con if ! id "$u" >/dev/null 2>&1; then log "user=$u status=skipped reason=no-such-user" - continue + return fi - before=$(run system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1) + before=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1) # Deliberately NOT `set -e`: a prune failing for one user must not stop the - # other, and a busy image is a normal, non-fatal outcome. - img=$(run image prune -af --filter "until={{ podman_prune_until }}" 2>&1 | tail -1) - vol=$(run volume prune -f 2>&1 | tail -1) + # others, and a busy image is a normal, non-fatal outcome. + # + # Containers are reaped BEFORE images on purpose -- an exited container pins + # the image it ran from, so pruning images first would leave those layers + # behind for another day. + con=none + if [ "$do_containers" = yes ]; then + con=$(run "$u" container prune -f --filter "until=$keep" 2>&1 | tail -1) + fi - after=$(run system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1) + img=$(run "$u" image prune -af --filter "until=$keep" 2>&1 | tail -1) + vol=$(run "$u" volume prune -f 2>&1 | tail -1) - log "user=$u images_before=$before images_after=$after" - log "user=$u image_prune=${img:-none} volume_prune=${vol:-none}" + after=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1) + + log "user=$u keep=$keep size_before=$before size_after=$after" + log "user=$u image_prune=${img:-none} volume_prune=${vol:-none} container_prune=${con:-none}" +} + +for u in {{ podman_prune_users | join(' ') }}; do + prune_user "$u" "{{ podman_prune_until }}" no +done + +for u in {{ podman_prune_ci_users | join(' ') }}; do + prune_user "$u" "{{ podman_prune_ci_until }}" yes done log "status=ok" diff --git a/ansible/roles/podman/templates/podman-prune.timer.j2 b/ansible/roles/podman/templates/podman-prune.timer.j2 index e74c66a..99b84f6 100644 --- a/ansible/roles/podman/templates/podman-prune.timer.j2 +++ b/ansible/roles/podman/templates/podman-prune.timer.j2 @@ -1,11 +1,11 @@ [Unit] -Description=Weekly podman image and volume prune +Description=Daily podman image, volume and CI container prune [Timer] OnCalendar={{ podman_prune_oncalendar | default('Sun *-*-* 02:00:00') }} RandomizedDelaySec=15m -# Weekly, and growth is one tag per deploy, so a missed run is worth catching -# up on rather than skipping. +# Growth is one tag per deploy plus every CI build, so a missed run is worth +# catching up on rather than skipping. Persistent=true [Install]