#!/bin/bash # {{ ansible_managed }} # Daily reclaim of unused podman images, volumes and exited job containers. # # Every image bump leaves the previous tag behind and nothing ever removed # them: this was written after finding 896 images totalling 59.6 GB, 75% of it # unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy. # # Two policies, because the stores serve different purposes: # # service users ({{ podman_prune_users | join(', ') }}) # until={{ podman_prune_until }} keeps recent images so a rollback does not # require a rebuild or re-pull. Containers are deliberately NOT pruned here: # they are the live services, and reaping one that merely happens to be # stopped would turn a transient crash into a unit that cannot start again # until the next deploy. # # CI users ({{ podman_prune_ci_users | join(', ') }}) # Build layers are throwaway and there is no rollback to protect, so these # get a much shorter window ({{ podman_prune_ci_until }}) and their exited # job containers are reaped too. They were never covered before: gitea- # runner had reached 1205 images / 113 GB, 100% of it reclaimable, and it # is the layer count that makes overlayfs lookups -- and so CI itself -- slow. # # Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts # under {{ podman_volumes }} that hold real service data -- those are # directories on the host and podman does not know about them. The dangling # ones seen in practice were 804 MB copies of Nextcloud's /var/www/html left # by container recreations, which are image content, not data. set -uo pipefail TAG=podman-prune log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; } # Rootless podman: -H so HOME points at the user's store, and the `cd;` # preamble is required (see CLAUDE.md) or podman cannot find its graph root. run() { local u=$1 shift sudo -H -u "$u" bash -c \ 'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d" exec podman "$@"' _ "$@" } # prune_user [image prune filter...] prune_user() { local u=$1 keep=$2 do_containers=$3 shift 3 local before after img vol con if ! id "$u" >/dev/null 2>&1; then log "user=$u status=skipped reason=no-such-user" return fi before=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1) # Deliberately NOT `set -e`: a prune failing for one user must not stop the # others, and a busy image is a normal, non-fatal outcome. # # Containers are reaped BEFORE images on purpose -- an exited container pins # the image it ran from, so pruning images first would leave those layers # behind for another day. con=none if [ "$do_containers" = yes ]; then con=$(run "$u" container prune -f --filter "until=$keep" 2>&1 | tail -1) fi img=$(run "$u" image prune -af --filter "until=$keep" "$@" 2>&1 | tail -1) vol=$(run "$u" volume prune -f 2>&1 | tail -1) after=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1) log "user=$u keep=$keep size_before=$before size_after=$after" log "user=$u image_prune=${img:-none} volume_prune=${vol:-none} container_prune=${con:-none}" } for u in {{ podman_prune_users | join(' ') }}; do prune_user "$u" "{{ podman_prune_until }}" no done for u in {{ podman_prune_ci_users | join(' ') }}; do # CI base images (gitea-ci, -espidf, -platformio) carry the keep label: they # are rebuilt or re-pulled from the registry only when missing, so pruning # them just forces a multi-GB re-download on the next job. prune_user "$u" "{{ podman_prune_ci_until }}" yes --filter "label!={{ podman_prune_ci_keep_label }}" done log "status=ok"