fix(podman-prune): reap the CI stores, which had reached 113 GB
podman_prune_users listed only the podman and git users, so the two stores that turn over fastest were never touched. gitea-runner had reached 1205 images / 113.1 GB with 100% of it reclaimable, and actions-runner had 137 exited job containers. That layer count is what makes overlayfs lookups -- and so CI itself -- slow; the disk was the lesser problem. Split into two policies, because the stores are not the same kind of thing: service users keep the 30-day rollback window, and their containers are deliberately NOT pruned. They are the live services, and reaping one that merely happens to be stopped would turn a transient crash into a unit that cannot start again until the next deploy. CI users get 48h and their exited job containers reaped too. Build layers carry no rollback value. Containers are reaped BEFORE images on purpose: an exited container pins the image it ran from, so pruning images first would leave those layers behind for another day. Timer moved weekly -> daily; a week of CI turnover is what let the store reach 113 GB between runs. Persistent=true is kept so a missed run catches up. First run reclaimed 134 GB: gitea-runner 113.1 -> 4.2 GB, actions-runner 7.6 GB -> 0, podman 25.7 -> 15.1 GB. Disk 449G -> 315G, all 26 containers up. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
f21be79452
commit
2a6390c0ad
@@ -269,7 +269,19 @@ podman_prune_users:
|
|||||||
- "{{ git_user }}"
|
- "{{ git_user }}"
|
||||||
# Keep 30 days of unused images so a rollback needs no rebuild or re-pull.
|
# Keep 30 days of unused images so a rollback needs no rebuild or re-pull.
|
||||||
podman_prune_until: 720h
|
podman_prune_until: 720h
|
||||||
podman_prune_oncalendar: "Sun *-*-* 02:00:00"
|
|
||||||
|
# The CI runners were never pruned and had run away: gitea-runner reached 1205
|
||||||
|
# images / 113 GB with 100% reclaimable, actions-runner 137 exited job
|
||||||
|
# containers. Build layers carry no rollback value, so they keep a far shorter
|
||||||
|
# window than the service stores and their exited containers are reaped too.
|
||||||
|
podman_prune_ci_users:
|
||||||
|
- gitea-runner
|
||||||
|
- actions-runner
|
||||||
|
podman_prune_ci_until: 48h
|
||||||
|
|
||||||
|
# Daily rather than weekly: CI turns over many images a day, and a week of that
|
||||||
|
# is what let the store reach 113 GB between runs.
|
||||||
|
podman_prune_oncalendar: "*-*-* 02:00:00"
|
||||||
|
|
||||||
# Hardened CIFS options for the TrueNAS shares (see containers/home/photos.yml).
|
# Hardened CIFS options for the TrueNAS shares (see containers/home/photos.yml).
|
||||||
# x-systemd.automount is what makes a failed mount recoverable without a human.
|
# x-systemd.automount is what makes a failed mount recoverable without a human.
|
||||||
|
|||||||
@@ -1,13 +1,26 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
# {{ ansible_managed }}
|
# {{ ansible_managed }}
|
||||||
# Weekly reclaim of unused podman images and volumes.
|
# Daily reclaim of unused podman images, volumes and exited job containers.
|
||||||
#
|
#
|
||||||
# Every image bump leaves the previous tag behind and nothing ever removed
|
# Every image bump leaves the previous tag behind and nothing ever removed
|
||||||
# them: this was written after finding 896 images totalling 59.6 GB, 75% of it
|
# them: this was written after finding 896 images totalling 59.6 GB, 75% of it
|
||||||
# unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy.
|
# unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy.
|
||||||
#
|
#
|
||||||
# --filter until={{ podman_prune_until }} keeps recent images so a rollback
|
# Two policies, because the stores serve different purposes:
|
||||||
# does not require a rebuild or re-pull. Anything older is unused AND stale.
|
#
|
||||||
|
# service users ({{ podman_prune_users | join(', ') }})
|
||||||
|
# until={{ podman_prune_until }} keeps recent images so a rollback does not
|
||||||
|
# require a rebuild or re-pull. Containers are deliberately NOT pruned here:
|
||||||
|
# they are the live services, and reaping one that merely happens to be
|
||||||
|
# stopped would turn a transient crash into a unit that cannot start again
|
||||||
|
# until the next deploy.
|
||||||
|
#
|
||||||
|
# CI users ({{ podman_prune_ci_users | join(', ') }})
|
||||||
|
# Build layers are throwaway and there is no rollback to protect, so these
|
||||||
|
# get a much shorter window ({{ podman_prune_ci_until }}) and their exited
|
||||||
|
# job containers are reaped too. They were never covered before: gitea-
|
||||||
|
# runner had reached 1205 images / 113 GB, 100% of it reclaimable, and it
|
||||||
|
# is the layer count that makes overlayfs lookups -- and so CI itself -- slow.
|
||||||
#
|
#
|
||||||
# Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts
|
# Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts
|
||||||
# under {{ podman_volumes }} that hold real service data -- those are
|
# under {{ podman_volumes }} that hold real service data -- those are
|
||||||
@@ -20,34 +33,54 @@ TAG=podman-prune
|
|||||||
|
|
||||||
log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; }
|
log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; }
|
||||||
|
|
||||||
total_before=0
|
|
||||||
total_after=0
|
|
||||||
|
|
||||||
for u in {{ podman_prune_users | join(' ') }}; do
|
|
||||||
# Rootless podman: -H so HOME points at the user's store, and the `cd;`
|
# Rootless podman: -H so HOME points at the user's store, and the `cd;`
|
||||||
# preamble is required (see CLAUDE.md) or podman cannot find its graph root.
|
# preamble is required (see CLAUDE.md) or podman cannot find its graph root.
|
||||||
run() {
|
run() {
|
||||||
|
local u=$1
|
||||||
|
shift
|
||||||
sudo -H -u "$u" bash -c \
|
sudo -H -u "$u" bash -c \
|
||||||
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
|
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
|
||||||
exec podman "$@"' _ "$@"
|
exec podman "$@"' _ "$@"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# prune_user <user> <until> <prune_containers: yes|no>
|
||||||
|
prune_user() {
|
||||||
|
local u=$1 keep=$2 do_containers=$3
|
||||||
|
local before after img vol con
|
||||||
|
|
||||||
if ! id "$u" >/dev/null 2>&1; then
|
if ! id "$u" >/dev/null 2>&1; then
|
||||||
log "user=$u status=skipped reason=no-such-user"
|
log "user=$u status=skipped reason=no-such-user"
|
||||||
continue
|
return
|
||||||
fi
|
fi
|
||||||
|
|
||||||
before=$(run system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
before=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
||||||
|
|
||||||
# Deliberately NOT `set -e`: a prune failing for one user must not stop the
|
# Deliberately NOT `set -e`: a prune failing for one user must not stop the
|
||||||
# other, and a busy image is a normal, non-fatal outcome.
|
# others, and a busy image is a normal, non-fatal outcome.
|
||||||
img=$(run image prune -af --filter "until={{ podman_prune_until }}" 2>&1 | tail -1)
|
#
|
||||||
vol=$(run volume prune -f 2>&1 | tail -1)
|
# Containers are reaped BEFORE images on purpose -- an exited container pins
|
||||||
|
# the image it ran from, so pruning images first would leave those layers
|
||||||
|
# behind for another day.
|
||||||
|
con=none
|
||||||
|
if [ "$do_containers" = yes ]; then
|
||||||
|
con=$(run "$u" container prune -f --filter "until=$keep" 2>&1 | tail -1)
|
||||||
|
fi
|
||||||
|
|
||||||
after=$(run system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
img=$(run "$u" image prune -af --filter "until=$keep" 2>&1 | tail -1)
|
||||||
|
vol=$(run "$u" volume prune -f 2>&1 | tail -1)
|
||||||
|
|
||||||
log "user=$u images_before=$before images_after=$after"
|
after=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
||||||
log "user=$u image_prune=${img:-none} volume_prune=${vol:-none}"
|
|
||||||
|
log "user=$u keep=$keep size_before=$before size_after=$after"
|
||||||
|
log "user=$u image_prune=${img:-none} volume_prune=${vol:-none} container_prune=${con:-none}"
|
||||||
|
}
|
||||||
|
|
||||||
|
for u in {{ podman_prune_users | join(' ') }}; do
|
||||||
|
prune_user "$u" "{{ podman_prune_until }}" no
|
||||||
|
done
|
||||||
|
|
||||||
|
for u in {{ podman_prune_ci_users | join(' ') }}; do
|
||||||
|
prune_user "$u" "{{ podman_prune_ci_until }}" yes
|
||||||
done
|
done
|
||||||
|
|
||||||
log "status=ok"
|
log "status=ok"
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
[Unit]
|
[Unit]
|
||||||
Description=Weekly podman image and volume prune
|
Description=Daily podman image, volume and CI container prune
|
||||||
|
|
||||||
[Timer]
|
[Timer]
|
||||||
OnCalendar={{ podman_prune_oncalendar | default('Sun *-*-* 02:00:00') }}
|
OnCalendar={{ podman_prune_oncalendar | default('Sun *-*-* 02:00:00') }}
|
||||||
RandomizedDelaySec=15m
|
RandomizedDelaySec=15m
|
||||||
# Weekly, and growth is one tag per deploy, so a missed run is worth catching
|
# Growth is one tag per deploy plus every CI build, so a missed run is worth
|
||||||
# up on rather than skipping.
|
# catching up on rather than skipping.
|
||||||
Persistent=true
|
Persistent=true
|
||||||
|
|
||||||
[Install]
|
[Install]
|
||||||
|
|||||||
Reference in New Issue
Block a user