Files
deploy_home/ansible/roles/podman/templates/podman-prune.sh.j2
T
Bastian de BylandClaude Opus 5 42e6f5271d fix(gitea-actions): serve CI images from the Gitea registry
The CI job images only existed under localhost/ in the gitea-runner store,
and the nightly CI prune deletes any image older than 48h that no container
holds. After every idle stretch CI failed in 0-1s pulling
localhost/gitea-ci:latest until the role was re-run and the images rebuilt.

- Build under git.debyl.io/gitbot/..., push after every run, and pull from the
  registry instead of rebuilding when the Containerfile is unchanged.
- Log gitea-runner in via ~/.docker/config.json, which both act_runner (job
  image pulls) and podman read.
- Label the base images io.debyl.ci-base and skip that label in the CI prune;
  its `until` counts from build time, so a re-pulled image would otherwise be
  deleted again the next night.

Workflows pinning `container: image: localhost/gitea-ci-*` must move to the
registry paths.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 16:40:57 -04:00

91 lines
3.7 KiB
Django/Jinja

#!/bin/bash
# {{ ansible_managed }}
# Daily reclaim of unused podman images, volumes and exited job containers.
#
# Every image bump leaves the previous tag behind and nothing ever removed
# them: this was written after finding 896 images totalling 59.6 GB, 75% of it
# unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy.
#
# Two policies, because the stores serve different purposes:
#
# service users ({{ podman_prune_users | join(', ') }})
# until={{ podman_prune_until }} keeps recent images so a rollback does not
# require a rebuild or re-pull. Containers are deliberately NOT pruned here:
# they are the live services, and reaping one that merely happens to be
# stopped would turn a transient crash into a unit that cannot start again
# until the next deploy.
#
# CI users ({{ podman_prune_ci_users | join(', ') }})
# Build layers are throwaway and there is no rollback to protect, so these
# get a much shorter window ({{ podman_prune_ci_until }}) and their exited
# job containers are reaped too. They were never covered before: gitea-
# runner had reached 1205 images / 113 GB, 100% of it reclaimable, and it
# is the layer count that makes overlayfs lookups -- and so CI itself -- slow.
#
# Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts
# under {{ podman_volumes }} that hold real service data -- those are
# directories on the host and podman does not know about them. The dangling
# ones seen in practice were 804 MB copies of Nextcloud's /var/www/html left
# by container recreations, which are image content, not data.
set -uo pipefail
TAG=podman-prune
log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; }
# Rootless podman: -H so HOME points at the user's store, and the `cd;`
# preamble is required (see CLAUDE.md) or podman cannot find its graph root.
run() {
local u=$1
shift
sudo -H -u "$u" bash -c \
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
exec podman "$@"' _ "$@"
}
# prune_user <user> <until> <prune_containers: yes|no> [image prune filter...]
prune_user() {
local u=$1 keep=$2 do_containers=$3
shift 3
local before after img vol con
if ! id "$u" >/dev/null 2>&1; then
log "user=$u status=skipped reason=no-such-user"
return
fi
before=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
# Deliberately NOT `set -e`: a prune failing for one user must not stop the
# others, and a busy image is a normal, non-fatal outcome.
#
# Containers are reaped BEFORE images on purpose -- an exited container pins
# the image it ran from, so pruning images first would leave those layers
# behind for another day.
con=none
if [ "$do_containers" = yes ]; then
con=$(run "$u" container prune -f --filter "until=$keep" 2>&1 | tail -1)
fi
img=$(run "$u" image prune -af --filter "until=$keep" "$@" 2>&1 | tail -1)
vol=$(run "$u" volume prune -f 2>&1 | tail -1)
after=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
log "user=$u keep=$keep size_before=$before size_after=$after"
log "user=$u image_prune=${img:-none} volume_prune=${vol:-none} container_prune=${con:-none}"
}
for u in {{ podman_prune_users | join(' ') }}; do
prune_user "$u" "{{ podman_prune_until }}" no
done
for u in {{ podman_prune_ci_users | join(' ') }}; do
# CI base images (gitea-ci, -espidf, -platformio) carry the keep label: they
# are rebuilt or re-pulled from the registry only when missing, so pruning
# them just forces a multi-GB re-download on the next job.
prune_user "$u" "{{ podman_prune_ci_until }}" yes --filter "label!={{ podman_prune_ci_keep_label }}"
done
log "status=ok"