The CI job images only existed under localhost/ in the gitea-runner store, and the nightly CI prune deletes any image older than 48h that no container holds. After every idle stretch CI failed in 0-1s pulling localhost/gitea-ci:latest until the role was re-run and the images rebuilt. - Build under git.debyl.io/gitbot/..., push after every run, and pull from the registry instead of rebuilding when the Containerfile is unchanged. - Log gitea-runner in via ~/.docker/config.json, which both act_runner (job image pulls) and podman read. - Label the base images io.debyl.ci-base and skip that label in the CI prune; its `until` counts from build time, so a re-pulled image would otherwise be deleted again the next night. Workflows pinning `container: image: localhost/gitea-ci-*` must move to the registry paths. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
91 lines
3.7 KiB
Django/Jinja
91 lines
3.7 KiB
Django/Jinja
#!/bin/bash
|
|
# {{ ansible_managed }}
|
|
# Daily reclaim of unused podman images, volumes and exited job containers.
|
|
#
|
|
# Every image bump leaves the previous tag behind and nothing ever removed
|
|
# them: this was written after finding 896 images totalling 59.6 GB, 75% of it
|
|
# unused -- 94 tags of greg-time-bot and 73 of fulfillr, one per deploy.
|
|
#
|
|
# Two policies, because the stores serve different purposes:
|
|
#
|
|
# service users ({{ podman_prune_users | join(', ') }})
|
|
# until={{ podman_prune_until }} keeps recent images so a rollback does not
|
|
# require a rebuild or re-pull. Containers are deliberately NOT pruned here:
|
|
# they are the live services, and reaping one that merely happens to be
|
|
# stopped would turn a transient crash into a unit that cannot start again
|
|
# until the next deploy.
|
|
#
|
|
# CI users ({{ podman_prune_ci_users | join(', ') }})
|
|
# Build layers are throwaway and there is no rollback to protect, so these
|
|
# get a much shorter window ({{ podman_prune_ci_until }}) and their exited
|
|
# job containers are reaped too. They were never covered before: gitea-
|
|
# runner had reached 1205 images / 113 GB, 100% of it reclaimable, and it
|
|
# is the layer count that makes overlayfs lookups -- and so CI itself -- slow.
|
|
#
|
|
# Volumes pruned here are podman's ANONYMOUS volumes, not the bind mounts
|
|
# under {{ podman_volumes }} that hold real service data -- those are
|
|
# directories on the host and podman does not know about them. The dangling
|
|
# ones seen in practice were 804 MB copies of Nextcloud's /var/www/html left
|
|
# by container recreations, which are image content, not data.
|
|
set -uo pipefail
|
|
|
|
TAG=podman-prune
|
|
|
|
log() { logger -t "$TAG" -p daemon.info -- "$*"; echo "$TAG: $*"; }
|
|
|
|
# Rootless podman: -H so HOME points at the user's store, and the `cd;`
|
|
# preamble is required (see CLAUDE.md) or podman cannot find its graph root.
|
|
run() {
|
|
local u=$1
|
|
shift
|
|
sudo -H -u "$u" bash -c \
|
|
'cd; d=/run/user/$(id -u); [ -d "$d" ] && export XDG_RUNTIME_DIR="$d"
|
|
exec podman "$@"' _ "$@"
|
|
}
|
|
|
|
# prune_user <user> <until> <prune_containers: yes|no> [image prune filter...]
|
|
prune_user() {
|
|
local u=$1 keep=$2 do_containers=$3
|
|
shift 3
|
|
local before after img vol con
|
|
|
|
if ! id "$u" >/dev/null 2>&1; then
|
|
log "user=$u status=skipped reason=no-such-user"
|
|
return
|
|
fi
|
|
|
|
before=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
|
|
|
# Deliberately NOT `set -e`: a prune failing for one user must not stop the
|
|
# others, and a busy image is a normal, non-fatal outcome.
|
|
#
|
|
# Containers are reaped BEFORE images on purpose -- an exited container pins
|
|
# the image it ran from, so pruning images first would leave those layers
|
|
# behind for another day.
|
|
con=none
|
|
if [ "$do_containers" = yes ]; then
|
|
con=$(run "$u" container prune -f --filter "until=$keep" 2>&1 | tail -1)
|
|
fi
|
|
|
|
img=$(run "$u" image prune -af --filter "until=$keep" "$@" 2>&1 | tail -1)
|
|
vol=$(run "$u" volume prune -f 2>&1 | tail -1)
|
|
|
|
after=$(run "$u" system df --format '{{ '{{' }}.Size{{ '}}' }}' 2>/dev/null | head -1)
|
|
|
|
log "user=$u keep=$keep size_before=$before size_after=$after"
|
|
log "user=$u image_prune=${img:-none} volume_prune=${vol:-none} container_prune=${con:-none}"
|
|
}
|
|
|
|
for u in {{ podman_prune_users | join(' ') }}; do
|
|
prune_user "$u" "{{ podman_prune_until }}" no
|
|
done
|
|
|
|
for u in {{ podman_prune_ci_users | join(' ') }}; do
|
|
# CI base images (gitea-ci, -espidf, -platformio) carry the keep label: they
|
|
# are rebuilt or re-pulled from the registry only when missing, so pruning
|
|
# them just forces a multi-GB re-download on the next job.
|
|
prune_user "$u" "{{ podman_prune_ci_until }}" yes --filter "label!={{ podman_prune_ci_keep_label }}"
|
|
done
|
|
|
|
log "status=ok"
|