podman_prune_users listed only the podman and git users, so the two stores that turn over fastest were never touched. gitea-runner had reached 1205 images / 113.1 GB with 100% of it reclaimable, and actions-runner had 137 exited job containers. That layer count is what makes overlayfs lookups -- and so CI itself -- slow; the disk was the lesser problem. Split into two policies, because the stores are not the same kind of thing: service users keep the 30-day rollback window, and their containers are deliberately NOT pruned. They are the live services, and reaping one that merely happens to be stopped would turn a transient crash into a unit that cannot start again until the next deploy. CI users get 48h and their exited job containers reaped too. Build layers carry no rollback value. Containers are reaped BEFORE images on purpose: an exited container pins the image it ran from, so pruning images first would leave those layers behind for another day. Timer moved weekly -> daily; a week of CI turnover is what let the store reach 113 GB between runs. Persistent=true is kept so a missed run catches up. First run reclaimed 134 GB: gitea-runner 113.1 -> 4.2 GB, actions-runner 7.6 GB -> 0, podman 25.7 -> 15.1 GB. Disk 449G -> 315G, all 26 containers up. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
296 lines
12 KiB
YAML
296 lines
12 KiB
YAML
---
|
|
# Where Nextcloud backup failure alerts are mailed (see containers/cloud-backup.yml).
|
|
backup_alert_email: bastian@debyl.io
|
|
bookstack_path: "{{ podman_volumes }}/bookstack"
|
|
cam2ip_path: "{{ podman_volumes }}/cam2ip"
|
|
cloud_path: "{{ podman_volumes }}/cloud"
|
|
cloud_skudak_path: "{{ podman_volumes }}/skudakcloud"
|
|
debyltech_path: "{{ podman_volumes }}/debyltech"
|
|
# drone_path: removed - Drone CI decommissioned
|
|
factorio_path: "{{ podman_volumes }}/factorio"
|
|
fulfillr_path: "{{ podman_volumes }}/fulfillr"
|
|
fulfillr_cases_table: "debyltech-cases-prod"
|
|
fulfillr_tickets_table: "debyltech-tickets-prod"
|
|
# Turso ecommerce store (self-hosted checkout).
|
|
# PROD store URL (non-secret); the RW token `fulfillr_prod_store_auth_token` is in the vault.
|
|
fulfillr_prod_store_database_url: "libsql://debyltech-store-prod-debyltech.aws-us-east-1.turso.io"
|
|
# Staging back-office (fulfillr-dev.debyltech.com, port 9055) -> staging Turso store.
|
|
# Its RW token is `fulfillr_dev_store_auth_token` and EasyPost test key is
|
|
# `fulfillr_dev_easypost_api_key`, both in the encrypted vault.
|
|
fulfillr_dev_path: "{{ podman_volumes }}/fulfillr-dev"
|
|
fulfillr_dev_server_name: fulfillr-dev.debyltech.com
|
|
fulfillr_dev_store_database_url: "libsql://debyltech-store-staging-debyltech.aws-us-east-1.turso.io"
|
|
gregtime_path: "{{ podman_volumes }}/gregtime"
|
|
hass_path: "{{ podman_volumes }}/hass"
|
|
# nginx_path: removed - nginx no longer used
|
|
# nosql_path: removed - nosql/redis no longer used
|
|
partsy_path: "{{ podman_volumes }}/partsy"
|
|
partsy_skudak_path: "{{ podman_volumes }}/partsy-skudak"
|
|
photos_path: "{{ podman_volumes }}/photos"
|
|
uptime_kuma_path: "{{ podman_volumes }}/uptime-kuma"
|
|
uptime_kuma_personal_path: "{{ podman_volumes }}/uptime-kuma-personal"
|
|
zomboid_path: "{{ podman_volumes }}/zomboid"
|
|
|
|
# Whether to deploy and run the Zomboid server at all.
|
|
#
|
|
# Was off for months because it was the heaviest tenant on this host: ~9 GB
|
|
# resident and ~10 MB/s of continuous log writes, which on a 4-core box also
|
|
# running Gitea, Immich, Graylog and two Nextclouds saturated the QLC SSD and
|
|
# starved CI -- a firmware build stretched to 17 minutes and one release job
|
|
# was killed outright by Gitea's zombie-task timeout.
|
|
#
|
|
# Back on now that log growth is bounded: the container logs to a rotating
|
|
# k8s-file instead of the shared journal, and logrotate caps the server's own
|
|
# server-console.txt and Logs/ (see containers/home/zomboid.yml). Memory is
|
|
# unchanged -- idle draw is nowhere near MAX_RAM.
|
|
#
|
|
# Pass -e zomboid_enabled=false to keep it down through a CI-heavy stretch.
|
|
# World data under {{ zomboid_path }} is left in place either way.
|
|
zomboid_enabled: true
|
|
|
|
# Server name, and also the on-disk identity: PZ derives Server/<name>.ini,
|
|
# Saves/Multiplayer/<name>/ and db/<name>.db from it. Renaming it therefore
|
|
# starts a brand-new world and leaves the previous one intact for rollback.
|
|
zomboid_server_name: debbzoid
|
|
zomboid_public_name: Debbzoid
|
|
|
|
# Sophie 42 ships map_distanciado as its map mod. Muldraugh must stay last --
|
|
# the base map is the fallback layer.
|
|
zomboid_map: "map_distanciado;Muldraugh, KY"
|
|
|
|
# Zomboid RCON port for remote administration
|
|
zomboid_rcon_port: "27015"
|
|
|
|
# Discord channel the server's own chat bridge posts into. Build 42 replaced
|
|
# B41's DiscordChannel/DiscordChannelID pair with DiscordChatChannel, which
|
|
# takes a channel *name*, not a snowflake ID.
|
|
zomboid_discord_chat_channel: zomboidbot
|
|
|
|
# JVM heap, written into ProjectZomboid64.json by the container entrypoint.
|
|
zomboid_min_ram: 8g
|
|
zomboid_max_ram: 24g
|
|
|
|
# The modlist itself lives in vars/zomboid_sophie_mods.yml -- 283 mod IDs and
|
|
# 256 workshop items vendored from the upstream Sophie 42 preset.
|
|
zomboid_preset_version: "sophie-42 @ 2026-07-31"
|
|
|
|
# Mods subtracted from the vendored Sophie lists before they reach the INI.
|
|
# Explicit and reasoned so that re-vendoring a newer preset cannot silently
|
|
# reintroduce something we deliberately dropped.
|
|
zomboid_mods_excluded:
|
|
# Already absent from the 2026-07-31 preset; listed so it stays that way.
|
|
- workshop_id: "3718412967"
|
|
mod_id: IconsInventory
|
|
reason: Sophie recommends removing Icon Inventory
|
|
# Build 41 mod (versionMin=41.60) that indexes the ModOptions framework, which
|
|
# Sophie's list does not include. Dies at VehicleDoorsHotkey_Options.lua:4 --
|
|
# "attempted index: ModOptions of non-table: null" -- and never initialises, so
|
|
# it is 21 exceptions a boot in exchange for a hotkey that cannot work.
|
|
- workshop_id: "2290459371"
|
|
mod_id: VehicleDoorsHotkey
|
|
reason: needs the absent ModOptions framework; throws on load and never runs
|
|
|
|
# Per-source-IP ceiling on 53-byte Steam/PZ query packets before they are
|
|
# logged and dropped. Set well above anything a real client produces -- opening
|
|
# the server browser or retrying a connection is a handful of queries, while a
|
|
# scanner flood is thousands. The previous 1/hour burst 2 was below normal play
|
|
# and made the server unreachable. See containers/home/zomboid.yml.
|
|
zomboid_query_rate_limit: 60/minute
|
|
zomboid_query_burst: 120
|
|
|
|
# How long to keep PZ's own log archive under data/Logs.
|
|
#
|
|
# PZ rolls logs into dated directories and never prunes them; logrotate cannot
|
|
# bound that, because the growth is in the number of files rather than the size
|
|
# of any one of them. See containers/home/zomboid.yml.
|
|
zomboid_log_retention_days: 14
|
|
|
|
# Mod IDs the Sophie preset's `Mods=` line gets wrong for a 42.20.3 server.
|
|
#
|
|
# Every workshop item below is in the preset's own WorkshopItems= list and
|
|
# downloads fine -- only the ID it is referenced by is stale, so the server
|
|
# logs `required mod ... not found` and quietly runs without it. Verified
|
|
# against the id= field of each mod.info on disk, not guessed from folder
|
|
# names (folder name and mod ID are unrelated). Applied by server.ini.j2 so a
|
|
# re-vendor of a newer preset keeps the corrections.
|
|
zomboid_mods_renamed:
|
|
- old: "FWOBenchPress&Treadmill"
|
|
new: FWOBenchPressTreadmill
|
|
reason: preset carries a stray ampersand; mod.info in 2940354599 has none
|
|
- old: Ladders42131
|
|
new: Ladders4220
|
|
reason: 3629835761 ships per-build variants; 42131 is the 42.16 one, 4220 is ours
|
|
- old: ServingPlatesB42
|
|
new: ServingPlates42
|
|
reason: mod.info in 3399320470 declares ServingPlates42
|
|
- old: NewMusic_OrchestraMix
|
|
new: NewMusic_CMM
|
|
reason: author renamed it to "Classical Music Mix" in 3760471726
|
|
|
|
# Server config is seeded once and then left alone, so it can be hand-edited
|
|
# between world regenerations. Set true to overwrite it from the templates.
|
|
zomboid_config_force: false
|
|
|
|
sshpass_cron_path: "{{ podman_volumes }}/sshpass_cron"
|
|
caddy_path: "{{ podman_volumes }}/caddy"
|
|
|
|
# Drone CI variables removed - infrastructure decommissioned
|
|
# drone_server_proto, drone_runner_proto, drone_runner_capacity
|
|
|
|
# Server names (used by Caddy)
|
|
base_server_name: bdebyl.net
|
|
assistant_server_name: assistant.bdebyl.net
|
|
bookstack_server_name: wiki.skudakrennsport.com
|
|
# ci_server_name: removed - Drone CI decommissioned
|
|
cloud_server_name: cloud.bdebyl.net
|
|
cloud_skudak_server_name: cloud.skudakrennsport.com
|
|
fulfillr_server_name: fulfillr.debyltech.com
|
|
home_server_name: home.debyl.io
|
|
uptime_kuma_server_name: uptime.debyltech.com
|
|
uptime_kuma_personal_server_name: uptime.debyl.io
|
|
parts_server_name: parts.bdebyl.net
|
|
photos_server_name: photos.bdebyl.net
|
|
|
|
# debyl.io domains (migration from bdebyl.net)
|
|
base_server_name_io: debyl.io
|
|
assistant_server_name_io: assistant.debyl.io
|
|
cloud_server_name_io: cloud.debyl.io
|
|
home_server_name_io: home.debyl.io
|
|
parts_server_name_io: parts.debyl.io
|
|
photos_server_name_io: photos.debyl.io
|
|
gitea_debyl_server_name: git.debyl.io
|
|
|
|
# skudak.com domains (migration from skudakrennsport.com)
|
|
parts_skudak_server_name: parts.skudak.com
|
|
bookstack_server_name_new: wiki.skudak.com
|
|
# partsy_skudak_admin_password: defined in vault
|
|
cloud_skudak_server_name_new: cloud.skudak.com
|
|
gitea_skudak_server_name: git.skudak.com
|
|
|
|
# LibreSign root certificate authority identity for skudak-cloud. This is the
|
|
# issuer name that appears on every signed document, so it must match the
|
|
# entity's legal name -- it was previously generated as "Skudak Rennsport LLP",
|
|
# the pre-rename name. Changing these values does NOT re-issue the CA on its
|
|
# own; see the guarded generate task in containers/skudak/cloud.yml.
|
|
# Skudak brand palette, mirroring ~/src/skudak/skudak-site/src/styles/variables.css.
|
|
# --color-gray-900 for the mail header band and UI chrome; the white signature
|
|
# logo is legible on it. --color-accent is spent only on the CTA button, and
|
|
# lives in the skudakmail app rather than here since theming has no second
|
|
# colour slot.
|
|
theming_skudak_primary: "#0A0A0A"
|
|
|
|
libresign_skudak_cert_cn: Skudak LLP
|
|
libresign_skudak_cert_o: Skudak LLP
|
|
libresign_skudak_cert_c: US
|
|
libresign_skudak_cert_st: New Hampshire
|
|
libresign_skudak_cert_l: Newbury
|
|
|
|
|
|
# Legacy nginx/ModSecurity configuration removed - Caddy provides built-in security
|
|
|
|
# Web server configuration (Caddy is the default)
|
|
# Legacy nginx variables kept for cleanup tasks
|
|
|
|
# Caddy configuration
|
|
caddy_email: "{{ ssl_email }}"
|
|
# Use staging for testing, production for real certificates
|
|
caddy_acme_ca: https://acme-v02.api.letsencrypt.org/directory
|
|
# For testing/staging:
|
|
# caddy_acme_ca: https://acme-staging-v02.api.letsencrypt.org/directory
|
|
|
|
# Caddy ports
|
|
caddy_admin_port: 2019
|
|
|
|
# Caddy network configuration
|
|
caddy_local_networks:
|
|
- 192.168.0.0/16
|
|
- 127.0.0.1
|
|
|
|
# Caddy logging configuration
|
|
caddy_log_level: INFO
|
|
caddy_log_format: json
|
|
|
|
# Log rotation. Caddy rotates on implicit defaults (100MiB / keep 10 / 90d)
|
|
# without these, and those were demonstrably not holding: 20 rotated files per
|
|
# stream for the busiest logs and a 190-day-old .gz, for 1.3 GB across 16
|
|
# streams. Kept short deliberately -- fluent-bit tails every one of these into
|
|
# Graylog (see roles/common/templates/fluent-bit/fluent-bit.conf.j2), so
|
|
# Graylog is the system of record and the local files are only a buffer.
|
|
caddy_log_roll_size: 10MiB
|
|
caddy_log_roll_keep: 3
|
|
caddy_log_roll_keep_for: 168h
|
|
|
|
# Caddy performance tuning
|
|
caddy_max_request_body_mb: 500
|
|
|
|
# Caddy security headers (global defaults)
|
|
caddy_security_headers:
|
|
Strict-Transport-Security: "max-age=31536000; includeSubDomains"
|
|
X-Content-Type-Options: "nosniff"
|
|
Referrer-Policy: "same-origin"
|
|
X-Frame-Options: "SAMEORIGIN"
|
|
|
|
# Graylog logging stack
|
|
graylog_path: "{{ podman_volumes }}/graylog"
|
|
logs_server_name: logs.debyl.io
|
|
# gelf_auth_token: defined in vault - X-Gelf-Token header for Lambda GELF HTTP auth
|
|
|
|
# Fluent Bit is deployed as a systemd service (not container)
|
|
# for direct journal access - see containers/base/fluent-bit.yml
|
|
|
|
# Fluent-bit Caddy log forwarding
|
|
caddy_log_path: "{{ caddy_path }}/logs"
|
|
caddy_log_names:
|
|
- caddy
|
|
- photos
|
|
- wiki
|
|
- assistant
|
|
- parts
|
|
- uptime-kuma
|
|
- uptime-kuma-personal
|
|
- graylog
|
|
- cloud
|
|
- cloud-skudak
|
|
- gitea-debyl
|
|
- gitea-skudak
|
|
- fulfillr
|
|
|
|
# GeoIP configuration for Graylog
|
|
# Requires free MaxMind account: https://dev.maxmind.com/geoip/geolite2-free-geolocation-data
|
|
geoip_path: "{{ graylog_path }}/geoip"
|
|
geoip_database_edition: GeoLite2-City
|
|
# geoip_maxmind_account_id: defined in vault
|
|
# geoip_maxmind_license_key: defined in vault
|
|
|
|
# Weekly podman prune (see templates/podman-prune.sh.j2). Both rootless users
|
|
# have their own image store; the git user runs the Gitea pods.
|
|
podman_prune_users:
|
|
- "{{ podman_user }}"
|
|
- "{{ git_user }}"
|
|
# Keep 30 days of unused images so a rollback needs no rebuild or re-pull.
|
|
podman_prune_until: 720h
|
|
|
|
# The CI runners were never pruned and had run away: gitea-runner reached 1205
|
|
# images / 113 GB with 100% reclaimable, actions-runner 137 exited job
|
|
# containers. Build layers carry no rollback value, so they keep a far shorter
|
|
# window than the service stores and their exited containers are reaped too.
|
|
podman_prune_ci_users:
|
|
- gitea-runner
|
|
- actions-runner
|
|
podman_prune_ci_until: 48h
|
|
|
|
# Daily rather than weekly: CI turns over many images a day, and a week of that
|
|
# is what let the store reach 113 GB between runs.
|
|
podman_prune_oncalendar: "*-*-* 02:00:00"
|
|
|
|
# Hardened CIFS options for the TrueNAS shares (see containers/home/photos.yml).
|
|
# x-systemd.automount is what makes a failed mount recoverable without a human.
|
|
cifs_mount_opts: >-
|
|
credentials=/etc/cifs/photos.creds,uid={{ podman_subuid.stdout }},gid={{ podman_subuid.stdout }},_netdev,nofail,soft,x-systemd.automount,x-systemd.mount-timeout=30,x-systemd.idle-timeout=600,vers=3.1.1,resilienthandles,echo_interval=10
|
|
# How often the watchdog checks mount health and heals immich.
|
|
cifs_watchdog_oncalendar: "*:0/5"
|
|
cifs_watchdog_mounts:
|
|
- "{{ photos_path }}/storage"
|
|
- "{{ photos_path }}/immich"
|
|
|