Files
deploy_home/ansible/roles/labelprint/templates/wifi-rescue.sh.j2
T
Bastian de BylandClaude Opus 5 6cd4d56de1 feat(labelprint): 4x6 label print proxy on a Raspberry Pi
A Pi 3B+ (stickah.local) shares a Phomemo PM246 to the LAN as a plain CUPS
queue, so any machine can print 4x6 labels -- fulfillr-site's shipping labels
in particular -- without installing the vendor driver, which is x86-64 only.
The role builds the TSPL CUPS driver from source instead.

It is Debian, not Fedora, so it lives in its own inventory and playbook
(make deploy-labelprint / check-labelprint) and the home.debyl.io roles can
never run against it. make bootfs renders its cloud-init first-boot files onto
a freshly imaged SD card from the same templates the role uses. The Wi-Fi
credentials for the home and rescue networks are in the vault.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-13 23:14:45 -04:00

133 lines
4.4 KiB
Django/Jinja

#!/bin/sh
# {{ ansible_managed }}
#
# Keeps the label print proxy reachable.
#
# Normally the Pi is a station on the home SSID. If that network is gone -- the
# password changed, the AP died, the Pi was carried somewhere else -- it brings
# up its own rescue access point so there is still a way in to reconfigure it.
# It keeps checking, and hands the radio back the moment the home SSID returns.
#
# The 3B+ has one radio and brcmfmac will not hold an AP and a station link at
# the same time, so this cannot listen for the home SSID while the AP is up.
# Instead the AP drops for a few seconds every {{ labelprint_ap_rescan_secs }}s
# to scan, and comes straight back if the home network is still missing.
#
# If you are working over the rescue AP and do not want the radio pulled out
# from under your SSH session:
#
# touch /run/wifi-rescue.hold
#
# The hold expires by itself after {{ (labelprint_hold_max_age_secs / 60) | int }} minutes, so a forgotten hold file
# cannot strand the Pi permanently.
set -eu
HOME_CON=home-wifi
AP_CON=rescue-ap
SSID='{{ stickah_ssid }}'
AP_SSID='{{ stickah_ssid_rescue }}'
IFACE=wlan0
HOLD=/run/wifi-rescue.hold
STAMP=/run/wifi-rescue.ap-since
RESCAN_SECS={{ labelprint_ap_rescan_secs }}
HOLD_MAX_AGE={{ labelprint_hold_max_age_secs }}
log() { logger -t wifi-rescue -- "$@"; }
con_active() { nmcli -t -f NAME connection show --active | grep -qxF "$1"; }
# A default IPv4 route is the honest test for "on a real network". The rescue
# AP uses NetworkManager's shared mode, which hands out addresses but installs
# no default route, so the AP can never make this look true.
online() { [ -n "$(ip -4 route show default 2>/dev/null)" ]; }
held() {
[ -e "$HOLD" ] || return 1
age=$(( $(date +%s) - $(stat -c %Y "$HOLD" 2>/dev/null || echo 0) ))
if [ "$age" -lt "$HOLD_MAX_AGE" ]; then
return 0
fi
log "hold file is ${age}s old; expiring it"
rm -f "$HOLD"
return 1
}
start_ap() {
con_active "$AP_CON" && return 0
log "starting rescue AP '$AP_SSID' on {{ labelprint_ap_addr }}"
if nmcli --wait 20 connection up "$AP_CON" >/dev/null 2>&1; then
date +%s > "$STAMP"
else
log "ERROR: rescue AP failed to start"
fi
}
join_home() {
# Bounded: the default 90s wait outlives the watchdog interval, and an
# SSID that is not there is not going to appear in the next minute.
#
# --wait is a GLOBAL nmcli option and has to precede the subcommand. Written
# as `connection up <id> --wait 20` it is rejected outright with "invalid
# extra argument" -- and only for a connection that exists, so it looks fine
# against a typo'd name. That failure mode is silent and total: every join
# returns failure and the rescue AP never starts.
nmcli --wait 20 connection up "$HOME_CON" >/dev/null 2>&1 || return 1
online
}
held && exit 0
if online; then
if con_active "$AP_CON"; then
log "back on the network; shutting the rescue AP down"
nmcli connection down "$AP_CON" >/dev/null 2>&1 || true
rm -f "$STAMP"
fi
exit 0
fi
if con_active "$AP_CON"; then
since=$(cat "$STAMP" 2>/dev/null || echo 0)
[ $(( $(date +%s) - since )) -lt "$RESCAN_SECS" ] && exit 0
log "rescue AP up for ${RESCAN_SECS}s; dropping it to scan for '$SSID'"
nmcli connection down "$AP_CON" >/dev/null 2>&1 || true
sleep 2
nmcli device wifi rescan ifname "$IFACE" >/dev/null 2>&1 || true
sleep 5
scan=$(nmcli -t -f SSID device wifi list ifname "$IFACE" 2>/dev/null || true)
if printf '%s\n' "$scan" | grep -qxF "$SSID"; then
log "'$SSID' is back; rejoining"
try=yes
elif [ -z "$(printf '%s' "$scan" | tr -d '[:space:]')" ]; then
# Nothing at all came back. Either the radio has not settled after
# dropping the AP, or the home SSID is hidden and will never show up in
# a scan. Worth 20 seconds to find out.
log "scan came back empty; trying '$SSID' anyway"
try=yes
else
# Other networks are visible and ours is not, so it really is gone.
# Straight back to the AP -- no point spending the join timeout.
try=no
fi
if [ "$try" = yes ]; then
if join_home; then
log "rejoined '$SSID'"
rm -f "$STAMP"
exit 0
fi
log "join failed; returning to the rescue AP"
fi
start_ap
exit 0
fi
log "offline and no rescue AP; trying '$SSID'"
if join_home; then
log "joined '$SSID'"
exit 0
fi
start_ap