74 lines
4.1 KiB
Bash
74 lines
4.1 KiB
Bash
#!/bin/sh
|
|
# aurora-netwatch (B20): LTE data-path watchdog for standalone use. Runs under supervise-daemon (/etc/init.d/aurora-netwatch).
|
|
# Why: on 2026-10-06 Beeline detached the UE (bearer "cm error: emm-detached"); the modem re-attached, but nothing re-connected the bearer,
|
|
# wwan0 kept its dead address/routes and the board sat without internet (and without VPN) until a manual reconnect.
|
|
# Every $NW_PERIOD s while the controller reports BEARER_CONNECTED:
|
|
# healthy = MM bearer connected AND at least one of the probe hosts (/32 via wwan0) answers ping.
|
|
# $NW_FAILS consecutive unhealthy checks -> `aurora-modem disconnect` + `aurora-modem start` (radio/MSS/MM stay up, ~6 s), then
|
|
# restore the default route (aurora-router adds it once at boot; disconnect drops it with the link).
|
|
# Controller state OFF (a start failed and rolled back) -> `aurora-modem start` again, at most every $NW_OFF_RETRY s.
|
|
# Never runs while another aurora-modem start/stop/disconnect is in progress (manual or boot). Config: /etc/aurora/netwatch.conf (optional).
|
|
NW_PERIOD=30 NW_FAILS=3 NW_OFF_RETRY=300 NW_BOOT_GRACE=600 NW_PROBES="77.88.8.8"
|
|
[ -f /etc/aurora/netwatch.conf ] && . /etc/aurora/netwatch.conf
|
|
M=/usr/sbin/aurora-modem; D=/run/aurora-modem; S=/run/aurora-netwatch; mkdir -p $S; LOG=$S/netwatch.log
|
|
now() { cut -d' ' -f1 /proc/uptime; }
|
|
log() { m="[$(now)] $*"; echo "$m" >> $LOG; echo "<5>[aurora-netwatch] $*" > /dev/kmsg 2>/dev/null; }
|
|
busy() { for a in start stop disconnect; do pgrep -f "$M $a" >/dev/null && return 0; done; return 1; }
|
|
mstate() { cat $D/state 2>/dev/null || echo NONE; }
|
|
up_s() { cut -d. -f1 /proc/uptime; }
|
|
|
|
bearer_ok() {
|
|
BR=$(sed -n 's/^BR=//p' $D/ipcfg 2>/dev/null)
|
|
[ -n "$BR" ] || return 1
|
|
timeout 10 mmcli -b "$BR" --output-keyvalue 2>/dev/null | grep -q "bearer.status.connected *: yes"
|
|
}
|
|
ping_ok() {
|
|
. $D/ipcfg 2>/dev/null
|
|
for h in $NW_PROBES $(echo "$DNS" | tr ',' ' '); do ping -c1 -W3 "$h" >/dev/null 2>&1 && return 0; done
|
|
return 1
|
|
}
|
|
defroute() {
|
|
IF=$(sed -n 's/^IF=//p' $D/ipcfg 2>/dev/null)
|
|
[ -n "$IF" ] && ! ip route | grep -q "^default" && ip route add default dev "$IF" 2>/dev/null && log "default route restored via $IF"
|
|
}
|
|
diag() {
|
|
timeout 10 mmcli -m any 2>/dev/null | sed 's/\x1b\[[0-9;]*m//g' | grep -E " state:|signal quality|access tech|registration|packet service" | tr -s ' ' | tr '\n' ';'
|
|
BR=$(sed -n 's/^BR=//p' $D/ipcfg 2>/dev/null)
|
|
[ -n "$BR" ] && timeout 10 mmcli -b "$BR" 2>/dev/null | sed 's/\x1b\[[0-9;]*m//g' | grep -E "connected:|error message|duration|bytes rx" | tr -s ' ' | tr '\n' ';'
|
|
}
|
|
recover() {
|
|
log "RECOVER ($1): $(diag)"
|
|
t0=$(up_s)
|
|
# timeouts above the controller's own deadlines (REG_TIMEOUT 600 + MM/connect) so it always finishes its own rollback
|
|
if [ "$(mstate)" = OFF ]; then timeout 1500 $M start >> $LOG 2>&1; rc=$?
|
|
else timeout 300 $M disconnect >> $LOG 2>&1; timeout 1500 $M start >> $LOG 2>&1; rc=$?; fi
|
|
defroute
|
|
if [ $rc = 0 ]; then log "RECOVER OK in $(( $(up_s) - t0 ))s: $(tail -1 $D/lifecycle.log | cut -c1-110)"
|
|
else log "RECOVER FAIL rc=$rc in $(( $(up_s) - t0 ))s, state $(mstate)"; fi
|
|
echo $(up_s) > $S/last-recover
|
|
}
|
|
|
|
log "=== NETWATCH START period=${NW_PERIOD}s fails=$NW_FAILS probes=[$NW_PROBES + bearer DNS] off_retry=${NW_OFF_RETRY}s"
|
|
fails=0; last_off=0
|
|
while :; do
|
|
sleep $NW_PERIOD
|
|
if busy; then fails=0; continue; fi
|
|
st=$(mstate)
|
|
case "$st" in
|
|
BEARER_CONNECTED)
|
|
defroute
|
|
if bearer_ok && ping_ok; then
|
|
[ $fails -gt 0 ] && log "healthy again after $fails failed check(s)"; fails=0; echo "ok $(up_s)" > $S/status
|
|
else
|
|
fails=$((fails+1)); echo "fail $fails $(up_s)" > $S/status
|
|
log "check failed ($fails/$NW_FAILS): bearer=$(bearer_ok && echo up || echo DOWN) ping=$(ping_ok && echo ok || echo FAIL)"
|
|
[ $fails -ge $NW_FAILS ] && { recover "bearer/data dead"; fails=0; }
|
|
fi;;
|
|
OFF)
|
|
# boot start still has its own deadline; only act after the grace period, then rate-limited
|
|
[ $(up_s) -lt $NW_BOOT_GRACE ] && continue
|
|
[ $(( $(up_s) - last_off )) -lt $NW_OFF_RETRY ] && continue
|
|
last_off=$(up_s); echo "off $(up_s)" > $S/status; recover "controller OFF";;
|
|
*) fails=0;; # starting / stopping: the controller owns the modem
|
|
esac
|
|
done
|