# services/wan-watchdog.nix — self-heal the router's WAN when the ISP flaps. # # Jul 2026: the ISP migrated us to a different gateway/subnet; eno1's DHCP # lease flapped and the LAN lost internet until a manual reboot. Root cause of # the "needs a reboot" part: after the WAN IP changed, the kernel NAT # conntrack table still mapped LAN flows to the OLD WAN IP, so masquerade kept # using stale info until entries aged out. A reboot flushed it. # # This watchdog (30s timer) fixes that without a reboot: # 1. WAN IP changed since last run -> flush conntrack (clears stale NAT). # 2. Internet unreachable via eno1: # ~90s down -> `networkctl renew eno1` + flush conntrack # ~4min down -> restart systemd-networkd + flush conntrack # Escalates once at each step, then stops (a real ISP outage can't be # fixed by thrashing the link) until connectivity returns. # 3. ntfy alert (reuses /var/secrets/ntfy-url) on each action + on recovery. { config, lib, pkgs, ... }: let # Ping out the WAN specifically (source-bound) so a healthy LAN doesn't read # as "up". Two anycast targets so one being down isn't a false alarm. watchdog = pkgs.writeShellScript "wan-watchdog" '' set -uo pipefail wan=eno1 state=/var/lib/wan-watchdog ipfile=$state/wan-ip failfile=$state/failcount secret=/var/secrets/ntfy-url ping=${pkgs.iputils}/bin/ping ip=${pkgs.iproute2}/bin/ip conntrack=${pkgs.conntrack-tools}/bin/conntrack networkctl=${pkgs.systemd}/bin/networkctl systemctl=${pkgs.systemd}/bin/systemctl notify() { # body [ -f "$secret" ] || return 0 url=$(${pkgs.coreutils}/bin/tr -d '\n' < "$secret") ${pkgs.curl}/bin/curl -fsS --max-time 10 \ -H "Title: WAN watchdog" -H "Priority: high" -H "Tags: satellite_antenna" \ -d "$1" "$url" >/dev/null 2>&1 || true } # --- 1. Flush stale NAT when the WAN IP changes --- cur_ip=$($ip -4 -o addr show "$wan" 2>/dev/null | ${pkgs.gawk}/bin/awk '{print $4}' | ${pkgs.coreutils}/bin/head -1) old_ip=$(${pkgs.coreutils}/bin/cat "$ipfile" 2>/dev/null || true) if [ -n "$cur_ip" ] && [ "$cur_ip" != "$old_ip" ]; then $conntrack -F >/dev/null 2>&1 || true ${pkgs.coreutils}/bin/printf '%s' "$cur_ip" > "$ipfile" # Only page on a real change, not the first-ever run. [ -n "$old_ip" ] && notify "WAN IP changed $old_ip -> $cur_ip; flushed NAT conntrack." fi # --- 2. Connectivity check via the WAN --- if $ping -c1 -W3 -I "$wan" 1.1.1.1 >/dev/null 2>&1 \ || $ping -c1 -W3 -I "$wan" 8.8.8.8 >/dev/null 2>&1; then prev=$(${pkgs.coreutils}/bin/cat "$failfile" 2>/dev/null || echo 0) [ "$prev" -ge 3 ] && notify "WAN back up (was down ~$(( prev * 30 ))s)." echo 0 > "$failfile" exit 0 fi # --- Down: escalate on consecutive failures (timer fires every 30s) --- fails=$(( $(${pkgs.coreutils}/bin/cat "$failfile" 2>/dev/null || echo 0) + 1 )) echo "$fails" > "$failfile" case "$fails" in 3) # ~90s down: renew the lease $networkctl renew "$wan" >/dev/null 2>&1 || true $conntrack -F >/dev/null 2>&1 || true notify "WAN unreachable ~90s: renewed $wan DHCP lease + flushed conntrack." ;; 8) # ~4min down: full network stack restart $systemctl restart systemd-networkd >/dev/null 2>&1 || true $conntrack -F >/dev/null 2>&1 || true notify "WAN still down ~4min: restarted systemd-networkd." ;; esac ''; in { config = lib.mkIf (config.networking.hostName == "FredOS-Mediaserver") { systemd.services.wan-watchdog = { description = "Self-heal WAN on ISP lease flap / IP change"; serviceConfig = { Type = "oneshot"; ExecStart = watchdog; StateDirectory = "wan-watchdog"; }; }; systemd.timers.wan-watchdog = { wantedBy = [ "timers.target" ]; timerConfig = { OnBootSec = "1min"; # let the network settle after boot first OnUnitActiveSec = "30s"; }; }; }; }