#!/bin/sh # Make the WAN match VRRP mastership. Idempotent; safe to run every 30s and on # every VRRP transition. # # Why a reconciler and not just transition scripts # ------------------------------------------------ # VyOS delivers `transition-script` through a helper process, # /usr/libexec/vyos/system/keepalived-fifo.py, fed by keepalived's notify_fifo. # Observed in labsim on 2026-09-02: the primary's Keepalived_vrrp logged # "(native) Entering MASTER STATE" for all six instances and the built-in # notify_master for conntrack-sync ran -- while the fifo helper logged NOTHING # and the master transition script never ran. The helper process was still # alive. The result was a router holding every VIP with no WAN at all: the exact # 2026-09-02 outage, re-created by the mechanism meant to prevent it. # # So transition scripts are kept for speed but nothing is trusted to them: this # also runs on a timer, and derives everything from ground truth rather than # from a marker that only exists if the script it depends on ran. # # vrrp-wan-reconcile reconcile once # vrrp-wan-reconcile --status what it thinks, changing nothing # # Ground truth for "am I master" is whether the management VIP is really on this # box. It is what VRRP actually does, it is observable, and it cannot silently # disagree with reality. VIP="${VRRP_WAN_VIP:-192.168.1.1}" # management VIP; sim overrides via env WAN_VIF=53 # bond0.53, the DHCP WAN STATE=/run/vrrp-wan LOCK=/run/vrrp-wan.lock cfg() { /opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands 2>/dev/null; } holds_vip() { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; } wan_up() { ip -4 addr show "bond0.${WAN_VIF}" 2>/dev/null | grep -q 'inet '; } wan_disabled(){ cfg | grep -q "vif ${WAN_VIF} disable"; } ppp_disabled(){ cfg | grep -q "pppoe pppoe0 disable"; } # --status must answer WITHOUT sourcing script-template. The template's `exit` # is a function that leaves configuration mode, not the shell builtin, so a # status run that had sourced it opened and closed a config session on every # call -- which is how a read-only query started colliding with the timer and # logging "Configuration system temporarily locked due to another commit". if [ "${1:-}" = "--status" ]; then printf 'vip=%s holds_vip=%s wan_disabled=%s wan_up=%s role=%s\n' \ "$VIP" "$(holds_vip && echo yes || echo no)" \ "$(wan_disabled && echo yes || echo no)" \ "$(wan_up && echo yes || echo no)" \ "$(cat "$STATE/role" 2>/dev/null || echo unset)" exit 0 fi # One writer. The 30s timer and a VRRP transition can fire together, and two # VyOS commits in flight on one box do not queue -- the second fails outright. exec 9>"$LOCK" flock -n 9 || exit 0 mkdir -p "$STATE" # Reap config sessions whose owning process is gone. VyOS creates # /opt/vyatta/config/tmp/new_config_ (a unionfs mount) per `configure`, and # a script that dies inside a session never removes it. One of those holds the # commit lock, and from then on EVERY commit fails with "Configuration system # temporarily locked due to another commit in progress" -- including the manual # one you try in order to fix it. A job on a 30s timer that can leak a session # per failure will wedge the router's config system on its own, so it cleans up # before it starts. `umount -l` first: the directory is a mount point and plain # rm returns "Device or resource busy". for d in /opt/vyatta/config/tmp/new_config_*; do [ -d "$d" ] || continue pid=${d##*_} kill -0 "$pid" 2>/dev/null && continue umount -l "$d" 2>/dev/null rm -rf "$d" 2>/dev/null done # The config edit lives in vrrp-wan-apply, because script-template must be the # first thing its script does -- sourced any later it terminates the script # silently with rc=0. See the header there. APPLY=/config/vrrp-wan-apply if holds_vip; then echo master > "$STATE/role" # Stamp only on entry to master, so the health check's grace window measures # time-since-promotion rather than time-since-last-tick. [ -f "$STATE/since" ] || date +%s > "$STATE/since" wan_disabled || ppp_disabled || exit 0 logger -t vrrp-wan "MASTER with WAN disabled -> enabling bond0.${WAN_VIF} + pppoe0" "$APPLY" enable else echo backup > "$STATE/role" rm -f "$STATE/since" { wan_disabled && ppp_disabled; } && exit 0 # Releasing matters more than taking. A demoted router that keeps the WAN up # holds the cloned MAC f0:9f:c2:12:9b:4f on VLAN 53 at the same time as the # new master, and the switch sends the ISP's replies to whichever port spoke # last -- the WAN-side twin of the eth2 incident. logger -t vrrp-wan "not MASTER but WAN enabled -> releasing bond0.${WAN_VIF} + pppoe0" "$APPLY" disable fi # Deliberately no `save`. config.boot keeps `disable` on BOTH routers, so a # reboot in any order comes up unable to claim the shared MAC, and only holding # the VIP re-enables it. NOTE: any `save` while this box is master (a hand # commit, or `pulumi up`) WILL persist the enabled state -- observed in labsim. # The Pulumi model asserts `disable` on both routers so an apply puts it back, # and vyos:verify reports it as drift if it does not.