#!/bin/sh
# Make the WAN match VRRP mastership. Idempotent; safe to run every 30s and on
# every VRRP transition.
#
# Why a reconciler and not just transition scripts
# ------------------------------------------------
# VyOS delivers `transition-script` through a helper process,
# /usr/libexec/vyos/system/keepalived-fifo.py, fed by keepalived's notify_fifo.
# Observed in labsim on 2026-09-02: the primary's Keepalived_vrrp logged
# "(native) Entering MASTER STATE" for all six instances and the built-in
# notify_master for conntrack-sync ran -- while the fifo helper logged NOTHING
# and the master transition script never ran. The helper process was still
# alive. The result was a router holding every VIP with no WAN at all: the exact
# 2026-09-02 outage, re-created by the mechanism meant to prevent it.
#
# So transition scripts are kept for speed but nothing is trusted to them: this
# also runs on a timer, and derives everything from ground truth rather than
# from a marker that only exists if the script it depends on ran.
#
#   vrrp-wan-reconcile          reconcile once
#   vrrp-wan-reconcile --status what it thinks, changing nothing
#
# Ground truth for "am I master" is whether the management VIP is really on this
# box. It is what VRRP actually does, it is observable, and it cannot silently
# disagree with reality.

VIP="${VRRP_WAN_VIP:-192.168.1.1}"      # management VIP; sim overrides via env
WAN_VIF=53                               # bond0.53, the DHCP WAN
STATE=/run/vrrp-wan
LOCK=/run/vrrp-wan.lock

cfg() { /opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands 2>/dev/null; }
holds_vip()   { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; }
wan_up()      { ip -4 addr show "bond0.${WAN_VIF}" 2>/dev/null | grep -q 'inet '; }
wan_disabled(){ cfg | grep -q "vif ${WAN_VIF} disable"; }
ppp_disabled(){ cfg | grep -q "pppoe pppoe0 disable"; }

# --status must answer WITHOUT sourcing script-template. The template's `exit`
# is a function that leaves configuration mode, not the shell builtin, so a
# status run that had sourced it opened and closed a config session on every
# call -- which is how a read-only query started colliding with the timer and
# logging "Configuration system temporarily locked due to another commit".
if [ "${1:-}" = "--status" ]; then
    printf 'vip=%s holds_vip=%s wan_disabled=%s wan_up=%s role=%s\n' \
        "$VIP" "$(holds_vip && echo yes || echo no)" \
        "$(wan_disabled && echo yes || echo no)" \
        "$(wan_up && echo yes || echo no)" \
        "$(cat "$STATE/role" 2>/dev/null || echo unset)"
    exit 0
fi

# One writer. The 30s timer and a VRRP transition can fire together, and two
# VyOS commits in flight on one box do not queue -- the second fails outright.
#
# The lock fd MUST be closed for children (`9>&-` on every call below). Entering
# VyOS config mode spawns a long-lived unionfs-fuse for the session, and it
# INHERITS this descriptor and never lets go -- `lsof /run/vrrp-wan.lock` showed
# it held by unionfs-fuse with fd 9w. From the first config change onward every
# later run lost the flock and exited 0 without doing anything, so the
# reconciler looked healthy in the journal ("Finished") while quietly having
# stopped reconciling. It is how a demoted router kept the WAN.
exec 9>"$LOCK"
flock -n 9 || exit 0

mkdir -p "$STATE"

# Reap config sessions whose owning process is gone. VyOS creates
# /opt/vyatta/config/tmp/new_config_<pid> (a unionfs mount) per `configure`, and
# a script that dies inside a session never removes it. One of those holds the
# commit lock, and from then on EVERY commit fails with "Configuration system
# temporarily locked due to another commit in progress" -- including the manual
# one you try in order to fix it. A job on a 30s timer that can leak a session
# per failure will wedge the router's config system on its own, so it cleans up
# before it starts. `umount -l` first: the directory is a mount point and plain
# rm returns "Device or resource busy".
for d in /opt/vyatta/config/tmp/new_config_*; do
    [ -d "$d" ] || continue
    pid=${d##*_}
    kill -0 "$pid" 2>/dev/null && continue
    umount -l "$d" 2>/dev/null
    rm -rf "$d" 2>/dev/null
done

# The config edit lives in vrrp-wan-apply, because script-template must be the
# first thing its script does -- sourced any later it terminates the script
# silently with rc=0. See the header there.
APPLY=/config/vrrp-wan-apply

if holds_vip; then
    echo master > "$STATE/role"
    # Stamp only on entry to master, so the health check's grace window measures
    # time-since-promotion rather than time-since-last-tick.
    [ -f "$STATE/since" ] || date +%s > "$STATE/since"
    wan_disabled || ppp_disabled || exit 0
    logger -t vrrp-wan "MASTER with WAN disabled -> enabling bond0.${WAN_VIF} + pppoe0"
    "$APPLY" enable 9>&-
else
    echo backup > "$STATE/role"
    rm -f "$STATE/since"
    { wan_disabled && ppp_disabled; } && exit 0
    # Releasing matters more than taking. A demoted router that keeps the WAN up
    # holds the cloned MAC f0:9f:c2:12:9b:4f on VLAN 53 at the same time as the
    # new master, and the switch sends the ISP's replies to whichever port spoke
    # last -- the WAN-side twin of the eth2 incident.
    logger -t vrrp-wan "not MASTER but WAN enabled -> releasing bond0.${WAN_VIF} + pppoe0"
    "$APPLY" disable 9>&-
fi

# Deliberately no `save`. config.boot keeps `disable` on BOTH routers, so a
# reboot in any order comes up unable to claim the shared MAC, and only holding
# the VIP re-enables it. NOTE: any `save` while this box is master (a hand
# commit, or `pulumi up`) WILL persist the enabled state -- observed in labsim.
# The Pulumi model asserts `disable` on both routers so an apply puts it back,
# and vyos:verify reports it as drift if it does not.
