Compare commits
56 Commits
27a343bc75
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c729275961 | ||
|
|
a4dcc9b379 | ||
|
|
97ae6dea89 | ||
|
|
2048515578 | ||
|
|
fd38c05c3e | ||
|
|
3ed2099083 | ||
|
|
ac8653f519 | ||
|
|
6c94371c8e | ||
|
|
7c2cbfaf31 | ||
|
|
3e43385639 | ||
|
|
d9f74aa294 | ||
|
|
c392bb9233 | ||
|
|
3c933b96e0 | ||
|
|
5c9f004759 | ||
|
|
914135c47d | ||
|
|
a9ff182bd6 | ||
|
|
51bf300474 | ||
|
|
cb03987c33 | ||
|
|
8a0ef52909 | ||
|
|
da86b60dce | ||
|
|
d0b733f831 | ||
|
|
395577850c | ||
|
|
061b9e3d7e | ||
|
|
bda5854563 | ||
|
|
d08f68e28b | ||
|
|
e1c571d004 | ||
|
|
47ce0c1aea | ||
|
|
85819e40c1 | ||
|
|
13dcdff1ef | ||
|
|
8aa3d0ebaa | ||
|
|
8fce03e705 | ||
|
|
5ed0e4888a | ||
|
|
97efb5abb2 | ||
|
|
6987b324f1 | ||
|
|
9221c71ff0 | ||
|
|
4d47b609a2 | ||
|
|
b659e0d47e | ||
|
|
d626750010 | ||
|
|
93fed7826b | ||
|
|
4efd70c987 | ||
|
|
2e828b8af2 | ||
|
|
7bf3f42e19 | ||
|
|
09ede73b67 | ||
|
|
a973b51b9c | ||
|
|
d727a50ca0 | ||
|
|
0481c38e09 | ||
|
|
23783b8486 | ||
|
|
11344eab92 | ||
|
|
e8679f45b5 | ||
|
|
527e0798ae | ||
|
|
842408c0d9 | ||
|
|
ad6eb7a9a6 | ||
|
|
7f551081ad | ||
|
|
f41ffdd039 | ||
|
|
a187703a3a | ||
|
|
86c2a36f00 |
6
.gitignore
vendored
6
.gitignore
vendored
@@ -31,3 +31,9 @@ node_modules/
|
||||
# Asahi build artifacts (large)
|
||||
bastion/.asahi-cache/
|
||||
bastion/asahi-repo/*.zip
|
||||
|
||||
# Regenerated by labsim/dualstack-lab.sh; derived state, not source.
|
||||
labsim/dualstack-evidence/
|
||||
|
||||
# Runtime snapshots from labsim/cilium-ipam-switch.sh
|
||||
labsim/.ipam-switch-state/
|
||||
|
||||
@@ -134,7 +134,13 @@ lang ${locale}
|
||||
keyboard uk
|
||||
timezone ${timezone} --utc
|
||||
|
||||
network --bootproto=dhcp --activate --hostname=${fqdn}
|
||||
# --ipv6=auto, not =dhcp: "auto" follows the router advertisement, so the RA's
|
||||
# managed-flag is what steers the node to DHCPv6, and a VLAN with no DHCPv6 yet
|
||||
# still installs instead of blocking on a lease that will never come. Failing an
|
||||
# OS install because IPv6 was not ready would be a worse trade than a node that
|
||||
# briefly has no v6. The address itself comes from a kea DHCPv6 reservation
|
||||
# keyed on MAC -- the same source of truth as the v4 address.
|
||||
network --bootproto=dhcp --ipv6=auto --activate --hostname=${fqdn}
|
||||
|
||||
${auth}
|
||||
${userDirective}
|
||||
@@ -366,6 +372,21 @@ fs.inotify.max_user_watches = 1048576
|
||||
SYSCTL
|
||||
sysctl --system || true
|
||||
|
||||
# -- IPv6 link-local address generation: EUI-64, fleet-wide --
|
||||
# A cluster node takes its IPv6 from a MAC-keyed DHCPv6 reservation. kea can only
|
||||
# match that reservation when it can recover the node's MAC, and for a modern
|
||||
# client that sends a DUID-UUID (no MAC in it) the only place kea can find one is
|
||||
# an EUI-64 link-local. NetworkManager's default is stable-privacy (RFC 7217),
|
||||
# whose link-local hides the MAC -- so a node on the default silently never gets
|
||||
# its reserved address, and with a reservations-only subnet it gets nothing at
|
||||
# all. Proven on 2026-09-06: worker0/worker2 (eui64) bound; worker1/spark
|
||||
# (default) did not, until flipped. Setting it here means a new node is correct
|
||||
# from first boot, before its connection is ever activated.
|
||||
cat > /etc/NetworkManager/conf.d/10-ipv6-eui64.conf << 'NMEUI64'
|
||||
[connection]
|
||||
ipv6.addr-gen-mode=eui64
|
||||
NMEUI64
|
||||
|
||||
# -- Disable firewalld permanently (k3s/Cilium manage iptables directly) --
|
||||
# Note: no '--now' — systemd is not running in the Anaconda chroot
|
||||
systemctl disable firewalld || true
|
||||
|
||||
@@ -40,34 +40,39 @@ export function renderUbuntuAutoinstall(params: UbuntuAutoinstallParams): string
|
||||
// Build the LVM layout to match Fedora kickstart sizes
|
||||
const extraLvs: string[] = [];
|
||||
if (hasLonghorn) {
|
||||
extraLvs.push(` - id: lv-longhorn
|
||||
name: longhorn
|
||||
type: lvm_partition
|
||||
volgroup: vg0
|
||||
size: -1
|
||||
- id: fs-longhorn
|
||||
type: format
|
||||
volume: lv-longhorn
|
||||
fstype: xfs
|
||||
- id: mount-longhorn
|
||||
type: mount
|
||||
device: fs-longhorn
|
||||
path: /var/lib/longhorn`);
|
||||
// 6 spaces for the list item, 8 for its keys -- these are siblings of the
|
||||
// lv-home/lv-srv entries in storage.config, which sit at 6. At 8 the
|
||||
// rendered document is not valid YAML at all ("expected <block end>, but
|
||||
// found '-'"), so an Ubuntu node with the longhorn role could never have
|
||||
// installed. Ubuntu + longhorn is exactly the worker shape.
|
||||
extraLvs.push(` - id: lv-longhorn
|
||||
name: longhorn
|
||||
type: lvm_partition
|
||||
volgroup: vg0
|
||||
size: -1
|
||||
- id: fs-longhorn
|
||||
type: format
|
||||
volume: lv-longhorn
|
||||
fstype: xfs
|
||||
- id: mount-longhorn
|
||||
type: mount
|
||||
device: fs-longhorn
|
||||
path: /var/lib/longhorn`);
|
||||
}
|
||||
if (hasRancher) {
|
||||
extraLvs.push(` - id: lv-rancher
|
||||
name: rancher
|
||||
type: lvm_partition
|
||||
volgroup: vg0
|
||||
size: 20G
|
||||
- id: fs-rancher
|
||||
type: format
|
||||
volume: lv-rancher
|
||||
fstype: xfs
|
||||
- id: mount-rancher
|
||||
type: mount
|
||||
device: fs-rancher
|
||||
path: /var/lib/rancher`);
|
||||
extraLvs.push(` - id: lv-rancher
|
||||
name: rancher
|
||||
type: lvm_partition
|
||||
volgroup: vg0
|
||||
size: 20G
|
||||
- id: fs-rancher
|
||||
type: format
|
||||
volume: lv-rancher
|
||||
fstype: xfs
|
||||
- id: mount-rancher
|
||||
type: mount
|
||||
device: fs-rancher
|
||||
path: /var/lib/rancher`);
|
||||
}
|
||||
|
||||
const extraLvsBlock = extraLvs.length > 0 ? "\n" + extraLvs.join("\n") : "";
|
||||
@@ -81,6 +86,11 @@ export function renderUbuntuAutoinstall(params: UbuntuAutoinstallParams): string
|
||||
`curtin in-target -- bash -c 'cat > /etc/modules-load.d/k3s.conf << EOF\nbr_netfilter\noverlay\nip_conntrack\nEOF'`,
|
||||
// Sysctl for k3s networking
|
||||
`curtin in-target -- bash -c 'cat > /etc/sysctl.d/90-k3s.conf << EOF\nnet.bridge.bridge-nf-call-iptables = 1\nnet.bridge.bridge-nf-call-ip6tables = 1\nnet.ipv4.ip_forward = 1\nnet.ipv6.conf.all.forwarding = 1\nfs.inotify.max_user_instances = 524288\nfs.inotify.max_user_watches = 1048576\nEOF'`,
|
||||
// IPv6 link-local = EUI-64, so a MAC-keyed DHCPv6 reservation can match: kea
|
||||
// recovers the node's MAC from an EUI-64 link-local when the client sends a
|
||||
// DUID-UUID (no MAC in it). NM's stable-privacy default hides the MAC and the
|
||||
// node silently never gets its reserved address. Proven 2026-09-06.
|
||||
`curtin in-target -- bash -c 'cat > /etc/NetworkManager/conf.d/10-ipv6-eui64.conf << EOF\n[connection]\nipv6.addr-gen-mode=eui64\nEOF'`,
|
||||
// Disable ufw firewall
|
||||
`curtin in-target -- systemctl disable ufw || true`,
|
||||
// Enable chrony/ntp
|
||||
@@ -121,7 +131,20 @@ export function renderUbuntuAutoinstall(params: UbuntuAutoinstallParams): string
|
||||
`curtin in-target -- bash -c 'IP_ADDR=$(ip -4 addr show | awk "/inet / && !/127.0.0/ {split(\\$2,a,\\"/\\"); print a[1]; exit}"); curl -sf -X POST "http://${serverIp}:${httpPort}/api/progress" -H "Content-Type: application/json" -d "{\\"mac\\":\\"$(ip link show | awk "/ether/ && !/00:00:00:00/ {print \\$2; exit}")\\",\\"stage\\":\\"complete\\",\\"detail\\":\\"ready at $IP_ADDR\\"}" || true'`,
|
||||
);
|
||||
|
||||
const lateCommandsYaml = lateCommands.map((c) => ` - "${c}"`).join("\n");
|
||||
// JSON.stringify, not `"${c}"`. JSON is a subset of YAML, so this produces a
|
||||
// correctly escaped double-quoted scalar for free -- and the naive version
|
||||
// was broken in two ways at once, for every role:
|
||||
//
|
||||
// * embedded double quotes ended the scalar early
|
||||
// (`echo "tmpfs /tmp ..." >> /etc/fstab` -> "expected <block end>")
|
||||
// * the heredocs contain REAL newlines, and YAML folds newlines inside a
|
||||
// double-quoted scalar into spaces -- so even where it parsed, the
|
||||
// heredoc arrived at the target as one long line and wrote a file with
|
||||
// no line breaks.
|
||||
//
|
||||
// JSON escaping turns the newlines into \n, which YAML unescapes back to
|
||||
// real newlines on parse, so the heredoc survives intact.
|
||||
const lateCommandsYaml = lateCommands.map((c) => ` - ${JSON.stringify(c)}`).join("\n");
|
||||
|
||||
return `#cloud-config
|
||||
autoinstall:
|
||||
@@ -139,6 +162,30 @@ autoinstall:
|
||||
allow-pw: false
|
||||
authorized-keys:
|
||||
${sshKeysYaml}
|
||||
# Both address families. Without dhcp6 the installer's default is IPv4-only,
|
||||
# so a node provisioned into a dual-stack cluster comes up with no IPv6, k3s
|
||||
# has no v6 node-ip to bind, and it joins as an IPv4-only member of a
|
||||
# dual-stack cluster -- which surfaces later as pods on that node being
|
||||
# unreachable over v6 while the node itself reads Ready.
|
||||
#
|
||||
# The address itself comes from a kea DHCPv6 reservation keyed on MAC, the
|
||||
# same source of truth as the v4 address, so nothing here needs to know it.
|
||||
#
|
||||
# optional: true matters -- it lets the install proceed if the v6 lease is
|
||||
# slow or the VLAN has no DHCPv6 yet, rather than blocking on a timeout. The
|
||||
# node still needs the address before k3s starts, but that is a later step's
|
||||
# problem and failing the OS install over it would be worse.
|
||||
# (No backticks in this comment: it lives inside a TS template literal, and a
|
||||
# backtick here ends the literal and breaks the build.)
|
||||
network:
|
||||
version: 2
|
||||
ethernets:
|
||||
primary:
|
||||
match:
|
||||
name: "en*"
|
||||
dhcp4: true
|
||||
dhcp6: true
|
||||
optional: true
|
||||
storage:
|
||||
config:
|
||||
- id: disk0
|
||||
|
||||
87
bastion/src/bastion/tests/ubuntu-autoinstall.test.ts
Normal file
87
bastion/src/bastion/tests/ubuntu-autoinstall.test.ts
Normal file
@@ -0,0 +1,87 @@
|
||||
// The Ubuntu autoinstall document must be valid YAML for EVERY role.
|
||||
//
|
||||
// There was no test here, and the template shipped a document that did not
|
||||
// parse: the longhorn/rancher LVM entries were indented 8 spaces while their
|
||||
// siblings in storage.config sit at 6, giving "expected <block end>, but found
|
||||
// '-'". Every role that gets a longhorn volume -- which is the worker shape --
|
||||
// rendered an uninstallable document. A `toContain` assertion would not have
|
||||
// caught that; only parsing does.
|
||||
//
|
||||
// Parsed with python3's yaml rather than a new npm dependency, mirroring how
|
||||
// kickstart.test.ts shells out to `ksvalidator`: the point is to check the
|
||||
// artefact with a real parser, not to grow the dependency tree.
|
||||
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { execFileSync } from "node:child_process";
|
||||
import { writeFileSync, unlinkSync } from "node:fs";
|
||||
import { renderUbuntuAutoinstall } from "../src/templates/ubuntu-autoinstall.js";
|
||||
|
||||
const base = {
|
||||
hostname: "n6",
|
||||
disk: "/dev/sda",
|
||||
domain: "ad.itaz.eu",
|
||||
ubuntuVersion: "24.04",
|
||||
timezone: "Europe/London",
|
||||
locale: "en_GB.UTF-8",
|
||||
serverIp: "10.0.0.1",
|
||||
httpPort: 8080,
|
||||
sshKeys: ["ssh-ed25519 AAAAtest test@lab"],
|
||||
adminUser: "root",
|
||||
};
|
||||
|
||||
/** Parse with python3's yaml and return the document as JSON. */
|
||||
function parseYaml(text: string, label: string): Record<string, any> {
|
||||
const tmp = `/tmp/autoinstall-test-${label}.yaml`;
|
||||
writeFileSync(tmp, text);
|
||||
try {
|
||||
const out = execFileSync(
|
||||
"python3",
|
||||
["-c", "import sys,yaml,json; json.dump(yaml.safe_load(open(sys.argv[1])), sys.stdout)", tmp],
|
||||
{ encoding: "utf-8" },
|
||||
);
|
||||
return JSON.parse(out);
|
||||
} catch (err: unknown) {
|
||||
const msg = err instanceof Error ? (err as { stderr?: string }).stderr ?? err.message : String(err);
|
||||
throw new Error(`autoinstall YAML did not parse for ${label}: ${msg}`);
|
||||
} finally {
|
||||
try { unlinkSync(tmp); } catch { /* ignore */ }
|
||||
}
|
||||
}
|
||||
|
||||
describe("renderUbuntuAutoinstall", () => {
|
||||
for (const role of ["vanilla", "worker", "infra"]) {
|
||||
it(`renders parseable YAML for role=${role}`, () => {
|
||||
const doc = parseYaml(renderUbuntuAutoinstall({ ...base, role }), role);
|
||||
expect(doc.autoinstall).toBeDefined();
|
||||
expect(doc.autoinstall.version).toBe(1);
|
||||
// storage.config must be a flat list; the indentation bug produced a
|
||||
// nested map here, which is how it went unnoticed.
|
||||
expect(Array.isArray(doc.autoinstall.storage.config)).toBe(true);
|
||||
});
|
||||
}
|
||||
|
||||
it("gives the longhorn role its volume as a sibling entry, not a nested map", () => {
|
||||
const doc = parseYaml(renderUbuntuAutoinstall({ ...base, role: "worker" }), "longhorn");
|
||||
const ids = doc.autoinstall.storage.config.map((e: { id: string }) => e.id);
|
||||
expect(ids).toContain("lv-longhorn");
|
||||
expect(ids).toContain("mount-longhorn");
|
||||
});
|
||||
|
||||
it("sets EUI-64 link-local so MAC-keyed DHCPv6 reservations can match", () => {
|
||||
const doc = parseYaml(renderUbuntuAutoinstall({ ...base, role: "worker" }), "eui64");
|
||||
const late = (doc.autoinstall["late-commands"] as string[]).join("\n");
|
||||
expect(late).toContain("10-ipv6-eui64.conf");
|
||||
expect(late).toContain("ipv6.addr-gen-mode=eui64");
|
||||
});
|
||||
|
||||
it("requests both address families on the primary NIC", () => {
|
||||
const doc = parseYaml(renderUbuntuAutoinstall({ ...base, role: "worker" }), "net");
|
||||
const eth = doc.autoinstall.network.ethernets.primary;
|
||||
expect(eth.dhcp4).toBe(true);
|
||||
// Without this a node provisioned into a dual-stack cluster comes up with
|
||||
// no IPv6 and joins as an IPv4-only member.
|
||||
expect(eth.dhcp6).toBe(true);
|
||||
// The install must not block waiting for a v6 lease that may never come.
|
||||
expect(eth.optional).toBe(true);
|
||||
});
|
||||
});
|
||||
42
bastion/src/modules/modules/k3s/bin/render-config.ts
Normal file
42
bastion/src/modules/modules/k3s/bin/render-config.ts
Normal file
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env node
|
||||
// Render /etc/rancher/k3s/config.yaml using the PRODUCTION generator.
|
||||
//
|
||||
// Exists so labsim (and anything else) can produce the exact config.yaml a node
|
||||
// would get from `labctl install`, without an SSH/OperationContext. Driving the
|
||||
// rehearsal through this means the sim tests the same code path production runs
|
||||
// -- if generateServerConfig ever changes shape, the sim moves with it.
|
||||
//
|
||||
// All input via env, so a sim's cloud-init or a shell can call it plainly:
|
||||
//
|
||||
// ROLE=infra HOSTNAME=k8s1 IP=172.31.2.11 render-config.ts # cluster-init server
|
||||
// ROLE=infra HOSTNAME=k8s2 IP=172.31.2.12 \
|
||||
// K3S_SERVER_URL=https://172.31.2.11:6443 K3S_TOKEN=... render-config.ts # joining server
|
||||
// ROLE=worker HOSTNAME=k8s4 IP=172.31.2.14 K3S_SERVER_URL=... K3S_TOKEN=... render-config.ts # agent
|
||||
//
|
||||
// Dual-stack is opt-in and matches K3sConfig exactly: set IPV6, CLUSTER_CIDR,
|
||||
// SERVICE_CIDR (comma-separated families) and the generator emits the dual
|
||||
// node-ip + CIDRs. Omit them and the output is byte-identical to a v4-only node.
|
||||
import { generateServerConfig, generateAgentConfig } from "../src/operations/k3s-config.js";
|
||||
import type { K3sConfig } from "../src/types.js";
|
||||
import type { Role } from "@lab/shared";
|
||||
|
||||
const env = process.env;
|
||||
const role = (env.ROLE ?? "worker") as Role;
|
||||
const isServer = role === "infra" || role === "labcontroller";
|
||||
|
||||
const splitCsv = (v: string | undefined): string[] | undefined =>
|
||||
v ? v.split(",").map((s) => s.trim()).filter(Boolean) : undefined;
|
||||
|
||||
const cfg: K3sConfig = {
|
||||
hostname: env.HOSTNAME ?? "node",
|
||||
ip: env.IP ?? (() => { throw new Error("IP is required"); })(),
|
||||
role,
|
||||
k3sServerUrl: env.K3S_SERVER_URL,
|
||||
k3sToken: env.K3S_TOKEN,
|
||||
tlsSans: splitCsv(env.TLS_SANS),
|
||||
ipv6: env.IPV6,
|
||||
clusterCidr: splitCsv(env.CLUSTER_CIDR),
|
||||
serviceCidr: splitCsv(env.SERVICE_CIDR),
|
||||
};
|
||||
|
||||
process.stdout.write(isServer ? generateServerConfig(cfg) : generateAgentConfig(cfg));
|
||||
@@ -215,7 +215,9 @@ echo " Using network device: \$DEFAULT_DEV"
|
||||
|
||||
KUBECONFIG=/etc/rancher/k3s/k3s.yaml cilium install \\
|
||||
--set kubeProxyReplacement=true \\
|
||||
--set ipam.mode=kubernetes \\
|
||||
--set ipam.mode=cluster-pool \\
|
||||
--set ipam.operator.clusterPoolIPv4PodCIDRList='{10.42.0.0/16}' \\
|
||||
--set ipam.operator.clusterPoolIPv4MaskSize=24 \\
|
||||
--set devices="\$DEFAULT_DEV" \\
|
||||
--set nodePort.directRoutingDevice="\$DEFAULT_DEV"
|
||||
|
||||
|
||||
@@ -47,7 +47,9 @@ export const installCilium: Operation = async (ctx): Promise<OperationResult> =>
|
||||
const installResult = await ctx.ssh.exec(
|
||||
`KUBECONFIG=/etc/rancher/k3s/k3s.yaml cilium install \
|
||||
--set kubeProxyReplacement=true \
|
||||
--set ipam.mode=kubernetes \
|
||||
--set ipam.mode=cluster-pool \
|
||||
--set ipam.operator.clusterPoolIPv4PodCIDRList='{10.42.0.0/16}' \
|
||||
--set ipam.operator.clusterPoolIPv4MaskSize=24 \
|
||||
--set k8sServiceHost=127.0.0.1 \
|
||||
--set k8sServicePort=6444 \
|
||||
--set cni.exclusive=false \
|
||||
|
||||
@@ -5,7 +5,7 @@ export { growRancherLv } from "./rancher-storage.js";
|
||||
export { enableIscsi } from "./iscsi.js";
|
||||
export { disableFirewall } from "./firewall.js";
|
||||
export { setSelinuxPermissive } from "./selinux.js";
|
||||
export { writeK3sConfig } from "./k3s-config.js";
|
||||
export { writeK3sConfig, generateServerConfig, generateAgentConfig } from "./k3s-config.js";
|
||||
export { writeAuditPolicy } from "./audit-policy.js";
|
||||
export { cleanupStaleCni } from "./cni-cleanup.js";
|
||||
export { installK3sBinary } from "./k3s-install.js";
|
||||
|
||||
@@ -7,8 +7,52 @@ function isServerRole(role: string): boolean {
|
||||
return role === "infra" || role === "labcontroller";
|
||||
}
|
||||
|
||||
function generateServerConfig(config: K3sConfig): string {
|
||||
const tlsSans = [config.hostname, config.ip, ...(config.tlsSans ?? [])];
|
||||
/**
|
||||
* The address-family block: `cluster-cidr`, `service-cidr` and `node-ip`.
|
||||
*
|
||||
* Emitted ONLY when the corresponding config is supplied, and that is
|
||||
* deliberate. With no `ipv6` and no CIDRs this returns "", so the generated
|
||||
* file is byte-identical to what every existing node already has -- no diff,
|
||||
* so `writeRemoteFile` reports unchanged and nothing restarts k3s. Dual-stack
|
||||
* is therefore opt-in per node rather than a flag day.
|
||||
*
|
||||
* `node-ip` is only written once there is a second family to name. k3s
|
||||
* auto-detects a sensible IPv4 on its own, and writing it out unconditionally
|
||||
* would rewrite the config of five healthy nodes to tell them what they had
|
||||
* already worked out.
|
||||
*/
|
||||
function addressFamilyLines(config: K3sConfig, opts: { cidrs: boolean }): string {
|
||||
const lines: string[] = [];
|
||||
if (opts.cidrs && config.clusterCidr?.length) {
|
||||
lines.push(`cluster-cidr: "${config.clusterCidr.join(",")}"`);
|
||||
}
|
||||
if (opts.cidrs && config.serviceCidr?.length) {
|
||||
lines.push(`service-cidr: "${config.serviceCidr.join(",")}"`);
|
||||
}
|
||||
if (config.ipv6) {
|
||||
// Order matters to k3s: the FIRST entry is the primary family, and the
|
||||
// supported single-to-dual-stack conversion is the one that preserves it.
|
||||
// IPv4 stays primary so existing Services keep their ClusterIP.
|
||||
lines.push(`node-ip: "${config.ip},${config.ipv6}"`);
|
||||
}
|
||||
return lines.length ? `${lines.join("\n")}\n` : "";
|
||||
}
|
||||
|
||||
// Exported so the exact production config.yaml can be rendered outside an SSH
|
||||
// context -- notably by the labsim 3-server-etcd rehearsal, which must drive its
|
||||
// nodes through THIS generator rather than a parallel set of INSTALL_K3S_EXEC
|
||||
// flags, or it proves a mechanism production does not run.
|
||||
export function generateServerConfig(config: K3sConfig): string {
|
||||
// The IPv6 address goes in the cert too. Without it, anything that reaches
|
||||
// this apiserver over v6 -- a peer server joining, or kubectl against the v6
|
||||
// address -- fails TLS verification, and the error names the certificate
|
||||
// rather than the missing SAN, which is a long way from the cause.
|
||||
const tlsSans = [
|
||||
config.hostname,
|
||||
config.ip,
|
||||
...(config.ipv6 ? [config.ipv6] : []),
|
||||
...(config.tlsSans ?? []),
|
||||
];
|
||||
const isJoining = !!config.k3sServerUrl;
|
||||
const clusterLines = isJoining
|
||||
? `server: "${config.k3sServerUrl}"\ntoken: "${config.k3sToken}"`
|
||||
@@ -21,7 +65,7 @@ function generateServerConfig(config: K3sConfig): string {
|
||||
// and never expire.
|
||||
return `# k3s server configuration — CIS hardened, etcd HA
|
||||
${clusterLines}
|
||||
protect-kernel-defaults: true
|
||||
${addressFamilyLines(config, { cidrs: true })}protect-kernel-defaults: true
|
||||
secrets-encryption: true
|
||||
write-kubeconfig-mode: "0640"
|
||||
|
||||
@@ -51,8 +95,13 @@ ${tlsSans.map((s) => ` - "${s}"`).join("\n")}
|
||||
`;
|
||||
}
|
||||
|
||||
function generateAgentConfig(): string {
|
||||
return `protect-kernel-defaults: true
|
||||
// Takes the config now: an agent needs its own dual `node-ip` just as much as a
|
||||
// server does. Without one it joins as an IPv4-only node into a dual-stack
|
||||
// cluster, gets no IPv6 pod CIDR, and the failure surfaces later as pods on that
|
||||
// node being unreachable over v6 while the node itself reads Ready.
|
||||
// It takes no cluster/service CIDRs -- those are server-side only.
|
||||
export function generateAgentConfig(config: K3sConfig): string {
|
||||
return `${addressFamilyLines(config, { cidrs: false })}protect-kernel-defaults: true
|
||||
node-label:
|
||||
- "node-role.kubernetes.io/worker=true"
|
||||
- "node.longhorn.io/create-default-disk=config"
|
||||
@@ -64,11 +113,41 @@ kubelet-arg:
|
||||
}
|
||||
|
||||
export const writeK3sConfig: Operation = async (ctx): Promise<OperationResult> => {
|
||||
// Refuse to name an address the node does not have.
|
||||
//
|
||||
// Most of this estate is SSH-onboard, not PXE-provisioned: Asahi cannot PXE
|
||||
// at all, and the DGX Sparks run NVIDIA's own OS and must never be
|
||||
// reinstalled. For those nodes the install templates govern nothing and this
|
||||
// module is the ONLY labctl touchpoint, so nothing upstream can guarantee the
|
||||
// vendor OS actually took a DHCPv6 lease.
|
||||
//
|
||||
// Writing node-ip for a missing address does not fail here -- it fails later,
|
||||
// when k3s will not start, with an error about binding rather than about
|
||||
// addressing. Checking costs one ssh round trip and turns a confusing
|
||||
// start-up failure into a sentence naming the address and the node.
|
||||
if (ctx.config.ipv6) {
|
||||
const probe = await ctx.ssh.exec(
|
||||
`ip -6 -o addr show 2>/dev/null | grep -qF " ${ctx.config.ipv6}/" && echo present || true`,
|
||||
sshOpts(ctx),
|
||||
);
|
||||
if (!probe.stdout.includes("present")) {
|
||||
return {
|
||||
success: false,
|
||||
changed: false,
|
||||
message: `Node does not have IPv6 ${ctx.config.ipv6}`,
|
||||
error:
|
||||
`k3s config would set node-ip to ${ctx.config.ipv6}, but that address is not on any ` +
|
||||
`interface. k3s resolves node-ip at start-up and would fail to bind. Check the node ` +
|
||||
`took its DHCPv6 lease (a kea reservation keyed on its MAC) before retrying.`,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
await ctx.ssh.exec("mkdir -p /etc/rancher/k3s", sshOpts(ctx));
|
||||
|
||||
const content = isServerRole(ctx.config.role)
|
||||
? generateServerConfig(ctx.config)
|
||||
: generateAgentConfig();
|
||||
: generateAgentConfig(ctx.config);
|
||||
|
||||
const changed = await writeRemoteFile(ctx, "/etc/rancher/k3s/config.yaml", content);
|
||||
|
||||
|
||||
@@ -16,6 +16,33 @@ export interface K3sConfig {
|
||||
|
||||
// Additional TLS SANs for API server certificate
|
||||
tlsSans?: string[] | undefined;
|
||||
|
||||
/**
|
||||
* IPv6 address of this node on the cluster VLAN. Its PRESENCE is what makes a
|
||||
* node dual-stack: supply it and `node-ip` is emitted as `<v4>,<v6>`; omit it
|
||||
* and the generated config is byte-identical to the IPv4-only one.
|
||||
*
|
||||
* It must be a real address on the node before k3s starts. k3s resolves
|
||||
* node-ip at boot, so a SLAAC or DHCPv6 lease that has not landed yet leaves
|
||||
* the node with no usable v6 identity. The estate takes it from a kea DHCPv6
|
||||
* reservation keyed on MAC, the same way it takes its IPv4 address.
|
||||
*/
|
||||
ipv6?: string | undefined;
|
||||
|
||||
/**
|
||||
* Pod and Service ranges, one entry per address family. Passed rather than
|
||||
* hardcoded so the ranges stay configuration -- and so labsim can rehearse
|
||||
* with its own addresses through this same generator, instead of a parallel
|
||||
* set of INSTALL_K3S_EXEC flags that would prove a different mechanism.
|
||||
*
|
||||
* k3s validates the two together and crash-loops on a mismatch (a dual
|
||||
* cluster-cidr with a single-family service-cidr, say), which is a safe
|
||||
* failure but an avoidable one: set both or neither.
|
||||
*
|
||||
* Servers only -- agents take neither.
|
||||
*/
|
||||
clusterCidr?: string[] | undefined;
|
||||
serviceCidr?: string[] | undefined;
|
||||
}
|
||||
|
||||
/** SSH execution interface injected into operations. */
|
||||
|
||||
@@ -268,6 +268,108 @@ describe("writeK3sConfig", () => {
|
||||
expect(writeCall).toContain("protect-kernel-defaults: true");
|
||||
expect(writeCall).not.toContain("secrets-encryption");
|
||||
});
|
||||
|
||||
// --- dual-stack ---
|
||||
//
|
||||
// The property that matters most is the NEGATIVE one: with no ipv6 and no
|
||||
// CIDRs the output must be byte-identical to what the five existing nodes
|
||||
// already have on disk. If it is not, rolling this out rewrites every node's
|
||||
// config.yaml and restarts a healthy cluster to tell it what it already knew.
|
||||
|
||||
// With an ipv6 configured there is an extra ssh round trip up front -- the
|
||||
// probe that checks the address is really on the node -- so the write lands
|
||||
// one call later.
|
||||
const writtenBy = async (config: Parameters<typeof mockCtx>[0]) => {
|
||||
const ctx = mockCtx(config);
|
||||
const hasV6 = !!(config as { ipv6?: string }).ipv6;
|
||||
if (hasV6) ctx.ssh.exec.mockResolvedValueOnce(stdout("present"));
|
||||
ctx.ssh.exec
|
||||
.mockResolvedValueOnce(OK)
|
||||
.mockResolvedValueOnce(stdout("__LABCTL_NOT_FOUND__"))
|
||||
.mockResolvedValueOnce(OK);
|
||||
await writeK3sConfig(ctx);
|
||||
return ctx.ssh.exec.mock.calls[hasV6 ? 3 : 2]![0] as string;
|
||||
};
|
||||
|
||||
it("emits no address-family lines at all when single-stack", async () => {
|
||||
const server = await writtenBy({ hostname: "n1.lab", ip: "10.0.1.1", role: "infra" });
|
||||
expect(server).not.toContain("node-ip");
|
||||
expect(server).not.toContain("cluster-cidr");
|
||||
expect(server).not.toContain("service-cidr");
|
||||
|
||||
const agent = await writtenBy({ role: "worker" });
|
||||
expect(agent).not.toContain("node-ip");
|
||||
// Nothing inserted ahead of, or between, the lines the agent config has
|
||||
// always opened with.
|
||||
expect(agent).toContain("protect-kernel-defaults: true\nnode-label:");
|
||||
});
|
||||
|
||||
it("emits dual node-ip and CIDRs on a server, IPv4 first", async () => {
|
||||
const out = await writtenBy({
|
||||
hostname: "n1.lab",
|
||||
ip: "192.168.8.23",
|
||||
role: "infra",
|
||||
ipv6: "2001:470:187e:2::23",
|
||||
clusterCidr: ["10.42.0.0/16", "2001:470:187e:1000::/56"],
|
||||
serviceCidr: ["10.43.0.0/16", "2001:470:187e:1fff::/112"],
|
||||
});
|
||||
expect(out).toContain('node-ip: "192.168.8.23,2001:470:187e:2::23"');
|
||||
expect(out).toContain('cluster-cidr: "10.42.0.0/16,2001:470:187e:1000::/56"');
|
||||
expect(out).toContain('service-cidr: "10.43.0.0/16,2001:470:187e:1fff::/112"');
|
||||
// IPv4 must stay the primary family -- that is the supported conversion
|
||||
// path and what lets existing Services keep their ClusterIP.
|
||||
expect(out.indexOf("10.42.0.0/16")).toBeLessThan(out.indexOf("2001:470:187e:1000::/56"));
|
||||
// still a valid server config
|
||||
expect(out).toContain("cluster-init: true");
|
||||
expect(out).toContain("secrets-encryption: true");
|
||||
// the v6 address must be a TLS SAN, or a peer joining over v6 fails
|
||||
// verification with an error that names the cert, not the missing SAN
|
||||
expect(out).toContain(' - "2001:470:187e:2::23"');
|
||||
});
|
||||
|
||||
it("gives an AGENT its own dual node-ip but no CIDRs", async () => {
|
||||
const out = await writtenBy({
|
||||
ip: "192.168.8.12",
|
||||
role: "worker",
|
||||
ipv6: "2001:470:187e:2::12",
|
||||
// deliberately supplied: an agent must ignore them
|
||||
clusterCidr: ["10.42.0.0/16", "2001:470:187e:1000::/56"],
|
||||
serviceCidr: ["10.43.0.0/16", "2001:470:187e:1fff::/112"],
|
||||
});
|
||||
expect(out).toContain('node-ip: "192.168.8.12,2001:470:187e:2::12"');
|
||||
expect(out).not.toContain("cluster-cidr");
|
||||
expect(out).not.toContain("service-cidr");
|
||||
});
|
||||
|
||||
it("refuses to write a config naming an IPv6 the node does not have", async () => {
|
||||
// The SSH-onboard case: Asahi and the DGX Sparks run an OS labctl never
|
||||
// installed, so nothing upstream guarantees a DHCPv6 lease was taken.
|
||||
// Writing the config anyway defers the failure to k3s start-up, where it
|
||||
// reads as a bind error rather than a missing address.
|
||||
const ctx = mockCtx({ ip: "192.168.8.12", role: "worker", ipv6: "2001:470:187e:2::12" });
|
||||
ctx.ssh.exec.mockResolvedValueOnce(stdout("")); // probe: address absent
|
||||
|
||||
const result = await writeK3sConfig(ctx);
|
||||
expect(result.success).toBe(false);
|
||||
expect(result.changed).toBe(false);
|
||||
expect(result.error).toContain("2001:470:187e:2::12");
|
||||
// and it must not have written anything
|
||||
expect(ctx.ssh.exec).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("does not write node-ip for a v4-only node even when CIDRs are given", async () => {
|
||||
// Guards the flag-day risk: supplying ranges alone must not start rewriting
|
||||
// node identity on nodes that have no IPv6 yet.
|
||||
const out = await writtenBy({
|
||||
hostname: "n1.lab",
|
||||
ip: "10.0.1.1",
|
||||
role: "infra",
|
||||
clusterCidr: ["10.42.0.0/16"],
|
||||
serviceCidr: ["10.43.0.0/16"],
|
||||
});
|
||||
expect(out).toContain('cluster-cidr: "10.42.0.0/16"');
|
||||
expect(out).not.toContain("node-ip");
|
||||
});
|
||||
});
|
||||
|
||||
// --- CNI Cleanup ---
|
||||
|
||||
@@ -161,7 +161,7 @@ EOF'
|
||||
CILIUM_CLI_VERSION=$(curl -s https://raw.githubusercontent.com/cilium/cilium-cli/main/stable.txt)
|
||||
curl -L --fail --silent "https://github.com/cilium/cilium-cli/releases/download/\${CILIUM_CLI_VERSION}/cilium-linux-amd64.tar.gz" | sudo tar xz -C /usr/local/bin
|
||||
DEFAULT_DEV=$(ip -4 route show default | awk '{print $5}' | head -1)
|
||||
sudo KUBECONFIG=/etc/rancher/k3s/k3s.yaml cilium install --set kubeProxyReplacement=true --set ipam.mode=kubernetes --set devices=$DEFAULT_DEV --set nodePort.directRoutingDevice=$DEFAULT_DEV
|
||||
sudo KUBECONFIG=/etc/rancher/k3s/k3s.yaml cilium install --set kubeProxyReplacement=true --set ipam.mode=cluster-pool --set ipam.operator.clusterPoolIPv4PodCIDRList='{10.42.0.0/16}' --set ipam.operator.clusterPoolIPv4MaskSize=24 --set devices=$DEFAULT_DEV --set nodePort.directRoutingDevice=$DEFAULT_DEV
|
||||
`.trim(), "cilium install", { keyPath: sshKeyPath, timeout: 120_000 });
|
||||
|
||||
log("Waiting for Cilium to be ready...");
|
||||
|
||||
4
labsim/.gitignore
vendored
4
labsim/.gitignore
vendored
@@ -2,3 +2,7 @@
|
||||
*.log
|
||||
labsim_matrix_lib.py
|
||||
__pycache__/
|
||||
|
||||
# Cluster-admin credentials for the rehearsal cluster, written by
|
||||
# `k8s-up.sh --kubeconfig`. Regenerate it rather than commit it.
|
||||
*.kubeconfig
|
||||
|
||||
176
labsim/README.md
176
labsim/README.md
@@ -143,6 +143,167 @@ unreserved MAC gets an unreserved address.
|
||||
- **http://localhost:9101/metrics** — `labsim_reachable{src,dst,proto}` and
|
||||
`labsim_rtt_ms{src,dst}`.
|
||||
|
||||
## WAN follows VRRP, and the PPPoE half of it
|
||||
|
||||
One consumer ISP account, two routers. The 10 gig line's lease is bound to a
|
||||
cloned MAC and the Vodafone line to a single credential, so neither can be live
|
||||
on both boxes: the WAN has to move with mastership.
|
||||
|
||||
The two halves use different control planes, and that asymmetry is the design:
|
||||
|
||||
| | plane | why |
|
||||
|---|---|---|
|
||||
| `bond0.53` | VyOS **config** (`disable`) | only config can move a MAC |
|
||||
| `pppoe0` | **systemd** unit gate | see below |
|
||||
|
||||
`set interfaces pppoe pppoe0 disable` cannot work as a resting state.
|
||||
`interfaces_pppoe.py` treats `disable` and `delete` identically and **unlinks
|
||||
`/etc/ppp/peers/pppoe0`** — which is pppd's own options file. The resting state
|
||||
therefore destroyed what the promotion path needed, and `ppp@pppoe0`
|
||||
restart-looped against it (47 restarts, zero sessions at the AC). It also makes
|
||||
op-mode `connect interface pppoe0` refuse, and puts every failover behind a
|
||||
priority-322 commit where one unrelated bad node fails the lot.
|
||||
|
||||
So `pppoe0` is configured identically and **enabled on both**, and dialling is
|
||||
gated by a drop-in:
|
||||
|
||||
```ini
|
||||
ConditionPathExists=/run/vrrp-wan/may-dial
|
||||
ConditionPathExists=/etc/ppp/peers/pppoe0
|
||||
```
|
||||
|
||||
`/run` is tmpfs, so the gate is shut at boot. That matters more than it looks:
|
||||
with the node enabled, `interfaces_pppoe.py` restarts ppp on **every** commit
|
||||
touching the pppoe subtree when the daemon isn't running — so the backup
|
||||
actively tries to dial whenever anything commits. The gate is the only thing
|
||||
making that a no-op, which is why the reconciler refuses to bless a box whose
|
||||
drop-in is missing (`/etc` is per-image; a VyOS upgrade would silently remove
|
||||
the protection).
|
||||
|
||||
`may-dial` is a **lease**, not a flag: `ConditionPathExists` is evaluated at
|
||||
start only, so it can prevent a dial but never revoke one. `vrrp-wan-reconcile`
|
||||
renews it every 30s; `vrrp-wan-guard` runs every 5s and only ever revokes.
|
||||
|
||||
### Testing it
|
||||
|
||||
```sh
|
||||
./labsim-pppoe-ha-test.sh --all
|
||||
```
|
||||
|
||||
The verdict is what the **routers** and the **access concentrator** did, never
|
||||
what a client happened to get — and the harness refuses to run at all while a
|
||||
router still has a default route via `eth2`, because the libvirt-NAT scaffold
|
||||
answers connectivity checks that the WAN under test would have failed. The
|
||||
invariant it enforces throughout: *the AC never reports two `simdsl` sessions,
|
||||
and no two routers ever hold `pppoe0`.*
|
||||
|
||||
Two failures the harness itself produced, both worth remembering: waiting for
|
||||
"exactly one holder" returns instantly during a handover (it was already true),
|
||||
and judging connectivity on a single ping 20s after a link drop reports an
|
||||
outage that has already healed. Ask **who** holds it, and poll.
|
||||
|
||||
Evidence in `wan-failover-evidence/`.
|
||||
|
||||
## Routing: BGP, dual WAN, and the ISP VMs
|
||||
|
||||
`sim-ha-config.py` covers the LAN side of the routers. `sim-net-config.py`
|
||||
covers everything that makes this a rehearsal for production *routing*:
|
||||
|
||||
| role | VM | what it generates |
|
||||
|---|---|---|
|
||||
| `primary` | `labsim-vyos` | BGP + dual WAN + health-checked failover |
|
||||
| `secondary` | `labsim-vyos2` | BGP only |
|
||||
| `isp-dhcp` | `labsim-isp-dhcp` | 10gig-equivalent ISP on VLAN 53 |
|
||||
| `isp-pppoe` | `labsim-isp-pppoe` | Vodafone-equivalent PPPoE ISP on VLAN 51 |
|
||||
|
||||
Both ISP VMs are VyOS with two NICs: one on the OVS trunk facing the sim
|
||||
router, one on libvirt's `default` network, NATing customers to the real
|
||||
internet. They use RFC 5737 documentation ranges (`203.0.113.0/24`,
|
||||
`198.51.100.0/24`) so a leaked sim route cannot blackhole anything real.
|
||||
|
||||
```sh
|
||||
./sim-net-apply.sh check # VM state vs what the code says — run this first
|
||||
./sim-net-apply.sh apply # push generated config over the serial console
|
||||
```
|
||||
|
||||
`check` is the important one. All of this previously existed only as running
|
||||
state, applied by hand over SSH; rebuilding a VM lost it, and nothing recorded
|
||||
why any of it was shaped the way it was.
|
||||
|
||||
### Known gaps vs production
|
||||
|
||||
- **WAN is on the primary router only.** Production has WAN on both. Two PPPoE
|
||||
clients sharing one credential against a single access concentrator is a
|
||||
failure mode production does not have, so the sim does not model it. VRRP and
|
||||
conntrack failover are still exercised.
|
||||
- **ISP VM interface names are not stable across a rebuild** — `isp-dhcp` came
|
||||
up as `eth0`/`eth1` and `isp-pppoe` as `eth2`/`eth3` from identical XML.
|
||||
Check `show interfaces` and pass `--wan-if` / `--uplink-if` rather than
|
||||
trusting the defaults.
|
||||
- **`eth2` on the primary router** is a libvirt-NAT uplink predating the ISP
|
||||
VMs: a third default route with no production equivalent that masks real WAN
|
||||
failures during a failover test. `--drop-scaffold` removes it.
|
||||
- **Committing on `isp-pppoe` drops the router's PPPoE session**, and the
|
||||
client does not redial promptly. After any change there, check `pppoe0` on
|
||||
the router and `sudo systemctl restart ppp@pppoe0` if it is missing.
|
||||
|
||||
## The trunk carries every VLAN tagged, including Management
|
||||
|
||||
There is deliberately **no native/untagged VLAN** on the trunks to the routers,
|
||||
and Management lives on `bond0.1`, not on the bare `bond0`.
|
||||
|
||||
A native VLAN is what puts a subnet on the bond **parent** while every other
|
||||
VLAN sits on a sub-interface of it. With `dhcp-socket-type: raw`, kea receives
|
||||
each tagged frame *twice* — once on the sub-interface and once on the parent —
|
||||
and answers from the parent's pool as well (ISC Kea
|
||||
[#1117](https://gitlab.isc.org/isc-projects/kea/-/issues/1117)). A client on
|
||||
VLAN 3 gets two OFFERs and keeps whichever arrives first:
|
||||
|
||||
```
|
||||
bond0.3 : 172.31.3.252 → 172.31.3.11 correct
|
||||
bond0 : 172.31.1.252 → 172.31.1.8 UNTAGGED, Management pool, wrong
|
||||
```
|
||||
|
||||
`./labsim-vlan-leak-test.sh` makes one client on a tagged VLAN send a DISCOVER
|
||||
and captures on the parent and the sub-interface at once. The verdict is how
|
||||
many OFFERs the **server** emitted and from which subnets — deliberately not
|
||||
"did the client get the right address", because a client picking correctly is
|
||||
exactly how this hid. Both orderings were observed across runs, so a passing
|
||||
client proves nothing.
|
||||
|
||||
```sh
|
||||
./labsim-vlan-leak-test.sh --vlan 3 # PASS on the current shape
|
||||
LABSIM_NATIVE_VLAN=1 ./router-up.sh # restore the old shape...
|
||||
./labsim-vlan-leak-test.sh --vlan 3 # ...and it FAILs again
|
||||
```
|
||||
|
||||
Three things this cost, all of which apply to production:
|
||||
|
||||
- **Kea must be restarted after the address moves.** VyOS does not restart it
|
||||
for an interface address change, so it keeps a raw socket bound with the old
|
||||
address and the bug survives the fix. In the sim kea had been running since
|
||||
16 Aug; the first post-fix test failed for this reason alone and looked like
|
||||
the fix simply not working.
|
||||
- **The firewall interface-group must move too.** `interface-group LAN` named
|
||||
the bare `bond0`; with a default-deny ruleset, moving the address without
|
||||
moving the group drops every management session and all VLAN 1 routing.
|
||||
- **Duplicate delivery does not stop.** #1117 says only that there is no longer
|
||||
a subnet on the parent to match, and that is exactly what happens: two replies
|
||||
per DISCOVER, both now from the correct pool. Harmless, but do not read a
|
||||
duplicate as a failure.
|
||||
|
||||
### Tagged and untagged Management coexist
|
||||
|
||||
Verified directly, and it is what makes the production cutover a rolling change
|
||||
rather than an outage: with the primary still untagged on `bond0` and the
|
||||
secondary already tagged on `bond0.1`, both routers were reachable, the VIP
|
||||
stayed up and a VLAN 1 client kept its gateway. One VLAN is one broadcast
|
||||
domain regardless of how each port tags it, so the two firewalls can be
|
||||
converted one at a time. See `migration/MANAGEMENT-VLAN-TAGGED.md`.
|
||||
|
||||
`./vlan1-move-monitor.sh` logs VIP/router liveness once a second during the
|
||||
change, because VRRP reconverges and leaves no trace of who held the VIP.
|
||||
|
||||
## Notes for whoever extends this
|
||||
|
||||
Things that cost time the first time round, all verified on this image:
|
||||
@@ -163,7 +324,14 @@ Things that cost time the first time round, all verified on this image:
|
||||
|
||||
## Not modelled (yet)
|
||||
|
||||
VLANs are separate L2 segments rather than one 802.1Q trunk, so this exercises
|
||||
inter-VLAN routing but not a `bond0.<vif>` trunk config specifically. A router
|
||||
VM would attach one NIC per VLAN. Adding a tagged-trunk variant is the obvious
|
||||
next step if the bond/vif config itself needs testing.
|
||||
- **The secondary's bond was fiction until 2026-09-02.** `ovs_bond_router`'s
|
||||
"already bonded, nothing to do" check compared only the trunk VLAN list, not
|
||||
the membership. Restarting a VM recreates its taps under new names, so the
|
||||
bond sat there holding two interfaces that no longer existed while the router's
|
||||
real taps ran in the bridge as two *independent* ports — no LACP, and carrying
|
||||
libvirt's own portgroup VLAN config rather than the bond's. It reconciles
|
||||
membership now, but the lesson generalises: a sim that reports success is not
|
||||
the same as a sim that models the thing.
|
||||
- **`labsim-vyos` has a third NIC** on libvirt's `default` network (the scaffold
|
||||
uplink, see `--drop-scaffold`). The tap count is filtered to `$OVS_NET` for
|
||||
that reason; an unfiltered count is 3 and silently skipped the primary's bond.
|
||||
|
||||
144
labsim/cilium-ipam-switch.sh
Executable file
144
labsim/cilium-ipam-switch.sh
Executable file
@@ -0,0 +1,144 @@
|
||||
#!/usr/bin/env bash
|
||||
# Procedure around a Cilium IPAM mode change. Works against any cluster, so the
|
||||
# rehearsal in labsim and the real thing in production run the SAME steps.
|
||||
#
|
||||
# It deliberately does NOT change the mode itself. In labsim that is `helm
|
||||
# upgrade`; in production Pulumi owns the release and a script racing it would
|
||||
# just reintroduce drift. What this owns is everything around the apply -- the
|
||||
# evidence, the deadlock, and the verdict.
|
||||
#
|
||||
# ./cilium-ipam-switch.sh preflight record what the cluster looks like now
|
||||
# ./cilium-ipam-switch.sh unstick break the agent-not-ready taint deadlock
|
||||
# ./cilium-ipam-switch.sh verify compare against preflight, report renumbering
|
||||
#
|
||||
# KUBECONFIG=... ./cilium-ipam-switch.sh preflight
|
||||
#
|
||||
# Whether a recycle is needed is CONDITIONAL, and `verify` is what decides it.
|
||||
#
|
||||
# The operator does not preserve which node held which /24 -- it adopts whatever
|
||||
# CiliumNode.spec.ipam.podCIDRs already says. So:
|
||||
#
|
||||
# * If CiliumNode already agrees with node.spec.podCIDRs on every node -- which
|
||||
# is the case for any cluster that has only ever run ipam=kubernetes, because
|
||||
# the operator syncs one from the other -- the pool adopts the existing
|
||||
# allocation, no node is renumbered, and NO pod recycle is needed. Verified
|
||||
# on the 3-node labsim cluster: CIDRs unchanged, nothing stranded, the only
|
||||
# blip was the cilium DaemonSet restarting itself.
|
||||
#
|
||||
# * If the two sources DISAGREE, nodes can swap /24s. Their running pods keep
|
||||
# addresses that no longer fall inside the node's range, every other node
|
||||
# routes that prefix to the wrong node, and those pods go unreachable
|
||||
# cross-node while still showing Running. Then a full recycle is mandatory.
|
||||
#
|
||||
# Do not skip `verify` on the assumption of the good case. Run it and read it.
|
||||
set -uo pipefail
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
STATE="${STATE:-$SCRIPT_DIR/.ipam-switch-state}"
|
||||
K="kubectl"
|
||||
|
||||
say() { printf '\033[0;36m[ipam]\033[0m %s\n' "$*"; }
|
||||
warn() { printf '\033[1;33m[ipam]\033[0m %s\n' "$*" >&2; }
|
||||
|
||||
snapshot() {
|
||||
echo "## nodes"
|
||||
$K get nodes -o jsonpath='{range .items[*]}{.metadata.name}{"\t"}{.spec.podCIDRs}{"\n"}{end}' 2>/dev/null
|
||||
echo "## ciliumnodes"
|
||||
$K get ciliumnode -o jsonpath='{range .items[*]}{.metadata.name}{"\t"}{.spec.ipam.podCIDRs}{"\n"}{end}' 2>/dev/null
|
||||
echo "## pods"
|
||||
$K get pods -A -o jsonpath='{range .items[*]}{.metadata.namespace}/{.metadata.name}{"\t"}{.status.podIP}{"\n"}{end}' 2>/dev/null \
|
||||
| grep -vP '\t$' | sort
|
||||
echo "## ipam"
|
||||
$K -n kube-system get cm cilium-config -o jsonpath='{.data.ipam}' 2>/dev/null; echo
|
||||
}
|
||||
|
||||
cmd_preflight() {
|
||||
mkdir -p "$STATE"
|
||||
snapshot > "$STATE/before.txt"
|
||||
say "recorded $(grep -c . "$STATE/before.txt") lines -> $STATE/before.txt"
|
||||
say "mode now: $(sed -n '/^## ipam/,$p' "$STATE/before.txt" | tail -1)"
|
||||
# The pod inventory is the rollback reference: if the switch renumbers, this
|
||||
# is the only record of what an address USED to be.
|
||||
say "pods on the pod network: $(sed -n '/^## pods/,/^## ipam/p' "$STATE/before.txt" | grep -c '10\.')"
|
||||
}
|
||||
|
||||
# The deadlock, in one place because it WILL happen and doing it by hand under
|
||||
# time pressure is how the wrong node gets untainted:
|
||||
# agent has no pod CIDR -> agent not ready -> node keeps
|
||||
# node.cilium.io/agent-not-ready:NoSchedule -> the operator that would assign
|
||||
# the CIDR cannot schedule -> agent still has no pod CIDR.
|
||||
# Removing the taint is safe: it exists to keep normal workloads off a node
|
||||
# without working networking, and the operator is precisely the thing that fixes
|
||||
# that. Kubernetes re-adds it on the next agent restart.
|
||||
cmd_unstick() {
|
||||
local stuck=0
|
||||
for n in $($K get nodes -o name 2>/dev/null); do
|
||||
$K get "$n" -o jsonpath='{.spec.taints[*].key}' 2>/dev/null | grep -q 'agent-not-ready' || continue
|
||||
warn "${n#node/} carries agent-not-ready; removing so the operator can schedule"
|
||||
$K taint "$n" node.cilium.io/agent-not-ready- >/dev/null 2>&1 && stuck=$((stuck+1))
|
||||
done
|
||||
[ "$stuck" -eq 0 ] && say "no node was stuck" || say "cleared $stuck node(s)"
|
||||
local pend
|
||||
pend="$($K -n kube-system get pods -l io.cilium/app=operator --no-headers 2>/dev/null | grep -c Pending)"
|
||||
[ "${pend:-0}" -gt 0 ] && warn "$pend operator pod(s) still Pending — check tolerations, not just taints"
|
||||
return 0
|
||||
}
|
||||
|
||||
cmd_verify() {
|
||||
[ -f "$STATE/before.txt" ] || { warn "no preflight snapshot; nothing to compare"; return 1; }
|
||||
snapshot > "$STATE/after.txt"
|
||||
echo
|
||||
say "mode: $(sed -n '/^## ipam/,$p' "$STATE/before.txt" | tail -1) -> $(sed -n '/^## ipam/,$p' "$STATE/after.txt" | tail -1)"
|
||||
|
||||
# The question that decides the size of the maintenance window: did per-node
|
||||
# CIDRs survive, or was every node renumbered (and every pod with it)?
|
||||
local moved=0
|
||||
while IFS=$'\t' read -r node cidr; do
|
||||
[ -z "${node:-}" ] && continue
|
||||
local now; now="$(sed -n '/^## ciliumnodes/,/^## pods/p' "$STATE/after.txt" | awk -F'\t' -v n="$node" '$1==n{print $2}')"
|
||||
if [ -n "$now" ] && [ "$now" != "$cidr" ]; then
|
||||
printf ' %-16s %s -> %s\n' "$node" "$cidr" "$now"; moved=$((moved+1))
|
||||
fi
|
||||
done < <(sed -n '/^## ciliumnodes/,/^## pods/p' "$STATE/before.txt" | grep -P '\t')
|
||||
if [ "$moved" -eq 0 ]; then
|
||||
say "per-node CIDRs UNCHANGED — the pool adopted the existing allocation"
|
||||
else
|
||||
warn "$moved node(s) renumbered — every pod on them must be recycled"
|
||||
fi
|
||||
|
||||
local before after same
|
||||
before="$(sed -n '/^## pods/,/^## ipam/p' "$STATE/before.txt" | grep -P '\t10\.' | wc -l)"
|
||||
after="$(sed -n '/^## pods/,/^## ipam/p' "$STATE/after.txt" | grep -P '\t10\.' | wc -l)"
|
||||
same="$(comm -12 <(sed -n '/^## pods/,/^## ipam/p' "$STATE/before.txt" | grep -P '\t10\.' | sort) \
|
||||
<(sed -n '/^## pods/,/^## ipam/p' "$STATE/after.txt" | grep -P '\t10\.' | sort) | wc -l)"
|
||||
say "pods: $before before, $after after, $same kept the SAME address"
|
||||
# Keeping the address is NOT the good outcome. If a node's CIDR moved, its
|
||||
# existing pods keep IPs that no longer fall inside it, every other node routes
|
||||
# that prefix to the WRONG node, and those pods go unreachable cross-node while
|
||||
# looking perfectly healthy. Observed in labsim: two nodes swapped CIDRs and
|
||||
# cross-node ping to their pods dropped 100%, with every pod still Running.
|
||||
# This is the check that decides whether a recycle is optional or mandatory.
|
||||
local stranded=0
|
||||
while read -r ns name ip node; do
|
||||
[ -z "${node:-}" ] && continue
|
||||
local cidr; cidr="$($K get ciliumnode "$node" -o jsonpath='{.spec.ipam.podCIDRs[0]}' 2>/dev/null)"
|
||||
[ -z "$cidr" ] && continue
|
||||
case "$ip" in
|
||||
"${cidr%.*/*}".*) ;;
|
||||
*) printf ' STRANDED %-40s %-15s on %s (now %s)\n' "$ns/$name" "$ip" "$node" "$cidr"; stranded=$((stranded+1)) ;;
|
||||
esac
|
||||
done < <($K get pods -A -o jsonpath='{range .items[?(@.status.podIP)]}{.metadata.namespace}{" "}{.metadata.name}{" "}{.status.podIP}{" "}{.spec.nodeName}{"\n"}{end}' 2>/dev/null | grep -E ' 10\.')
|
||||
if [ "$stranded" -gt 0 ]; then
|
||||
warn "$stranded pod(s) sit OUTSIDE their node CIDR — unreachable cross-node until recycled"
|
||||
warn "recycle: for ns in $(kubectl get ns -o name | cut -d/ -f2); do kubectl -n $ns rollout restart deploy,ds,sts 2>/dev/null; done"
|
||||
else
|
||||
say "every pod is inside its node CIDR — no recycle needed"
|
||||
fi
|
||||
say "not-Running pods: $($K get pods -A --no-headers 2>/dev/null | grep -vcE 'Running|Completed')"
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
preflight) cmd_preflight ;;
|
||||
unstick) cmd_unstick ;;
|
||||
verify) cmd_verify ;;
|
||||
*) sed -n '2,16p' "$0"; exit 1 ;;
|
||||
esac
|
||||
@@ -22,6 +22,12 @@ def main() -> int:
|
||||
ap.add_argument("--config", required=True)
|
||||
ap.add_argument("--user", default="vyos")
|
||||
ap.add_argument("--password", default="vyos")
|
||||
# `save` writes config.boot. For WAN work that is dangerous: the resting
|
||||
# state must stay `vif 53 disable` on both routers, and saving while a box
|
||||
# is master persists the ENABLED state -- so a reboot would have it claim
|
||||
# the cloned MAC. Observed in labsim on 2026-09-02.
|
||||
ap.add_argument("--no-save", action="store_true",
|
||||
help="commit without saving (leave config.boot untouched)")
|
||||
args = ap.parse_args()
|
||||
|
||||
cmds = [l.rstrip() for l in open(args.config)
|
||||
@@ -83,8 +89,9 @@ def main() -> int:
|
||||
c.sendline("commit")
|
||||
c.expect(r"# ", timeout=300)
|
||||
commit_out = c.before or ""
|
||||
c.sendline("save")
|
||||
c.expect(r"# ", timeout=120)
|
||||
if not args.no_save:
|
||||
c.sendline("save")
|
||||
c.expect(r"# ", timeout=120)
|
||||
# Accept either prompt on the way out. Insisting on `$ ` here hangs against
|
||||
# a healthy box -- and worse, leaves the console parked in config mode, so
|
||||
# the NEXT run finds a `# ` it was not expecting either. One strict expect
|
||||
|
||||
318
labsim/dualstack-lab.sh
Executable file
318
labsim/dualstack-lab.sh
Executable file
@@ -0,0 +1,318 @@
|
||||
#!/usr/bin/env bash
|
||||
# Differential study: what ACTUALLY differs between a k3s cluster born
|
||||
# dual-stack and one converted in place?
|
||||
#
|
||||
# k3s says dual-stack "cannot be enabled on an existing cluster". The stated
|
||||
# reason is narrow -- nodes get Pod CIDRs only at join and the Kubernetes IPAM
|
||||
# controller will not hand out a new IPv6 CIDR later -- and it does not obviously
|
||||
# apply to a cluster where Cilium owns IPAM. Rather than argue from docs, build
|
||||
# both shapes and diff them.
|
||||
#
|
||||
# ./dualstack-lab.sh up v4 single-node k3s, IPv4 only (.21)
|
||||
# ./dualstack-lab.sh up dual single-node k3s, dual-stack (.22)
|
||||
# ./dualstack-lab.sh pristine v4 reflink copy of v4's disk, so the upgrade
|
||||
# attempt can be rolled back and retried
|
||||
# ./dualstack-lab.sh restore v4 put that copy back
|
||||
# ./dualstack-lab.sh collect <n> normalized state dump -> evidence/<n>/
|
||||
# ./dualstack-lab.sh compare a b semantic diff of two collections
|
||||
# ./dualstack-lab.sh virtdiff a b whole-filesystem diff, offline (libguestfs)
|
||||
# ./dualstack-lab.sh down [name]
|
||||
#
|
||||
# The comparison that matters is `compare dual upgraded`: everything it prints
|
||||
# is a way the converted cluster failed to reach the shape of a native one.
|
||||
#
|
||||
# Single node on purpose. Dual-stack is decided by server flags and CNI config,
|
||||
# both of which a one-node cluster exercises fully, and it rebuilds in minutes.
|
||||
# Node-rejoin behaviour needs the 3-node cluster and is a separate question.
|
||||
set -euo pipefail
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
source "$SCRIPT_DIR/lib.sh"
|
||||
source "$SCRIPT_DIR/ovs.sh"
|
||||
|
||||
K8S_VLAN="${K8S_VLAN:-2}"
|
||||
DS_PREFIX="${DS_PREFIX:-172.31.2}"
|
||||
MEM="${MEM:-4096}"; CPUS="${CPUS:-2}"; DISK_GB="${DISK_GB:-12}"
|
||||
TOKEN="${TOKEN:-labsim-ds-token}"
|
||||
CILIUM_VERSION="${CILIUM_VERSION:-1.19.1}" # same as production
|
||||
DEB_BASE="${DEB_BASE:-$IMG_DIR/debian-13-genericcloud-amd64.qcow2}"
|
||||
EVIDENCE="$SCRIPT_DIR/dualstack-evidence"
|
||||
|
||||
# Pod/Service ranges. IPv4 halves are k3s's own defaults, so the v4-only build is
|
||||
# a stock cluster and the diff is not polluted by gratuitous differences.
|
||||
# IPv6 halves are ULA: this cluster never routes off-box, and using the real /48
|
||||
# here would put lab addresses into a prefix that production also announces.
|
||||
V4_CLUSTER="10.42.0.0/16"; V4_SERVICE="10.43.0.0/16"
|
||||
V6_CLUSTER="${V6_CLUSTER:-fd00:42::/56}"
|
||||
V6_SERVICE="${V6_SERVICE:-fd00:43::/112}" # /112 -- apiserver caps v6 service ranges
|
||||
V6_PREFIX="${V6_PREFIX:-fd00:2}" # node addresses: fd00:2::<octet>
|
||||
|
||||
vm_name() { echo "labsim-ds-$1"; }
|
||||
vm_ip() { case "$1" in v4) echo "$DS_PREFIX.21";; dual) echo "$DS_PREFIX.22";; *) die "unknown build '$1'";; esac; }
|
||||
vm_ip6() { case "$1" in v4) echo "$V6_PREFIX::21";; dual) echo "$V6_PREFIX::22";; *) die "unknown build '$1'";; esac; }
|
||||
disk_of() { echo "$IMG_DIR/$(vm_name "$1").qcow2"; }
|
||||
|
||||
ssh_vm() { local ip="$1"; shift; ssh -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
|
||||
-o LogLevel=ERROR -o ConnectTimeout=8 -o BatchMode=yes "debian@$ip" "$@"; }
|
||||
|
||||
# --- seed -----------------------------------------------------------------
|
||||
build_seed() {
|
||||
local iso="$1" vm="$2" mode="$3" pubkey="$4"
|
||||
local ip ip6 tmp; ip="$(vm_ip "$mode")"; ip6="$(vm_ip6 "$mode")"; tmp="$(mktemp -d)"
|
||||
|
||||
echo "instance-id: $vm" > "$tmp/meta-data"
|
||||
# Static v6 on both builds. The v4-only cluster still gets an IPv6 ADDRESS --
|
||||
# only its Kubernetes config is v4-only. Otherwise the diff would be dominated
|
||||
# by host addressing rather than by what Kubernetes did differently.
|
||||
cat > "$tmp/network-config" <<EOF
|
||||
version: 2
|
||||
ethernets:
|
||||
enp1s0:
|
||||
addresses: [${ip}/24, ${ip6}/64]
|
||||
routes:
|
||||
- to: default
|
||||
via: ${DS_PREFIX}.1
|
||||
nameservers:
|
||||
addresses: [8.8.8.8, 1.1.1.1]
|
||||
EOF
|
||||
|
||||
local exec_args="server --flannel-backend=none --disable-network-policy --disable=servicelb --disable=traefik --tls-san=$ip --cluster-init"
|
||||
if [ "$mode" = dual ]; then
|
||||
exec_args="$exec_args --cluster-cidr=${V4_CLUSTER},${V6_CLUSTER} --service-cidr=${V4_SERVICE},${V6_SERVICE} --node-ip=${ip},${ip6}"
|
||||
else
|
||||
exec_args="$exec_args --node-ip=${ip}"
|
||||
fi
|
||||
|
||||
cat > "$tmp/user-data" <<EOF
|
||||
#cloud-config
|
||||
hostname: $vm
|
||||
fqdn: $vm
|
||||
users:
|
||||
- name: debian
|
||||
groups: [sudo]
|
||||
shell: /bin/bash
|
||||
sudo: ["ALL=(ALL) NOPASSWD:ALL"]
|
||||
lock_passwd: false
|
||||
plain_text_passwd: labsim
|
||||
ssh_authorized_keys: [$pubkey]
|
||||
ssh_pwauth: true
|
||||
disable_root: false
|
||||
package_update: true
|
||||
packages: [curl, jq, iproute2, nftables]
|
||||
write_files:
|
||||
- path: /etc/modules-load.d/cilium.conf
|
||||
content: |
|
||||
br_netfilter
|
||||
- path: /etc/dualstack-lab-mode
|
||||
content: |
|
||||
$mode
|
||||
runcmd:
|
||||
- modprobe br_netfilter || true
|
||||
- |
|
||||
curl -sfL https://get.k3s.io | INSTALL_K3S_EXEC="$exec_args" K3S_TOKEN="$TOKEN" sh -
|
||||
- |
|
||||
# Cilium via helm, matching the production version. IPAM stays 'kubernetes'
|
||||
# in BOTH builds on purpose: that is what production runs, and it is the
|
||||
# mode the k3s objection is actually about. If the converted cluster needs
|
||||
# cluster-pool to work, the diff should be what tells us so.
|
||||
curl -sfL https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash || true
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
helm repo add cilium https://helm.cilium.io >/dev/null 2>&1 || true
|
||||
helm repo update >/dev/null 2>&1 || true
|
||||
for i in \$(seq 1 60); do kubectl get nodes >/dev/null 2>&1 && break; sleep 5; done
|
||||
if [ "$mode" = dual ]; then
|
||||
helm install cilium cilium/cilium --version $CILIUM_VERSION -n kube-system \\
|
||||
--set kubeProxyReplacement=false --set ipam.mode=kubernetes \\
|
||||
--set ipv4.enabled=true --set ipv6.enabled=true \\
|
||||
--set k8sServiceHost=$ip --set k8sServicePort=6443 || true
|
||||
else
|
||||
helm install cilium cilium/cilium --version $CILIUM_VERSION -n kube-system \\
|
||||
--set kubeProxyReplacement=false --set ipam.mode=kubernetes \\
|
||||
--set ipv4.enabled=true --set ipv6.enabled=false \\
|
||||
--set k8sServiceHost=$ip --set k8sServicePort=6443 || true
|
||||
fi
|
||||
touch /etc/dualstack-lab-ready
|
||||
EOF
|
||||
sudo mkdir -p "$(dirname "$iso")"
|
||||
sudo genisoimage -quiet -output "$iso" -volid cidata -joliet -rock \
|
||||
"$tmp/user-data" "$tmp/meta-data" "$tmp/network-config"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
cmd_up() {
|
||||
local mode="${1:?usage: up <v4|dual>}"
|
||||
local vm ip disk seed pubkey
|
||||
vm="$(vm_name "$mode")"; ip="$(vm_ip "$mode")"; disk="$(disk_of "$mode")"
|
||||
seed="$IMG_DIR/${vm}-seed.iso"; pubkey="$(find_ssh_pubkey)"
|
||||
|
||||
[ -f "$DEB_BASE" ] || die "base image missing: $DEB_BASE (run ./k8s-up.sh once)"
|
||||
if virsh_q dominfo "$vm" >/dev/null 2>&1; then
|
||||
log "$vm exists — starting if stopped"
|
||||
[ "$(virsh_q domstate "$vm" | head -1)" = "running" ] || virsh_q start "$vm" >/dev/null
|
||||
return
|
||||
fi
|
||||
selected_vlans; ovs_up
|
||||
log "creating $vm ($mode) at $ip / $(vm_ip6 "$mode")"
|
||||
sudo qemu-img create -q -f qcow2 -F qcow2 -b "$DEB_BASE" "$disk" "${DISK_GB}G" >/dev/null
|
||||
build_seed "$seed" "$vm" "$mode" "$pubkey"
|
||||
sudo virt-install --connect "$LIBVIRT_URI" --name "$vm" \
|
||||
--memory "$MEM" --vcpus "$CPUS" \
|
||||
--disk "path=$disk,format=qcow2,bus=virtio" \
|
||||
--disk "path=$seed,device=cdrom" \
|
||||
--network "network=$OVS_NET,portgroup=vlan${K8S_VLAN},model=virtio" \
|
||||
--os-variant debian12 --graphics none --noautoconsole --import >/dev/null
|
||||
log "installing in background; watch: ssh debian@$ip 'ls /etc/dualstack-lab-ready'"
|
||||
}
|
||||
|
||||
# --- pristine copy / restore ---------------------------------------------
|
||||
# reflink so the copy is instant and independent on btrfs/xfs. A qcow2 backing
|
||||
# chain would be cheaper still but makes the parent read-only in practice: boot
|
||||
# the parent again and every child silently corrupts.
|
||||
cmd_pristine() {
|
||||
local mode="${1:?usage: pristine <v4|dual>}" vm disk
|
||||
vm="$(vm_name "$mode")"; disk="$(disk_of "$mode")"
|
||||
[ "$(virsh_q domstate "$vm" 2>/dev/null | head -1)" = "running" ] && \
|
||||
die "$vm is running — shut it down first (virsh shutdown $vm), a copy of a live disk is not consistent"
|
||||
sudo cp --reflink=auto "$disk" "${disk}.pristine"
|
||||
log "pristine copy: ${disk}.pristine"
|
||||
}
|
||||
cmd_restore() {
|
||||
local mode="${1:?usage: restore <v4|dual>}" vm disk
|
||||
vm="$(vm_name "$mode")"; disk="$(disk_of "$mode")"
|
||||
[ -f "${disk}.pristine" ] || die "no pristine copy for $mode"
|
||||
[ "$(virsh_q domstate "$vm" 2>/dev/null | head -1)" = "running" ] && \
|
||||
die "$vm is running — shut it down first"
|
||||
sudo cp --reflink=auto "${disk}.pristine" "$disk"
|
||||
log "restored $mode from pristine"
|
||||
}
|
||||
|
||||
|
||||
# --- the experiment ------------------------------------------------------
|
||||
# Convert the IPv4-only cluster in place, mirroring the flags the native build
|
||||
# was BORN with. Each step prints what the cluster did, because the interesting
|
||||
# output is which step refuses rather than whether the end state is pretty.
|
||||
cmd_upgrade() {
|
||||
local ip; ip="$(vm_ip v4)"; local ip6; ip6="$(vm_ip6 v4)"
|
||||
log "step 1/4: add dual CIDRs + dual node-ip to the k3s unit"
|
||||
# Done with python on the box, not nested sed: quoting a multi-line systemd
|
||||
# continuation through ssh -> sh -> sed produced a literal \\n in the unit, and
|
||||
# k3s then saw a dual cluster-cidr with a still-IPv4 service-cidr and refused
|
||||
# to start. All three flags go on one line -- systemd does not care, and there
|
||||
# is nothing left to escape.
|
||||
ssh_vm "$ip" "sudo python3 - <<'PYEOF'
|
||||
import re
|
||||
u = '/etc/systemd/system/k3s.service'
|
||||
s = open(u).read()
|
||||
old = \"'--node-ip=${ip}'\"
|
||||
new = \"'--cluster-cidr=${V4_CLUSTER},${V6_CLUSTER}' '--service-cidr=${V4_SERVICE},${V6_SERVICE}' '--node-ip=${ip},${ip6}'\"
|
||||
assert old in s, 'node-ip flag not found in unit'
|
||||
open(u,'w').write(s.replace(old, new))
|
||||
print(' unit rewritten')
|
||||
PYEOF
|
||||
sudo systemctl daemon-reload" || die "unit edit failed"
|
||||
ssh_vm "$ip" "grep -oE \"'--(cluster|service)-cidr=[^']*'|'--node-ip=[^']*'\" /etc/systemd/system/k3s.service | sed 's/^/ /'"
|
||||
|
||||
log "step 2/4: restart k3s and see whether it accepts the changed ranges"
|
||||
ssh_vm "$ip" "sudo systemctl restart k3s" || true
|
||||
for i in $(seq 1 40); do
|
||||
ssh_vm "$ip" "sudo k3s kubectl get --raw /readyz >/dev/null 2>&1" && break
|
||||
sleep 5
|
||||
done
|
||||
ssh_vm "$ip" "sudo journalctl -u k3s --since '2 min ago' --no-pager 2>/dev/null | grep -iE 'cidr|dual|ipv6|invalid|cannot|fail' | tail -12 | sed 's/^/ /'" || true
|
||||
|
||||
log "step 3/4: what the API says now"
|
||||
ssh_vm "$ip" "echo -n ' servicecidr: '; sudo k3s kubectl get servicecidr -o jsonpath='{.items[*].spec.cidrs}'; echo; \
|
||||
echo -n ' node podCIDRs: '; sudo k3s kubectl get node -o jsonpath='{.items[0].spec.podCIDRs}'; echo; \
|
||||
echo -n ' node addresses: '; sudo k3s kubectl get node -o jsonpath='{.items[0].status.addresses[*].address}'; echo" || true
|
||||
|
||||
log "step 4/4: turn on IPv6 in Cilium"
|
||||
ssh_vm "$ip" "export KUBECONFIG=/etc/rancher/k3s/k3s.yaml; sudo -E helm upgrade cilium cilium/cilium --version ${CILIUM_VERSION} -n kube-system --reuse-values --set ipv6.enabled=true >/dev/null 2>&1 && echo ' cilium upgraded' || echo ' cilium upgrade FAILED'" || true
|
||||
ssh_vm "$ip" "sudo k3s kubectl -n kube-system rollout restart ds/cilium >/dev/null 2>&1; sleep 20; sudo k3s kubectl -n kube-system get pods -l k8s-app=cilium --no-headers | sed 's/^/ /'" || true
|
||||
log "now: ./dualstack-lab.sh collect upgraded ${ip} && ./dualstack-lab.sh compare dual upgraded"
|
||||
}
|
||||
|
||||
# --- evidence collection --------------------------------------------------
|
||||
# Normalized on purpose. Two independently built clusters differ in certs,
|
||||
# tokens, UUIDs, timestamps and log lines; left raw, that noise buries the
|
||||
# handful of differences that actually mean something.
|
||||
cmd_collect() {
|
||||
local name="${1:?usage: collect <name> [ip]}"
|
||||
local ip="${2:-}"
|
||||
[ -n "$ip" ] || ip="$(vm_ip "$name" 2>/dev/null || true)"
|
||||
[ -n "$ip" ] || die "collect: give an ip for a non-standard name"
|
||||
local out="$EVIDENCE/$name"; mkdir -p "$out"
|
||||
log "collecting from $name ($ip) -> $out"
|
||||
|
||||
ssh_vm "$ip" 'sudo cat /etc/rancher/k3s/config.yaml 2>/dev/null; sudo systemctl cat k3s 2>/dev/null | grep -A30 ExecStart' \
|
||||
> "$out/k3s-config.txt" 2>/dev/null || true
|
||||
ssh_vm "$ip" 'sudo tr "\0" "\n" < /proc/$(pgrep -f "k3s server" | head -1)/cmdline | grep -v "^$"' \
|
||||
> "$out/k3s-cmdline.txt" 2>/dev/null || true
|
||||
ssh_vm "$ip" 'ip -o addr show | awk "{print \$2, \$3, \$4}"; echo ---; ip -4 route show; echo ---; ip -6 route show' \
|
||||
> "$out/host-net.txt" 2>/dev/null || true
|
||||
ssh_vm "$ip" 'sudo sysctl -a 2>/dev/null | grep -E "net\.ipv6\.conf\.(all|default)\.(forwarding|disable_ipv6)|net\.ipv4\.ip_forward"' \
|
||||
> "$out/sysctl.txt" 2>/dev/null || true
|
||||
|
||||
local K='sudo k3s kubectl'
|
||||
ssh_vm "$ip" "$K get servicecidr -o yaml" > "$out/servicecidr.yaml" 2>/dev/null || true
|
||||
ssh_vm "$ip" "$K get nodes -o yaml" > "$out/nodes.yaml.raw" 2>/dev/null || true
|
||||
ssh_vm "$ip" "$K get ciliumnodes -o yaml" > "$out/ciliumnodes.yaml.raw" 2>/dev/null || true
|
||||
ssh_vm "$ip" "$K -n kube-system get cm cilium-config -o yaml" > "$out/cilium-config.yaml.raw" 2>/dev/null || true
|
||||
ssh_vm "$ip" "$K get svc -A -o custom-columns=NS:.metadata.namespace,NAME:.metadata.name,FAMILYPOLICY:.spec.ipFamilyPolicy,FAMILIES:.spec.ipFamilies,IPS:.spec.clusterIPs" \
|
||||
> "$out/services.txt" 2>/dev/null || true
|
||||
ssh_vm "$ip" "$K get pods -A -o custom-columns=NS:.metadata.namespace,NAME:.metadata.name,IPS:.status.podIPs" \
|
||||
> "$out/podips.txt" 2>/dev/null || true
|
||||
|
||||
# Strip the things that differ every build regardless of configuration.
|
||||
for f in "$out"/*.raw; do
|
||||
[ -e "$f" ] || continue
|
||||
sed -E \
|
||||
-e 's/[0-9]{4}-[0-9]{2}-[0-9]{2}T[0-9:]+Z?/<TIME>/g' \
|
||||
-e 's/(uid|resourceVersion|creationTimestamp|generation|observedGeneration): .*/\1: <X>/' \
|
||||
-e 's/[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}/<UUID>/g' \
|
||||
-e 's/(LS0tLS1|[A-Za-z0-9+\/]{60,}=*)/<B64>/g' \
|
||||
"$f" > "${f%.raw}"
|
||||
rm -f "$f"
|
||||
done
|
||||
log "collected $(ls "$out" | wc -l) artefacts"
|
||||
}
|
||||
|
||||
cmd_compare() {
|
||||
local a="${1:?usage: compare <a> <b>}" b="${2:?}"
|
||||
[ -d "$EVIDENCE/$a" ] && [ -d "$EVIDENCE/$b" ] || die "collect both first"
|
||||
echo "### semantic diff: $a (<) vs $b (>)"
|
||||
diff -ru "$EVIDENCE/$a" "$EVIDENCE/$b" || true
|
||||
}
|
||||
|
||||
cmd_virtdiff() {
|
||||
local a="${1:?usage: virtdiff <a> <b>}" b="${2:?}"
|
||||
for m in "$a" "$b"; do
|
||||
[ "$(virsh_q domstate "$(vm_name "$m")" 2>/dev/null | head -1)" = "running" ] && \
|
||||
die "$(vm_name "$m") is running — virt-diff needs the disks quiescent"
|
||||
done
|
||||
log "whole-filesystem diff (slow); noise is expected — use it to find what the collector missed"
|
||||
sudo virt-diff -a "$(disk_of "$a")" -A "$(disk_of "$b")" \
|
||||
| grep -vE '/(var/log|tmp|run|proc|sys)/|\.log$|/var/lib/rancher/k3s/(server/(tls|cred|db)|agent)' || true
|
||||
}
|
||||
|
||||
cmd_down() {
|
||||
local only="${1:-}"
|
||||
for m in v4 dual; do
|
||||
[ -n "$only" ] && [ "$only" != "$m" ] && continue
|
||||
local vm; vm="$(vm_name "$m")"
|
||||
virsh_q dominfo "$vm" >/dev/null 2>&1 || continue
|
||||
log "removing $vm"
|
||||
virsh_q destroy "$vm" >/dev/null 2>&1 || true
|
||||
virsh_q undefine "$vm" --remove-all-storage >/dev/null 2>&1 || true
|
||||
done
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
up) shift; cmd_up "$@" ;;
|
||||
upgrade) shift; cmd_upgrade "$@" ;;
|
||||
pristine) shift; cmd_pristine "$@" ;;
|
||||
restore) shift; cmd_restore "$@" ;;
|
||||
collect) shift; cmd_collect "$@" ;;
|
||||
compare) shift; cmd_compare "$@" ;;
|
||||
virtdiff) shift; cmd_virtdiff "$@" ;;
|
||||
down) shift; cmd_down "$@" ;;
|
||||
*) sed -n '2,30p' "$0"; exit 1 ;;
|
||||
esac
|
||||
43
labsim/ipv6-ha-evidence/V0-baseline/state.txt
Normal file
43
labsim/ipv6-ha-evidence/V0-baseline/state.txt
Normal file
@@ -0,0 +1,43 @@
|
||||
=== 2026-09-06T14:25:25+01:00 ===
|
||||
--- HE endpoint ---
|
||||
HE address : 192.0.2.10/32
|
||||
he-sim : he-sim: ipv6/ip remote 198.51.100.137 local 192.0.2.10 ttl 64 6rd-prefix 2002::/16
|
||||
he-sim v6 : 2001:db8:1f1c:f6::1/64
|
||||
API : running
|
||||
API bound : nohost
|
||||
API calls : 4
|
||||
route back : 198.51.100.0/24 via 192.168.122.63 dev eth1
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup tun=DOWN radvd=inactive
|
||||
inactive
|
||||
tun0@NONE DOWN 203.0.113.108 <POINTOPOINT,NOARP>
|
||||
tun0: ipv6/ip remote 192.0.2.10 local 203.0.113.108 ttl 64 tos inherit 6rd-prefix 2002::/16
|
||||
inactive
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:29:01 apitest vrrp-wan[16369]: GUARD: lease stale (200s > 75s; is vrrp-wan-reconcile.timer running?) -- hanging up pppoe0
|
||||
Sep 05 23:29:27 apitest vrrp-wan[17398]: MASTER: dialling pppoe0
|
||||
Sep 05 23:29:57 apitest vrrp-wan[18610]: MASTER: dialling pppoe0
|
||||
-- Boot 6efc3c9ba47f455e9668454ee6c2fc37 --
|
||||
Sep 05 23:34:48 apitest vrrp-wan[6266]: MASTER: dialling pppoe0
|
||||
Sep 05 23:34:49 apitest vrrp-wan[6422]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:34:54 apitest vrrp-wan[7130]: bond0.53 enable commit took 5s
|
||||
-- Boot 866611afd7c542d4bf8c5117978dcd4a --
|
||||
Sep 06 13:21:54 apitest vrrp-wan[544215]: not MASTER: stopping radvd (deprecates the v6 gateway)
|
||||
Sep 06 13:21:54 apitest vrrp-wan[544221]: not MASTER: bringing tun0 down
|
||||
Sep 06 13:22:01 apitest he-tunnel-follow[544377]: role is now backup
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=25 dropin=yes role=master tun=UNKNOWN radvd=active
|
||||
tun0@NONE UNKNOWN 198.51.100.137 <POINTOPOINT,NOARP,UP,LOWER_UP>
|
||||
tun0: ipv6/ip remote 192.0.2.10 local 198.51.100.137 ttl 64 tos inherit 6rd-prefix 2002::/16
|
||||
default nhid 111 via 2001:db8:1f1c:f6::1 dev tun0 proto static metric 20 pref medium
|
||||
active
|
||||
Sep 06 09:29:21 vyos vrrp-wan[422441]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 06 09:29:25 vyos vrrp-wan[423516]: bond0.53 enable commit took 4s
|
||||
Sep 06 13:22:01 vyos he-tunnel-follow[593096]: role is now master
|
||||
Sep 06 13:22:01 vyos he-tunnel-follow[593109]: change seen (203.0.113.108 -> 198.51.100.137) but waiting for stability (1/2)
|
||||
Sep 06 13:23:01 vyos he-tunnel-follow[594799]: HE endpoint set to 198.51.100.137 (good 198.51.100.137)
|
||||
Sep 06 13:23:01 vyos he-tunnel-follow[594805]: moved tun0 to pppoe0: src 203.0.113.108 -> 198.51.100.137, mtu 1480 -> 1472
|
||||
Sep 06 13:24:02 vyos he-tunnel-follow[595700]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
Sep 06 13:24:26 vyos he-tunnel-follow[595886]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
Sep 06 13:24:26 vyos he-tunnel-follow[595904]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
Sep 06 13:25:01 vyos he-tunnel-follow[596311]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
23
labsim/ipv6-ha-evidence/mechanism-2026-09-06.txt
Normal file
23
labsim/ipv6-ha-evidence/mechanism-2026-09-06.txt
Normal file
@@ -0,0 +1,23 @@
|
||||
=== mechanism evidence, labsim, 2026-09-06T14:28:11+01:00 ===
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup tun=DOWN radvd=inactive
|
||||
inactive
|
||||
tun0@NONE DOWN 203.0.113.108 <POINTOPOINT,NOARP>
|
||||
inactive
|
||||
Sep 06 13:22:01 apitest he-tunnel-follow[544377]: role is now backup
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=26 dropin=yes role=master tun=UNKNOWN radvd=active
|
||||
tun0@NONE UNKNOWN 198.51.100.137 <POINTOPOINT,NOARP,UP,LOWER_UP>
|
||||
active
|
||||
Sep 06 13:25:01 vyos he-tunnel-follow[596311]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
Sep 06 13:26:01 vyos he-tunnel-follow[598274]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
Sep 06 13:27:01 vyos he-tunnel-follow[599393]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
Sep 06 13:28:01 vyos he-tunnel-follow[600596]: in sync: tun0 via pppoe0 src 198.51.100.137 mtu 1472
|
||||
--- HE endpoint ---
|
||||
HE address : 192.0.2.10/32
|
||||
he-sim : he-sim: ipv6/ip remote 198.51.100.137 local 192.0.2.10 ttl 64 6rd-prefix 2002::/16
|
||||
he-sim v6 : 2001:db8:1f1c:f6::1/64
|
||||
API : running
|
||||
API bound : nohost
|
||||
API calls : 5
|
||||
route back : 198.51.100.0/24 via 192.168.122.63 dev eth1
|
||||
266
labsim/k8s-up.sh
Executable file
266
labsim/k8s-up.sh
Executable file
@@ -0,0 +1,266 @@
|
||||
#!/bin/bash
|
||||
# A real Kubernetes cluster inside labsim, on the OVS fabric, for rehearsing
|
||||
# Cilium <-> VyOS BGP before it goes near the production routers.
|
||||
#
|
||||
# Why VMs and not k3d: the thing under test is eBGP between Cilium and VyOS
|
||||
# across the switch fabric — nodes on VLAN 2, peering with the router's bond0.2
|
||||
# leg, directly connected. k3d would put the nodes on a container bridge, which
|
||||
# is a different L2 path and would prove something else. (It also needs Docker;
|
||||
# this host has podman.)
|
||||
#
|
||||
# Why not the existing micro VMs: they are Alpine with 256 MB and 1 vCPU. k3s
|
||||
# plus Cilium needs an order of magnitude more, and a glibc distro with a stock
|
||||
# kernel that Cilium's eBPF probes are actually tested against.
|
||||
#
|
||||
# Three nodes, not two: ECMP is only meaningfully tested if a node can be
|
||||
# drained and MORE THAN ONE path survives.
|
||||
#
|
||||
# Layout (mirrors production's shape, not its addresses):
|
||||
# labsim-k8s1 172.31.2.11 k3s server
|
||||
# labsim-k8s2 172.31.2.12 agent
|
||||
# labsim-k8s3 172.31.2.13 agent
|
||||
# gateway 172.31.2.1 the VRRP VIP of the router pair under test
|
||||
# BGP peers 172.31.2.252 / .253 the routers' real per-box addresses
|
||||
#
|
||||
# Idempotent: re-running only creates what is missing.
|
||||
#
|
||||
# Usage:
|
||||
# ./k8s-up.sh create/start the cluster
|
||||
# ./k8s-up.sh --kubeconfig fetch kubeconfig to ./labsim-k8s.kubeconfig
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
source "$SCRIPT_DIR/lib.sh"
|
||||
source "$SCRIPT_DIR/ovs.sh"
|
||||
|
||||
# --- knobs ----------------------------------------------------------------
|
||||
K8S_VLAN="${K8S_VLAN:-2}"
|
||||
K8S_PREFIX="${K8S_PREFIX:-172.31.2}"
|
||||
K8S_NODES="${K8S_NODES:-3}"
|
||||
K8S_FIRST_OCTET="${K8S_FIRST_OCTET:-11}"
|
||||
# Dual-stack, mirroring production's shape (not its addresses). VLAN 2 v6 is a
|
||||
# ULA so nothing here can leak into the real HE /48; node fd00:2::1x matches the
|
||||
# BGP peers in sim-net-config.py (K8S_NODES_V6). cluster/service v6 are ULAs too;
|
||||
# the LB pool fd61:1e00::/64 is what Cilium advertises (SERVICE_CIDR_V6 there).
|
||||
K8S_PREFIX_V6="${K8S_PREFIX_V6:-fd00:2}" # nodes fd00:2::11/12/13, router ::252/::253
|
||||
K8S_ROUTER_V6="${K8S_ROUTER_V6:-fd00:2::252}" # v6 next-hop (primary router bond0.2)
|
||||
CLUSTER_CIDR_V6="${CLUSTER_CIDR_V6:-fd00:42::/56}"
|
||||
SERVICE_CIDR_V6="${SERVICE_CIDR_V6:-fd00:43::/112}"
|
||||
K8S_MEM="${K8S_MEM:-4096}" # MB — k3s + cilium + a workload
|
||||
K8S_CPUS="${K8S_CPUS:-2}"
|
||||
K8S_DISK_GB="${K8S_DISK_GB:-12}"
|
||||
K8S_TOKEN="${K8S_TOKEN:-labsim-k3s-token}"
|
||||
|
||||
# Debian rather than Alpine: glibc, a stock kernel, and cloud-init that applies
|
||||
# network-config properly (the Alpine base in this sim notably does not).
|
||||
DEB_URL="${DEB_URL:-https://cloud.debian.org/images/cloud/trixie/latest/debian-13-genericcloud-amd64.qcow2}"
|
||||
DEB_BASE="${DEB_BASE:-$IMG_DIR/debian-13-genericcloud-amd64.qcow2}"
|
||||
|
||||
# Same version production runs, so CRD shapes and chart flags transfer exactly.
|
||||
CILIUM_VERSION="${CILIUM_VERSION:-1.19.1}"
|
||||
|
||||
node_name() { echo "labsim-k8s$1"; }
|
||||
node_ip() { echo "${K8S_PREFIX}.$((K8S_FIRST_OCTET + $1 - 1))"; }
|
||||
node_ip6() { echo "${K8S_PREFIX_V6}::$((K8S_FIRST_OCTET + $1 - 1))"; }
|
||||
|
||||
# --- base image -----------------------------------------------------------
|
||||
ensure_base_image() {
|
||||
if [ -f "$DEB_BASE" ]; then
|
||||
log "base image present: $(basename "$DEB_BASE")"
|
||||
return
|
||||
fi
|
||||
log "fetching Debian cloud image (~330 MB) -> $DEB_BASE"
|
||||
sudo mkdir -p "$IMG_DIR"
|
||||
# .tmp + mv so an interrupted download never leaves a half image that later
|
||||
# runs treat as valid.
|
||||
sudo curl -fsSL --retry 3 -o "${DEB_BASE}.tmp" "$DEB_URL" \
|
||||
|| die "could not fetch $DEB_URL"
|
||||
sudo mv "${DEB_BASE}.tmp" "$DEB_BASE"
|
||||
log "base image ready"
|
||||
}
|
||||
|
||||
# --- cloud-init -----------------------------------------------------------
|
||||
# The server node writes the join token; agents wait for the API to answer
|
||||
# before joining, because cloud-init ordering across VMs is not guaranteed and
|
||||
# a failed join leaves an agent that never retries.
|
||||
build_k8s_seed() {
|
||||
local iso="$1" vm="$2" ip="$3" role="$4" server_ip="$5" pubkey="$6" ip6="$7" server_ip6="$8"
|
||||
local tmp; tmp="$(mktemp -d)"
|
||||
|
||||
cat > "$tmp/meta-data" <<EOF
|
||||
instance-id: $vm
|
||||
local-hostname: $vm
|
||||
EOF
|
||||
|
||||
cat > "$tmp/network-config" <<EOF
|
||||
version: 2
|
||||
ethernets:
|
||||
enp1s0:
|
||||
match:
|
||||
name: "en*"
|
||||
addresses: [$ip/24, $ip6/64]
|
||||
routes:
|
||||
- to: default
|
||||
via: ${K8S_PREFIX}.1
|
||||
nameservers:
|
||||
addresses: [8.8.8.8, 1.1.1.1]
|
||||
EOF
|
||||
|
||||
local k3s_exec
|
||||
if [ "$role" = "server" ]; then
|
||||
# flannel/servicelb/traefik off: Cilium is the CNI under test, and k3s's
|
||||
# own ServiceLB would fight Cilium for LoadBalancer addresses. Dual-stack:
|
||||
# both families in cluster-cidr/service-cidr and a dual node-ip, mirroring
|
||||
# production. Pod IPs come from Cilium cluster-pool (node.spec.podCIDRs is
|
||||
# immutable), so cluster-cidr v6 just marks the cluster dual-stack.
|
||||
k3s_exec="server --flannel-backend=none --disable-network-policy --disable=servicelb --disable=traefik --node-ip=$ip,$ip6 --tls-san=$ip --cluster-cidr=10.42.0.0/16,${CLUSTER_CIDR_V6} --service-cidr=10.43.0.0/16,${SERVICE_CIDR_V6} --cluster-init"
|
||||
else
|
||||
k3s_exec="agent --server https://${server_ip}:6443 --node-ip=$ip,$ip6"
|
||||
fi
|
||||
|
||||
cat > "$tmp/user-data" <<EOF
|
||||
#cloud-config
|
||||
hostname: $vm
|
||||
fqdn: $vm
|
||||
users:
|
||||
- name: debian
|
||||
groups: [sudo]
|
||||
shell: /bin/bash
|
||||
sudo: ["ALL=(ALL) NOPASSWD:ALL"]
|
||||
lock_passwd: false
|
||||
plain_text_passwd: labsim
|
||||
ssh_authorized_keys:
|
||||
- $pubkey
|
||||
ssh_pwauth: true
|
||||
disable_root: false
|
||||
ssh_authorized_keys:
|
||||
- $pubkey
|
||||
|
||||
package_update: true
|
||||
packages: [curl, jq, iproute2, tcpdump, bird2]
|
||||
|
||||
write_files:
|
||||
# Cilium replaces kube-proxy and needs these; Debian cloud images ship
|
||||
# neither loaded nor persisted.
|
||||
- path: /etc/modules-load.d/cilium.conf
|
||||
content: |
|
||||
br_netfilter
|
||||
overlay
|
||||
- path: /etc/sysctl.d/99-k8s.conf
|
||||
content: |
|
||||
net.ipv4.ip_forward = 1
|
||||
net.bridge.bridge-nf-call-iptables = 1
|
||||
|
||||
runcmd:
|
||||
- [ modprobe, br_netfilter ]
|
||||
- [ modprobe, overlay ]
|
||||
- [ sysctl, --system ]
|
||||
- |
|
||||
# Wait for the server's API before an agent tries to join. Without this the
|
||||
# agent fails once and the unit backs off for minutes.
|
||||
if [ "$role" != "server" ]; then
|
||||
for i in \$(seq 1 60); do
|
||||
curl -sk --max-time 3 https://${server_ip}:6443/ping >/dev/null 2>&1 && break
|
||||
sleep 5
|
||||
done
|
||||
fi
|
||||
- |
|
||||
curl -sfL https://get.k3s.io | \
|
||||
INSTALL_K3S_EXEC="$k3s_exec" \
|
||||
K3S_TOKEN="$K8S_TOKEN" \
|
||||
sh -
|
||||
EOF
|
||||
|
||||
sudo mkdir -p "$(dirname "$iso")"
|
||||
sudo genisoimage -quiet -output "$iso" -volid cidata -joliet -rock \
|
||||
"$tmp/user-data" "$tmp/meta-data" "$tmp/network-config"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
# --- VM creation ----------------------------------------------------------
|
||||
create_node() {
|
||||
local n="$1" pubkey="$2"
|
||||
local vm; vm="$(node_name "$n")"
|
||||
local ip; ip="$(node_ip "$n")"
|
||||
local ip6; ip6="$(node_ip6 "$n")"
|
||||
local role="agent"; [ "$n" -eq 1 ] && role="server"
|
||||
local server_ip; server_ip="$(node_ip 1)"
|
||||
local server_ip6; server_ip6="$(node_ip6 1)"
|
||||
|
||||
if virsh_q dominfo "$vm" >/dev/null 2>&1; then
|
||||
local state; state="$(virsh_q domstate "$vm" 2>/dev/null | head -1 | tr -d '\n')"
|
||||
if [ "$state" = "running" ]; then
|
||||
log "$vm already running ($ip, $role)"
|
||||
else
|
||||
log "$vm exists but is $state — starting"
|
||||
virsh_q start "$vm" >/dev/null
|
||||
fi
|
||||
return
|
||||
fi
|
||||
|
||||
local disk="$IMG_DIR/${vm}.qcow2"
|
||||
local seed="$IMG_DIR/${vm}-seed.iso"
|
||||
|
||||
log "creating $vm ($ip, $role, ${K8S_MEM}MB/${K8S_CPUS}cpu)"
|
||||
sudo qemu-img create -q -f qcow2 -F qcow2 -b "$DEB_BASE" "$disk" "${K8S_DISK_GB}G" >/dev/null
|
||||
build_k8s_seed "$seed" "$vm" "$ip" "$role" "$server_ip" "$pubkey" "$ip6" "$server_ip6"
|
||||
|
||||
# Access port on the k8s VLAN — same broadcast domain as the routers'
|
||||
# bond0.2 leg, so BGP peering is directly connected exactly as in production.
|
||||
sudo virt-install --connect "$LIBVIRT_URI" --name "$vm" \
|
||||
--memory "$K8S_MEM" --vcpus "$K8S_CPUS" \
|
||||
--disk "path=$disk,format=qcow2,bus=virtio" \
|
||||
--disk "path=$seed,device=cdrom" \
|
||||
--network "network=$OVS_NET,portgroup=vlan${K8S_VLAN},model=virtio" \
|
||||
--os-variant debian12 \
|
||||
--graphics none --noautoconsole --import >/dev/null
|
||||
}
|
||||
|
||||
fetch_kubeconfig() {
|
||||
local server_ip; server_ip="$(node_ip 1)"
|
||||
local out="$SCRIPT_DIR/labsim-k8s.kubeconfig"
|
||||
log "fetching kubeconfig from $server_ip"
|
||||
ssh -o StrictHostKeyChecking=no -o ConnectTimeout=10 \
|
||||
"debian@${server_ip}" "sudo cat /etc/rancher/k3s/k3s.yaml" \
|
||||
| sed "s|127.0.0.1|${server_ip}|" > "$out"
|
||||
chmod 600 "$out"
|
||||
log "wrote $out"
|
||||
log "use: KUBECONFIG=$out kubectl get nodes"
|
||||
}
|
||||
|
||||
main() {
|
||||
if [ "${1:-}" = "--kubeconfig" ]; then
|
||||
fetch_kubeconfig
|
||||
return
|
||||
fi
|
||||
|
||||
require_tools
|
||||
command -v genisoimage >/dev/null || die "genisoimage missing (dnf install genisoimage)"
|
||||
|
||||
local pubkey; pubkey="$(find_ssh_pubkey)"
|
||||
log "using SSH key: ${pubkey%% *} ...${pubkey##* }"
|
||||
|
||||
ensure_base_image
|
||||
|
||||
# Select EVERY VLAN, not just the k8s one. ovs_up re-defines the libvirt
|
||||
# network from SELECTED, so narrowing it here silently drops the portgroups
|
||||
# for every other VLAN -- running VMs keep working (their taps are already
|
||||
# attached) and nothing complains until the next VM cannot be attached.
|
||||
# Observed: this deleted vlan1/3/9/10/200/51/53 and only surfaced when the
|
||||
# ISP VMs needed vlan51 and vlan53.
|
||||
selected_vlans
|
||||
log "ensuring OVS fabric (all VLANs, so no portgroup is dropped)"
|
||||
ovs_up
|
||||
|
||||
for n in $(seq 1 "$K8S_NODES"); do
|
||||
create_node "$n" "$pubkey"
|
||||
done
|
||||
|
||||
echo
|
||||
log "nodes created. k3s installs on first boot (a few minutes)."
|
||||
log "watch: ssh debian@$(node_ip 1) 'sudo systemctl status k3s'"
|
||||
log "then: $0 --kubeconfig"
|
||||
log "then install Cilium $CILIUM_VERSION and the BGP resources (see README)."
|
||||
}
|
||||
|
||||
main "$@"
|
||||
116
labsim/labsim-dualstack-convert.sh
Executable file
116
labsim/labsim-dualstack-convert.sh
Executable file
@@ -0,0 +1,116 @@
|
||||
#!/bin/bash
|
||||
# Rehearse the k3s single -> dual-stack conversion on the 3-server etcd cluster,
|
||||
# the way production Phase 4 will do it: edit each server's config.yaml (rendered
|
||||
# by the PRODUCTION generator) to add the second address family, restart ONE
|
||||
# server at a time, and watch what happens in between.
|
||||
#
|
||||
# Answers the questions a single-node lab cannot:
|
||||
# - does quorum survive a rolling config.yaml change across 3 etcd servers?
|
||||
# - what does a MIXED control plane do (one server dual, two still v4-only)?
|
||||
# - does ServiceCIDR pick up the v6 range on the FIRST server's restart, or
|
||||
# only once all three agree?
|
||||
#
|
||||
# The node-ip v6 must exist on the box before k3s reads it, so each server first
|
||||
# gets a ULA on its interface (fd00:2::3x), mirroring how production nodes get a
|
||||
# DHCPv6 address before k3s starts.
|
||||
#
|
||||
# ./labsim-dualstack-convert.sh baseline snapshot the v4-only starting state
|
||||
# ./labsim-dualstack-convert.sh convert roll the conversion, snapshotting each step
|
||||
# ./labsim-dualstack-convert.sh snapshot print current SC / podCIDRs / quorum
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
NET="${NET:-172.31.2}"; FIRST_OCTET="${FIRST_OCTET:-31}"; SERVERS="${SERVERS:-3}"
|
||||
TOKEN="${TOKEN:-labsim-etcd-token}"
|
||||
RENDER="${RENDER:-$SCRIPT_DIR/../bastion/src/modules/dist/modules/k3s/bin/render-config.js}"
|
||||
EVID="${EVID:-$SCRIPT_DIR/dualstack-evidence}"
|
||||
|
||||
# Dual-stack target ranges. ULA/v4 -- the mechanism is what's under test, not the
|
||||
# addresses; using ULA keeps sim traffic out of the real /48.
|
||||
V4_CLUSTER="10.42.0.0/16"; V6_CLUSTER="fd00:42::/56"
|
||||
V4_SERVICE="10.43.0.0/16"; V6_SERVICE="fd00:43::/112" # /112: apiserver caps v6 service ranges
|
||||
V6_NODE_PREFIX="fd00:2" # node-ip v6: fd00:2::3x
|
||||
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o LogLevel=ERROR -o ConnectTimeout=8 -o BatchMode=yes)
|
||||
node_ip() { echo "${NET}.$((FIRST_OCTET + $1 - 1))"; }
|
||||
node_v6() { echo "${V6_NODE_PREFIX}::$((FIRST_OCTET + $1 - 1))"; }
|
||||
node_name() { echo "labsim-etcd$1"; }
|
||||
s() { local n="$1"; shift; timeout "${TMO:-25}" ssh "${SSH[@]}" "debian@$(node_ip "$n")" "$@" 2>/dev/null; }
|
||||
k() { s 1 "sudo k3s kubectl $*"; }
|
||||
log() { printf '\033[36m==>\033[0m %s\n' "$*"; }
|
||||
|
||||
snapshot() {
|
||||
# custom-columns, not jsonpath: the jsonpath range/quotes get mangled through
|
||||
# ssh -> sudo -> kubectl and came back empty.
|
||||
echo " servicecidr :"
|
||||
s 1 'sudo k3s kubectl get servicecidr -o custom-columns=NAME:.metadata.name,CIDRS:.spec.cidrs --no-headers' 2>/dev/null | sed 's/^/ /'
|
||||
echo " node podCIDRs:"
|
||||
s 1 'sudo k3s kubectl get nodes -o custom-columns=NAME:.metadata.name,PODCIDRS:.spec.podCIDRs --no-headers' 2>/dev/null | sed 's/^/ /'
|
||||
local ready; ready="$(k get nodes --no-headers 2>/dev/null | wc -l)"
|
||||
echo " nodes registered: $ready ; apiserver: $(k get --raw /readyz >/dev/null 2>&1 && echo ok || echo DOWN)"
|
||||
}
|
||||
|
||||
# Re-render node n's config.yaml WITH the dual families and write it back.
|
||||
convert_node() {
|
||||
local n="$1"
|
||||
local ip; ip="$(node_ip "$n")"; local v6; v6="$(node_v6 "$n")"
|
||||
log "server $n ($ip): add $v6, re-render dual config.yaml, restart k3s"
|
||||
|
||||
# 1. node-ip v6 must exist before k3s reads it.
|
||||
s "$n" "sudo ip -6 addr add ${v6}/64 dev enp1s0 2>/dev/null; ip -6 -br addr show enp1s0 | grep -o '${v6}/64'" | sed 's/^/ addr: /'
|
||||
|
||||
# 2. render the dual config from the PRODUCTION generator, preserving this
|
||||
# node's role (node 1 cluster-init, else joining server).
|
||||
local extra=""
|
||||
[ "$n" -ne 1 ] && extra="K3S_SERVER_URL=https://$(node_ip 1):6443 K3S_TOKEN=$TOKEN"
|
||||
local cfg
|
||||
cfg="$(env ROLE=infra HOSTNAME="$(node_name "$n")" IP="$ip" TLS_SANS="$ip" \
|
||||
IPV6="$v6" CLUSTER_CIDR="${V4_CLUSTER},${V6_CLUSTER}" SERVICE_CIDR="${V4_SERVICE},${V6_SERVICE}" \
|
||||
$extra node "$RENDER")"
|
||||
# sanity: the render must actually carry both families or the restart is pointless
|
||||
echo "$cfg" | grep -q "$V6_CLUSTER" || { echo " RENDER MISSING v6 -- aborting"; return 1; }
|
||||
|
||||
# 3. write it back and restart k3s on this one server.
|
||||
printf '%s\n' "$cfg" | s "$n" "sudo tee /etc/rancher/k3s/config.yaml >/dev/null && sudo systemctl restart k3s"
|
||||
|
||||
# 4. wait for THIS server's k3s to come back and the apiserver to answer.
|
||||
local i
|
||||
for i in $(seq 1 30); do
|
||||
[ "$(s "$n" 'sudo systemctl is-active k3s' 2>/dev/null)" = active ] && \
|
||||
s "$n" 'sudo k3s kubectl get --raw /readyz >/dev/null 2>&1' && break
|
||||
sleep 10
|
||||
done
|
||||
echo " server $n k3s=$(s "$n" 'systemctl is-active k3s') restarts=$(s "$n" 'systemctl show k3s -p NRestarts --value')"
|
||||
}
|
||||
|
||||
cmd_baseline() {
|
||||
mkdir -p "$EVID"
|
||||
{ echo "=== BASELINE (v4-only) $(date -u +%FT%TZ) ==="; snapshot; } | tee "$EVID/convert-baseline.txt"
|
||||
}
|
||||
|
||||
cmd_snapshot() { snapshot; }
|
||||
|
||||
cmd_convert() {
|
||||
mkdir -p "$EVID"
|
||||
local out="$EVID/convert-run.txt"
|
||||
{
|
||||
echo "=== 3-SERVER DUAL-STACK CONVERSION $(date -u +%FT%TZ) ==="
|
||||
echo "--- before ---"; snapshot
|
||||
local n
|
||||
for n in $(seq 1 "$SERVERS"); do
|
||||
echo; echo "### converting server $n of $SERVERS ###"
|
||||
convert_node "$n" || { echo "convert_node $n failed"; break; }
|
||||
echo "--- state after server $n (MIXED until n=$SERVERS) ---"
|
||||
snapshot
|
||||
done
|
||||
echo; echo "--- FINAL ---"; snapshot
|
||||
} 2>&1 | tee "$out"
|
||||
log "evidence -> $out"
|
||||
}
|
||||
|
||||
case "${1:-snapshot}" in
|
||||
baseline) cmd_baseline ;;
|
||||
convert) cmd_convert ;;
|
||||
snapshot) cmd_snapshot ;;
|
||||
*) echo "usage: $0 {baseline|convert|snapshot}" >&2; exit 2 ;;
|
||||
esac
|
||||
221
labsim/labsim-dualstack-net.sh
Executable file
221
labsim/labsim-dualstack-net.sh
Executable file
@@ -0,0 +1,221 @@
|
||||
#!/bin/bash
|
||||
# VLAN 2 gets IPv6 in labsim: addresses, router advertisements and a DHCPv6
|
||||
# server with per-MAC reservations.
|
||||
#
|
||||
# This is the rehearsal of the production change (dual-stack plan, phase 2b) and
|
||||
# runs the SAME VyOS config, against the sim router pair, so the production
|
||||
# apply is a repeat rather than a first attempt.
|
||||
#
|
||||
# WHY DHCPv6 AND NOT SLAAC. The cluster needs each node's IPv6 to be knowable in
|
||||
# advance and stable: k3s resolves node-ip once at start-up, and a node's
|
||||
# identity cannot be allowed to change under it. The estate already answers that
|
||||
# question for IPv4 with kea reservations keyed on MAC, so IPv6 answers it the
|
||||
# same way and stays one source of truth.
|
||||
#
|
||||
# DHCPv6 normally keys on DUID, not MAC -- a DUID is generated by the client and
|
||||
# is not derivable from its MAC, which would have meant a second, client-owned
|
||||
# source of truth. VyOS's static-mapping accepts `mac` as well as `duid`
|
||||
# (verified on the sim: the node.tag directory offers duid, mac, ipv6-address,
|
||||
# ipv6-prefix), so the reservation can key on the same MAC the v4 one does.
|
||||
#
|
||||
# The RA carries managed-flag WITH no-autonomous-flag. That combination is what
|
||||
# makes the node's address unambiguous: managed sends it to DHCPv6, and
|
||||
# non-autonomous stops it also forming a SLAAC address from the same prefix.
|
||||
# Leave autonomous on and every node has two global addresses, only one of which
|
||||
# anybody reserved -- and whichever labctl happens to find is the one that ends
|
||||
# up in node-ip.
|
||||
#
|
||||
# Both are VALUELESS nodes on this VyOS: `managed-flag` not `managed-flag true`,
|
||||
# and `prefix <p> no-autonomous-flag` not `autonomous-flag false`. The value
|
||||
# forms are rejected with "is not valid".
|
||||
#
|
||||
# A FAILED COMMIT DOES NOT MEAN NOTHING CHANGED. VyOS commits node groups
|
||||
# independently, so an earlier run of this script left the bond0.2 address and
|
||||
# most of the router-advert block applied while `[[service dhcpv6-server]]
|
||||
# failed` -- the interface and RA groups had already succeeded. Re-read the
|
||||
# config after any failure rather than assuming a clean rollback; that matters
|
||||
# more in production than here.
|
||||
#
|
||||
# TWO THINGS THIS REHEARSAL FOUND, both of which would have bitten production:
|
||||
#
|
||||
# 1. managed-flag is NECESSARY BUT NOT SUFFICIENT. It only tells the host to use
|
||||
# DHCPv6; the kernel's accept_ra implements SLAAC and nothing else, so a node
|
||||
# with no DHCPv6 *client* running takes no lease at all. The sim's Debian
|
||||
# nodes have neither NetworkManager nor a networkd .network file managing the
|
||||
# interface, so they ended up with a kernel SLAAC address
|
||||
# (2001:db8:187e:2:5054:ff:fe53:731 -- EUI-64 from the MAC) and never asked
|
||||
# for the reserved ::11. Production's Fedora nodes use NetworkManager, which
|
||||
# does run a DHCPv6 client on the managed flag, and the DGX Sparks run
|
||||
# NetworkManager too -- but that must be VERIFIED per node, not assumed. The
|
||||
# k3s module's preflight catches the consequence; the fix is node-side.
|
||||
#
|
||||
# 2. TURNING AUTONOMOUS OFF DOES NOT RETRACT ADDRESSES ALREADY FORMED. An
|
||||
# earlier partial apply advertised the prefix while autonomous was still on;
|
||||
# the nodes autoconfigured, and adding no-autonomous-flag afterwards left
|
||||
# those addresses in place with a 30-day valid lifetime. So in production,
|
||||
# where VLAN 2 has no IPv6 at all yet, no-autonomous-flag MUST be in the same
|
||||
# commit that first advertises the prefix. Advertise first and tighten later
|
||||
# and every node carries an unreserved EUI-64 address that labctl might pick
|
||||
# up as node-ip.
|
||||
#
|
||||
# ./labsim-dualstack-net.sh up apply to both sim routers
|
||||
# ./labsim-dualstack-net.sh down remove it again
|
||||
# ./labsim-dualstack-net.sh status what the routers and nodes think
|
||||
# ./labsim-dualstack-net.sh leases who has taken an address
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
R1="${R1:-172.31.1.252}" # sim vyos001 -> ::1
|
||||
R2="${R2:-172.31.1.253}" # sim vyos002 -> ::2
|
||||
PW="${VYOS_PW:-vyos}"
|
||||
|
||||
# Mirrors production's scheme (2001:470:187e:<vlan>::/64) using the
|
||||
# documentation prefix, so the shape is rehearsed without putting sim addresses
|
||||
# inside a range production announces.
|
||||
V6_PREFIX="${V6_PREFIX:-2001:db8:187e:2}"
|
||||
LINK_MTU="${LINK_MTU:-1472}"
|
||||
# VyOS requires a unique subnet-id per DHCPv6 subnet ("Unique subnet ID not
|
||||
# specified for subnet"). Using the VLAN id keeps it self-documenting and
|
||||
# collision-free across VLANs.
|
||||
VLAN_ID="${VLAN_ID:-2}"
|
||||
|
||||
# node -> last hextet. Mirrors the v4 host part (172.31.2.11 -> ::11) so a
|
||||
# reservation is readable next to its IPv4 twin.
|
||||
NODES=(labsim-k8s1:11 labsim-k8s2:12 labsim-k8s3:13)
|
||||
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6 -o PreferredAuthentications=password)
|
||||
|
||||
log() { printf '\033[0;36m[ds-net]\033[0m %s\n' "$*"; }
|
||||
warn() { printf '\033[1;33m[ds-net]\033[0m %s\n' "$*" >&2; }
|
||||
die() { printf '\033[0;31m[ds-net]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
# Drive VyOS from a script FILE, never `vbash -c`. The latter never starts a
|
||||
# config session: commit fails to stderr, a helper discards it, and the run
|
||||
# reports success having changed nothing. labsim-pppoe-ha-test.sh lost a whole
|
||||
# policy matrix to exactly that.
|
||||
vyos_apply() { # host, set-lines on stdin
|
||||
local h="$1" out
|
||||
out="$({ printf '#!/bin/vbash\nsource /opt/vyatta/etc/functions/script-template\nconfigure\n'
|
||||
cat
|
||||
printf 'commit\nsave\nexit\n'
|
||||
} | timeout 120 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$h" \
|
||||
'cat > /tmp/ds-net.sh && chmod +x /tmp/ds-net.sh && sudo /tmp/ds-net.sh' 2>&1)"
|
||||
# Read the outcome instead of assuming it. The first version of this script
|
||||
# printed "up" after BOTH routers had failed to commit -- the same shape of
|
||||
# lie this repo has been bitten by before (a matrix reporting coverage it did
|
||||
# not have). vbash exits 0 even when the commit fails, so the text is the
|
||||
# only honest signal.
|
||||
printf '%s\n' "$out" | grep -vE '^\s*$' | sed 's/^/ /' | tail -6
|
||||
if printf '%s' "$out" | grep -qiE 'Commit failed|\[\[.*\]\] failed|Set failed'; then
|
||||
return 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
r() { timeout 40 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$1" "${@:2}" 2>/dev/null; }
|
||||
|
||||
mac_of() { sudo virsh domiflist "$1" 2>/dev/null | awk '/52:54/{print $5; exit}'; }
|
||||
|
||||
up() {
|
||||
log "collecting node MACs (the reservation key, same as IPv4)"
|
||||
local mappings="" name hextet mac
|
||||
for entry in "${NODES[@]}"; do
|
||||
name="${entry%%:*}"; hextet="${entry##*:}"
|
||||
mac="$(mac_of "$name")"
|
||||
[ -n "$mac" ] || die "no MAC for $name -- is the sim cluster up? (./k8s-up.sh)"
|
||||
log " $name $mac -> ${V6_PREFIX}::${hextet}"
|
||||
# No trailing newline: `${mappings}` sits on its own line in the heredoc
|
||||
# below and supplies its own. A heredoc terminator is matched LITERALLY
|
||||
# in the source before any expansion, so `${mappings}EOF` is not the
|
||||
# delimiter -- the heredoc ran on, swallowed the rest of this function
|
||||
# and part of down(), and the apply failed with "Invalid command: [EOF]".
|
||||
[ -n "$mappings" ] && mappings+=$'\n'
|
||||
mappings+="set service dhcpv6-server shared-network-name K8S subnet ${V6_PREFIX}::/64 static-mapping ${name} mac '${mac}'
|
||||
set service dhcpv6-server shared-network-name K8S subnet ${V6_PREFIX}::/64 static-mapping ${name} ipv6-address '${V6_PREFIX}::${hextet}'"
|
||||
done
|
||||
|
||||
local host pref hextet_self
|
||||
for host in "$R1" "$R2"; do
|
||||
if [ "$host" = "$R1" ]; then pref=high; hextet_self=1; else pref=low; hextet_self=2; fi
|
||||
log "applying to $host (router ${V6_PREFIX}::${hextet_self}, RA preference $pref)"
|
||||
vyos_apply "$host" <<EOF || die "commit failed on $host -- nothing applied there"
|
||||
set interfaces bonding bond0 vif 2 address '${V6_PREFIX}::${hextet_self}/64'
|
||||
set service router-advert interface bond0.2 prefix ${V6_PREFIX}::/64 valid-lifetime '2592000'
|
||||
set service router-advert interface bond0.2 prefix ${V6_PREFIX}::/64 preferred-lifetime '604800'
|
||||
set service router-advert interface bond0.2 prefix ${V6_PREFIX}::/64 no-autonomous-flag
|
||||
set service router-advert interface bond0.2 managed-flag
|
||||
set service router-advert interface bond0.2 link-mtu '${LINK_MTU}'
|
||||
set service router-advert interface bond0.2 default-preference '${pref}'
|
||||
set service dhcpv6-server shared-network-name K8S subnet ${V6_PREFIX}::/64 subnet-id '${VLAN_ID}'
|
||||
${mappings}
|
||||
EOF
|
||||
done
|
||||
log "up. Nodes need a DHCPv6 client on their VLAN 2 interface -- see 'status'."
|
||||
}
|
||||
|
||||
down() {
|
||||
local host
|
||||
for host in "$R1" "$R2"; do
|
||||
log "removing from $host"
|
||||
vyos_apply "$host" <<EOF
|
||||
delete service dhcpv6-server
|
||||
delete service router-advert interface bond0.2
|
||||
delete interfaces bonding bond0 vif 2 address '${V6_PREFIX}::$([ "$host" = "$R1" ] && echo 1 || echo 2)/64'
|
||||
EOF
|
||||
done
|
||||
log "down"
|
||||
}
|
||||
|
||||
status() {
|
||||
local host
|
||||
for host in "$R1" "$R2"; do
|
||||
printf ' --- %s ---\n' "$host"
|
||||
printf ' bond0.2 v6 : %s\n' "$(r "$host" 'ip -6 -br addr show bond0.2 | tr -s " "' || echo '<unreachable>')"
|
||||
printf ' radvd : %s\n' "$(r "$host" 'systemctl is-active radvd')"
|
||||
# ps|grep, not pgrep: the bracket idiom that stops a self-match gets
|
||||
# mangled through ssh -> vbash quoting and reported "NOT running" for a
|
||||
# daemon that was plainly up.
|
||||
printf ' kea-dhcp6 : %s\n' "$(r "$host" 'c=$(ps -ef | grep -c "[k]ea-dhcp6"); [ "$c" -gt 0 ] && echo running || echo "NOT running"')"
|
||||
# radvd is expected INACTIVE on the backup: vrrp-wan-reconcile's IPv6
|
||||
# plane stops it there so only the master advertises on any VLAN. Not a
|
||||
# fault -- see the step-0 IPv6-follows-master work.
|
||||
printf ' role : %s\n' "$(r "$host" 'sudo /config/vrrp-wan-reconcile --status 2>/dev/null | grep -o "role=[a-z]*"')"
|
||||
done
|
||||
printf ' --- nodes ---\n'
|
||||
local entry name
|
||||
for entry in "${NODES[@]}"; do
|
||||
name="${entry%%:*}"
|
||||
printf ' %-14s %s\n' "$name" "$(node_v6 "$name")"
|
||||
done
|
||||
}
|
||||
|
||||
# What global IPv6 does the node actually hold? This is the question labctl asks
|
||||
# before it will write node-ip, so ask it the same way.
|
||||
node_v6() {
|
||||
# Ask the node, not the hypervisor: virsh domifaddr --source agent needs
|
||||
# qemu-guest-agent, which these images do not carry, and returned nothing
|
||||
# while the nodes plainly had addresses.
|
||||
local name="$1" v4 out
|
||||
case "$name" in *k8s1) v4=172.31.2.11 ;; *k8s2) v4=172.31.2.12 ;; *k8s3) v4=172.31.2.13 ;; *) echo "<unknown>"; return ;; esac
|
||||
out="$(timeout 20 ssh -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6 -o BatchMode=yes "debian@$v4" \
|
||||
'ip -6 -br addr show scope global 2>/dev/null | tr -s " "' 2>/dev/null)"
|
||||
echo "${out:-<unreachable>}"
|
||||
}
|
||||
|
||||
leases() {
|
||||
local host
|
||||
for host in "$R1" "$R2"; do
|
||||
printf ' --- %s ---\n' "$host"
|
||||
r "$host" '/opt/vyatta/bin/vyatta-op-cmd-wrapper show dhcpv6-server leases 2>/dev/null || echo " (no leases command / no leases)"'
|
||||
done
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
up) up ;;
|
||||
down) down ;;
|
||||
status) status ;;
|
||||
leases) leases ;;
|
||||
*) die "usage: $0 {up|down|status|leases}" ;;
|
||||
esac
|
||||
280
labsim/labsim-he-endpoint.sh
Executable file
280
labsim/labsim-he-endpoint.sh
Executable file
@@ -0,0 +1,280 @@
|
||||
#!/bin/bash
|
||||
# Build a fake Hurricane Electric 6in4 endpoint inside labsim.
|
||||
#
|
||||
# WHY THIS EXISTS
|
||||
# The production HE tunnel was never rehearsed. The override that introduced it
|
||||
# says so in its own reason text -- "the sim has no public IPv4 and no HE
|
||||
# endpoint, so there is nothing to tunnel to" -- and that gap is why IPv6 was
|
||||
# the one half of the WAN story with no matrix behind it. Then the WAN became
|
||||
# HA and IPv6 did not follow, which nobody caught, because nothing tests it.
|
||||
#
|
||||
# THE PROBLEM THIS HAD TO SOLVE
|
||||
# The sim's two WANs are isolated islands. Verified:
|
||||
#
|
||||
# 203.0.113.1 from 203.0.113.107 : OK <- 10 gig analogue
|
||||
# 203.0.113.1 from 198.51.100.137 : unreachable
|
||||
# 198.51.100.1 from 198.51.100.137 : OK <- PPPoE analogue
|
||||
# 198.51.100.1 from 203.0.113.107 : unreachable
|
||||
#
|
||||
# A 6in4 tunnel has ONE remote address, and production never changes it -- so an
|
||||
# endpoint reachable over only one WAN could not rehearse the case that matters
|
||||
# most: the WITHIN-box fall back from the 10 gig to PPPoE, where he-tunnel-follow
|
||||
# re-points the tunnel and calls the HE API. That is exactly where the 2026-09-06
|
||||
# near-miss lived.
|
||||
#
|
||||
# So the sim needs a minimal "internet": both ISP boxes already sit on the
|
||||
# libvirt default network (192.168.122.0/24) and both forward, so that becomes
|
||||
# the backbone, and HE lives on a single address behind it, reachable over
|
||||
# either WAN. No new VMs, no new networks.
|
||||
#
|
||||
# 192.0.2.10 "HE" -- on isp-dhcp, reached from the PPPoE island via
|
||||
# 192.168.122.136, and directly from the 10 gig island
|
||||
#
|
||||
# EVERYTHING HERE IS KERNEL-LEVEL, not VyOS config. The ISP boxes are scaffold,
|
||||
# not the thing under test: `ip` commands leave no config to drift, no commit to
|
||||
# fail, and a reboot cleans up. The ROUTER side is deliberately the opposite --
|
||||
# it goes through real VyOS config, because "will VyOS commit a tunnel whose
|
||||
# source-address does not exist on this box?" is one of the questions.
|
||||
#
|
||||
# ./labsim-he-endpoint.sh up build it
|
||||
# ./labsim-he-endpoint.sh down tear it down
|
||||
# ./labsim-he-endpoint.sh status what is live
|
||||
# ./labsim-he-endpoint.sh calls how many times the HE API was called
|
||||
# ./labsim-he-endpoint.sh point IP point the endpoint by hand (sim bookkeeping)
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
DHCP_ISP="${DHCP_ISP:-192.168.122.136}" # owns the 10 gig segment, hosts "HE"
|
||||
PPPOE_ISP="${PPPOE_ISP:-192.168.122.63}" # owns the PPPoE segment
|
||||
PW="${VYOS_PW:-vyos}"
|
||||
|
||||
# TEST-NET-1 for the endpoint, and the documentation prefix for v6. Production
|
||||
# uses 2001:470:187e::/48 from HE; the sim mirrors its SHAPE
|
||||
# (2001:db8:187e:<vlan>::/64) so the scheme is exercised, not just the tunnel.
|
||||
HE_ADDR="${HE_ADDR:-192.0.2.10}"
|
||||
HE_LINK6="${HE_LINK6:-2001:db8:1f1c:f6::1}" # HE side of the tunnel /64
|
||||
RT_LINK6="${RT_LINK6:-2001:db8:1f1c:f6::2}" # router side
|
||||
SITE6="${SITE6:-2001:db8:187e::/48}" # routed to the router side
|
||||
TENGIG_NET="${TENGIG_NET:-203.0.113.0/24}"
|
||||
# Any valid address that will never be a router WAN -- see the tunnel creation
|
||||
# below for why this cannot be 0.0.0.0.
|
||||
PLACEHOLDER_REMOTE="${PLACEHOLDER_REMOTE:-203.0.113.1}"
|
||||
PPPOE_NET="${PPPOE_NET:-198.51.100.0/24}"
|
||||
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6 -o PreferredAuthentications=password)
|
||||
dhcp_isp() { timeout 40 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$DHCP_ISP" "$@" 2>/dev/null; }
|
||||
pppoe_isp() { timeout 40 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$PPPOE_ISP" "$@" 2>/dev/null; }
|
||||
|
||||
log() { printf '\033[0;36m[he-sim]\033[0m %s\n' "$*"; }
|
||||
die() { printf '\033[0;31m[he-sim]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
# --- the stub tunnelbroker API ---------------------------------------------
|
||||
# HE's real endpoint is a dyndns-style updater that re-points the tunnel's remote
|
||||
# address. This is that, in 40 lines, so the sim can exercise the HE-SIDE half of
|
||||
# a failover -- the half production can never safely test.
|
||||
#
|
||||
# It logs every call to /run/he-sim-api.log, which is what lets the matrix assert
|
||||
# the invariant that matters: a ROUTER-level failover must call this ZERO times,
|
||||
# because the 10 gig address follows the cloned MAC to the other box unchanged.
|
||||
API_PY='
|
||||
import http.server, subprocess, urllib.parse, datetime, sys, re
|
||||
|
||||
TUN = "he-sim"
|
||||
LOCAL = sys.argv[1] if len(sys.argv) > 1 else "192.0.2.10"
|
||||
V6_LOCAL = sys.argv[2] if len(sys.argv) > 2 else "2001:db8:1f1c:f6::1/64"
|
||||
V6_PEER = sys.argv[3] if len(sys.argv) > 3 else "2001:db8:1f1c:f6::2"
|
||||
SITE6 = sys.argv[4] if len(sys.argv) > 4 else "2001:db8:187e::/48"
|
||||
LOG = "/run/he-sim-api.log"
|
||||
|
||||
def note(msg):
|
||||
with open(LOG, "a") as f:
|
||||
f.write("%s %s\n" % (datetime.datetime.now().isoformat(timespec="seconds"), msg))
|
||||
|
||||
def current_remote():
|
||||
out = subprocess.run(["ip", "tunnel", "show", TUN], capture_output=True, text=True).stdout
|
||||
m = re.search(r"remote ([0-9.]+)", out)
|
||||
return m.group(1) if m else None
|
||||
|
||||
class H(http.server.BaseHTTPRequestHandler):
|
||||
def reply(self, body):
|
||||
b = body.encode()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/plain")
|
||||
self.send_header("Content-Length", str(len(b)))
|
||||
self.end_headers()
|
||||
self.wfile.write(b)
|
||||
|
||||
def handle_update(self, qs):
|
||||
q = urllib.parse.parse_qs(qs)
|
||||
ip = (q.get("myip") or [""])[0]
|
||||
if not ip:
|
||||
note("REFUSED no myip"); return self.reply("nohost")
|
||||
cur = current_remote()
|
||||
if cur == ip:
|
||||
note("nochg %s" % ip); return self.reply("nochg %s" % ip)
|
||||
rc = subprocess.run(["ip", "tunnel", "change", TUN, "mode", "sit",
|
||||
"local", LOCAL, "remote", ip],
|
||||
capture_output=True, text=True)
|
||||
if rc.returncode != 0:
|
||||
# Recreate rather than report a success we did not achieve. This is
|
||||
# the path that a multipoint tunnel takes; keeping it means a stub
|
||||
# that cannot silently no-op.
|
||||
note("change failed (%s) -- recreating" % rc.stderr.strip())
|
||||
subprocess.run(["ip", "tunnel", "del", TUN], check=False)
|
||||
subprocess.run(["ip", "tunnel", "add", TUN, "mode", "sit",
|
||||
"local", LOCAL, "remote", ip, "ttl", "64"], check=False)
|
||||
subprocess.run(["ip", "link", "set", TUN, "up", "mtu", "1480"], check=False)
|
||||
subprocess.run(["ip", "-6", "addr", "replace", V6_LOCAL, "dev", TUN], check=False)
|
||||
subprocess.run(["ip", "-6", "route", "replace", SITE6, "via", V6_PEER,
|
||||
"dev", TUN], check=False)
|
||||
# Verify rather than trust: read the remote back.
|
||||
got = current_remote()
|
||||
if got != ip:
|
||||
note("FAILED to point %s at %s (reads %s)" % (TUN, ip, got))
|
||||
return self.reply("dnserr")
|
||||
note("good %s (was %s)" % (ip, cur))
|
||||
self.reply("good %s" % ip)
|
||||
|
||||
def do_GET(self):
|
||||
u = urllib.parse.urlparse(self.path)
|
||||
if u.path == "/nic/update": self.handle_update(u.query)
|
||||
else: self.reply("badauth")
|
||||
|
||||
def do_POST(self):
|
||||
n = int(self.headers.get("Content-Length") or 0)
|
||||
self.handle_update(self.rfile.read(n).decode())
|
||||
|
||||
def log_message(self, *a): pass
|
||||
|
||||
http.server.HTTPServer(("0.0.0.0", 80), H).serve_forever()
|
||||
'
|
||||
|
||||
up() {
|
||||
log "backbone: teaching each ISP box how to reach the other island"
|
||||
# isp-dhcp owns HE and must be able to answer a router that arrived over
|
||||
# PPPoE, so it needs a route back to that island via the backbone.
|
||||
dhcp_isp "sudo ip route replace $PPPOE_NET via $PPPOE_ISP" \
|
||||
|| die "could not add the PPPoE-island route on isp-dhcp"
|
||||
# isp-pppoe must forward its clients' traffic for HE across the backbone.
|
||||
pppoe_isp "sudo ip route replace $HE_ADDR/32 via $DHCP_ISP" \
|
||||
|| die "could not add the HE route on isp-pppoe"
|
||||
|
||||
log "HE endpoint: $HE_ADDR on isp-dhcp"
|
||||
# A dummy interface, not a loopback alias: `ip tunnel` wants a real local
|
||||
# address and a dummy is the honest way to have one that is not tied to
|
||||
# either WAN segment -- which is the point, HE is neither.
|
||||
dhcp_isp "sudo modprobe dummy 2>/dev/null;
|
||||
sudo ip link add he-lo type dummy 2>/dev/null;
|
||||
sudo ip link set he-lo up;
|
||||
sudo ip addr replace $HE_ADDR/32 dev he-lo"
|
||||
|
||||
log "6in4 tunnel he-sim: local $HE_ADDR, remote set by the API on demand"
|
||||
# A PLACEHOLDER remote, not 0.0.0.0. A sit tunnel created with `remote any`
|
||||
# is multipoint (6rd-shaped), and `ip tunnel change` then refuses to convert
|
||||
# it to point-to-point -- "add tunnel he-sim failed: Invalid argument". The
|
||||
# API's update silently did nothing, so the stub logged "good", production's
|
||||
# he-tunnel-follow logged success, and the tunnel still pointed nowhere.
|
||||
# Created point-to-point from the start, `change` works.
|
||||
dhcp_isp "sudo ip tunnel del he-sim 2>/dev/null;
|
||||
sudo ip tunnel add he-sim mode sit local $HE_ADDR remote $PLACEHOLDER_REMOTE ttl 64;
|
||||
sudo ip link set he-sim up mtu 1480;
|
||||
sudo ip -6 addr replace $HE_LINK6/64 dev he-sim;
|
||||
sudo ip -6 route replace $SITE6 via $RT_LINK6 dev he-sim;
|
||||
sudo sysctl -qw net.ipv6.conf.all.forwarding=1"
|
||||
|
||||
log "stub tunnelbroker API on $HE_ADDR:80"
|
||||
printf '%s' "$API_PY" | dhcp_isp "cat > /tmp/he-sim-api.py"
|
||||
# Launch from a script FILE, not an inline ssh command. The remote login
|
||||
# shell is vbash, and a multi-line inlined `sudo setsid nohup ... &` through
|
||||
# it silently ran nothing at all: no process, no /run/he-sim-api.out, and a
|
||||
# `pgrep -f he-sim-api.py` status check that reported "running" because the
|
||||
# unbracketed pattern matched its OWN ssh command line. Two self-inflicted
|
||||
# illusions stacked on each other.
|
||||
#
|
||||
# This repo already learned this once -- see isp_session_control() in
|
||||
# labsim-pppoe-ha-test.sh, where driving vbash inline made every iteration
|
||||
# of the T4 matrix test the wrong policy while printing the right one.
|
||||
printf '%s\n' \
|
||||
'#!/bin/sh' \
|
||||
'# started detached so it outlives the ssh session that launched it' \
|
||||
'pkill -f "he-sim-api[.]py" 2>/dev/null' \
|
||||
'rm -f /run/he-sim-api.log /run/he-sim-api.out' \
|
||||
"exec setsid python3 /tmp/he-sim-api.py $HE_ADDR '$HE_LINK6/64' $RT_LINK6 $SITE6 >/run/he-sim-api.out 2>&1 </dev/null &" \
|
||||
| dhcp_isp "cat > /tmp/he-sim-start.sh"
|
||||
dhcp_isp "chmod +x /tmp/he-sim-start.sh && sudo /tmp/he-sim-start.sh" >/dev/null
|
||||
# Poll for the bind rather than sleeping a guessed interval.
|
||||
local i probe=""
|
||||
for i in $(seq 1 10); do
|
||||
sleep 1
|
||||
probe="$(dhcp_isp "curl -sS --max-time 3 'http://$HE_ADDR/nic/update' 2>&1")"
|
||||
[ "$probe" = nohost ] && break
|
||||
done
|
||||
case "$probe" in
|
||||
nohost) log "API answering (returned 'nohost' for a call with no myip -- correct)" ;;
|
||||
*) die "stub API not answering on $HE_ADDR:80 (got: ${probe:-<nothing>})" ;;
|
||||
esac
|
||||
|
||||
log "up. Router side is NOT configured by this script -- that is real VyOS"
|
||||
log "config and belongs to the matrix; see labsim-ipv6-ha-test.sh --setup."
|
||||
}
|
||||
|
||||
down() {
|
||||
log "tearing down"
|
||||
dhcp_isp "sudo pkill -f 'he-sim-api[.]py' 2>/dev/null;
|
||||
sudo ip tunnel del he-sim 2>/dev/null;
|
||||
sudo ip link del he-lo 2>/dev/null;
|
||||
sudo ip route del $PPPOE_NET via $PPPOE_ISP 2>/dev/null" >/dev/null
|
||||
pppoe_isp "sudo ip route del $HE_ADDR/32 via $DHCP_ISP 2>/dev/null" >/dev/null
|
||||
log "down"
|
||||
}
|
||||
|
||||
status() {
|
||||
printf ' HE address : %s\n' "$(dhcp_isp "ip -4 -br addr show he-lo 2>/dev/null | awk '{print \$3}'" || echo '<absent>')"
|
||||
printf ' he-sim : %s\n' "$(dhcp_isp "ip tunnel show he-sim 2>/dev/null" || echo '<absent>')"
|
||||
printf ' he-sim v6 : %s\n' "$(dhcp_isp "ip -6 -br addr show he-sim 2>/dev/null | awk '{print \$3}'" || echo '-')"
|
||||
# Bracketed so the pattern cannot match the ssh command line carrying it --
|
||||
# unbracketed, this reported "running" while nothing was listening at all.
|
||||
printf ' API : %s\n' "$(dhcp_isp "pgrep -f 'he-sim-api[.]py' >/dev/null && echo running || echo stopped")"
|
||||
printf ' API bound : %s\n' "$(dhcp_isp "curl -sS --max-time 3 'http://$HE_ADDR/nic/update' 2>/dev/null" || echo 'NOT ANSWERING')"
|
||||
printf ' API calls : %s\n' "$(dhcp_isp "grep -c . /run/he-sim-api.log 2>/dev/null" || echo 0)"
|
||||
printf ' route back : %s\n' "$(dhcp_isp "ip route show $PPPOE_NET 2>/dev/null" || echo '<none>')"
|
||||
}
|
||||
|
||||
# Count of endpoint-CHANGING calls. `nochg` does not count: production's
|
||||
# he-tunnel-follow re-sends the same address happily and HE treats it as a
|
||||
# no-op, so only a real move is evidence that something re-pointed the tunnel.
|
||||
calls() { dhcp_isp "grep -c ' good ' /run/he-sim-api.log 2>/dev/null" | tr -d ' \n'; }
|
||||
|
||||
# Point the endpoint at an address WITHOUT going through the API, and without
|
||||
# counting as an API call.
|
||||
#
|
||||
# Needed because rebuilding the sim endpoint resets its remote, while the
|
||||
# routers' he-tunnel-follow still reads "in sync" and therefore never re-asserts
|
||||
# -- it only calls HE when its own LOCAL source changes, and has no way to learn
|
||||
# that the far end drifted. The real HE does not forget, so this is sim
|
||||
# bookkeeping, not a behaviour production needs. Keeping it out of the call
|
||||
# counter is the point: the matrix asserts on that counter.
|
||||
point() {
|
||||
local ip="$1"
|
||||
[ -n "$ip" ] || die "usage: $0 point <ipv4>"
|
||||
dhcp_isp "sudo ip tunnel change he-sim mode sit local $HE_ADDR remote $ip 2>/dev/null \
|
||||
|| { sudo ip tunnel del he-sim 2>/dev/null;
|
||||
sudo ip tunnel add he-sim mode sit local $HE_ADDR remote $ip ttl 64;
|
||||
sudo ip link set he-sim up mtu 1480;
|
||||
sudo ip -6 addr replace $HE_LINK6/64 dev he-sim;
|
||||
sudo ip -6 route replace $SITE6 via $RT_LINK6 dev he-sim; }"
|
||||
local got
|
||||
got="$(dhcp_isp "ip tunnel show he-sim 2>/dev/null | sed -nE 's/.* remote ([0-9.]+).*/\\1/p'" | tr -d ' \n')"
|
||||
[ "$got" = "$ip" ] || die "endpoint still points at ${got:-nothing}, wanted $ip"
|
||||
log "endpoint now points at $ip"
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
up) up ;;
|
||||
down) down ;;
|
||||
status) status ;;
|
||||
calls) calls; echo ;;
|
||||
point) point "${2:-}" ;;
|
||||
*) die "usage: $0 {up|down|status|calls}" ;;
|
||||
esac
|
||||
430
labsim/labsim-ipv6-ha-test.sh
Executable file
430
labsim/labsim-ipv6-ha-test.sh
Executable file
@@ -0,0 +1,430 @@
|
||||
#!/bin/bash
|
||||
# Does IPv6 follow VRRP mastership, and does it do so WITHOUT touching HE?
|
||||
#
|
||||
# The WAN became HA on 2026-09-06 and IPv6 did not follow it. Nothing caught
|
||||
# that, because nothing tested it: wan-drill measured IPv4 only, and PPPOE-HA.md
|
||||
# recorded "IPv6 stayed up at 15.5ms" from a reading taken outside the failover
|
||||
# window. This is the matrix that would have caught it.
|
||||
#
|
||||
# TWO INVARIANTS, checked independently of any individual test:
|
||||
#
|
||||
# 1. At most ONE router ever has a live tunnel. An UP tunnel on a box that
|
||||
# does not own the source address is not harmless -- see V4, it is a
|
||||
# BLACKHOLE that will happily attract the v6 default route.
|
||||
# 2. A ROUTER-level failover calls the HE API ZERO times. The 10 gig address
|
||||
# is bound to a cloned MAC and follows the VIP to the other box unchanged,
|
||||
# so there is nothing to tell HE. A non-zero count means something
|
||||
# re-pointed the tunnel at a PPPoE address -- which the ISP re-issues on
|
||||
# every dial, so it would be wrong within minutes.
|
||||
#
|
||||
# KNOWN SIM GAP, 2026-09-06 -- read this before believing a v6_online failure.
|
||||
# The MECHANISM is proven here: tun0 up on the master and down on the backup,
|
||||
# radvd following mastership, he-tunnel-follow's master guard, its hysteresis,
|
||||
# its HE call and the 1480->1472 MTU switch, and VLAN 9 hosts autoconfiguring
|
||||
# from the RA (observed: real SLAAC traffic from 2001:db8:187e:9::/64 arriving
|
||||
# at the endpoint encapsulated).
|
||||
#
|
||||
# What is NOT yet proven is the end-to-end v6 DATAPATH, because the sim's
|
||||
# "internet" is asymmetric: 6in4 packets from the PPPoE island reach the
|
||||
# endpoint with an outer source of 192.168.122.1 -- the libvirt host's NAT --
|
||||
# so HE's replies go back to the tunnel remote by a path with no NAT state and
|
||||
# are lost. Forward works, return does not.
|
||||
#
|
||||
# That is a topology fault in the scaffold, not in the thing under test. Fixing
|
||||
# it means giving the two ISP islands a real transit path that does not traverse
|
||||
# libvirt NAT. Until then, treat v6_online failures as UNPROVEN rather than as
|
||||
# evidence the design is wrong -- and do not let that ambiguity leak into
|
||||
# production sign-off, which is exactly the mistake wan-drill made by asserting
|
||||
# "IPv6 stayed up" from a reading taken outside the window.
|
||||
#
|
||||
# Requires the fake HE endpoint: ./labsim-he-endpoint.sh up
|
||||
#
|
||||
# ./labsim-ipv6-ha-test.sh --setup configure the router side (once)
|
||||
# ./labsim-ipv6-ha-test.sh --list
|
||||
# ./labsim-ipv6-ha-test.sh V1
|
||||
# ./labsim-ipv6-ha-test.sh --all
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
R1="${R1:-172.31.1.252}"; R2="${R2:-172.31.1.253}"
|
||||
VIP="${VIP:-172.31.1.1}"
|
||||
LAN9="${LAN9:-172.31.9.10}" # VLAN 9 client, for RA tests
|
||||
HE_ADDR="${HE_ADDR:-192.0.2.10}"
|
||||
HE_LINK6="${HE_LINK6:-2001:db8:1f1c:f6::1}"
|
||||
RT_LINK6="${RT_LINK6:-2001:db8:1f1c:f6::2}"
|
||||
V9_PREFIX="${V9_PREFIX:-2001:db8:187e:9}"
|
||||
# Mirrors production: one MAC, one lease, whichever router holds the VIP.
|
||||
WAN_MAC="${WAN_MAC:-02:9f:c2:12:9b:4f}"
|
||||
PW="${VYOS_PW:-vyos}"; LANPW="${LANPW:-labsim}"
|
||||
EVID="$SCRIPT_DIR/ipv6-ha-evidence"
|
||||
HE="$SCRIPT_DIR/labsim-he-endpoint.sh"
|
||||
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6 -o PreferredAuthentications=password)
|
||||
r() { timeout 45 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$1" "${@:2}" 2>/dev/null; }
|
||||
# The Alpine LAN VMs do not offer `password` auth -- reusing the routers' option
|
||||
# set makes ssh exit 255 before running anything, which reads as "the network is
|
||||
# broken". Same trap as lan() in labsim-pppoe-ha-test.sh.
|
||||
LAN_SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6)
|
||||
lan9() { timeout 30 sshpass -p "$LANPW" ssh "${LAN_SSH[@]}" "root@$LAN9" "$@" 2>/dev/null; }
|
||||
|
||||
log() { printf '\033[36m==>\033[0m %s\n' "$*"; }
|
||||
pass() { printf ' \033[32mPASS\033[0m %s\n' "$*"; }
|
||||
fail() { printf ' \033[31mFAIL\033[0m %s\n' "$*"; FAILED=$((FAILED+1)); }
|
||||
warn() { printf ' \033[33mWARN\033[0m %s\n' "$*"; }
|
||||
FAILED=0
|
||||
|
||||
# --- observations ----------------------------------------------------------
|
||||
# A destroyed or unreachable router is emphatically NOT holding the tunnel, but
|
||||
# ssh returns an EMPTY string, and `[ "" = 0 ]` is false -- the pppoe matrix hung
|
||||
# on exactly this waiting for a dead box to report zero. Default everything to 0.
|
||||
tun_state() { local v; v="$(r "$1" "ip -br link show tun0 2>/dev/null | awk '{print \$2}'" | tr -d ' \n')"; echo "${v:-absent}"; }
|
||||
tun_src() { r "$1" 'ip tunnel show tun0 2>/dev/null | sed -nE "s/.* local ([0-9.]+).*/\1/p"' | tr -d ' \n'; }
|
||||
# Whichever router currently holds the 10 gig lease -- under the cloned MAC only
|
||||
# one ever does. Read rather than assumed: the address changes when the MAC does.
|
||||
tengig_addr() { local h a; for h in "$R1" "$R2"; do
|
||||
a="$(r "$h" 'ip -4 addr show bond0.53 2>/dev/null | sed -nE "s/.*inet ([0-9.]+).*/\1/p"' | tr -d ' \n')"
|
||||
[ -n "$a" ] && { echo "$a"; return; }
|
||||
done; echo ""; }
|
||||
tun_mtu() { local v; v="$(r "$1" 'cat /sys/class/net/tun0/mtu 2>/dev/null' | tr -d ' \n')"; echo "${v:-0}"; }
|
||||
holder() { for h in "$R1" "$R2"; do
|
||||
[ "$(r "$h" "ip -4 -o addr show | grep -c ' ${VIP}/'" | tr -d ' \n')" != 0 ] \
|
||||
&& { echo "$h"; return; }; done; echo none; }
|
||||
# Ask the ROUTER, not a client: a client can be answered by the wrong path.
|
||||
v6_online() { [ "$(r "$1" "ping -6 -c1 -W3 $HE_LINK6 >/dev/null 2>&1 && echo y" | tr -d ' \n')" = y ]; }
|
||||
radvd_on() { [ "$(r "$1" 'systemctl is-active radvd 2>/dev/null' | tr -d ' \n')" = active ]; }
|
||||
he_calls() { timeout 60 "$HE" calls | tr -d ' \n'; }
|
||||
|
||||
# How many routers have a tunnel that is UP. Invariant 1's numerator.
|
||||
tun_holders() { local n=0 h; for h in "$R1" "$R2"; do
|
||||
case "$(tun_state "$h")" in UP|UNKNOWN) n=$((n+1)) ;; esac
|
||||
done; echo "$n"; }
|
||||
|
||||
check_invariants() {
|
||||
local t ok=0
|
||||
t="$(tun_holders)"
|
||||
[ "${t:-0}" -le 1 ] || { fail "INVARIANT: $t routers have tun0 up"; ok=1; }
|
||||
return $ok
|
||||
}
|
||||
|
||||
save_evidence() {
|
||||
local name="$1"; local d="$EVID/$name"; mkdir -p "$d"
|
||||
{ echo "=== $(date -Is) ==="
|
||||
echo "--- HE endpoint ---"; timeout 60 "$HE" status
|
||||
for h in "$R1" "$R2"; do echo "--- $h ---"
|
||||
r "$h" 'sudo /config/vrrp-wan-reconcile --status 2>/dev/null
|
||||
ip -br link show tun0 2>/dev/null; ip tunnel show tun0 2>/dev/null
|
||||
ip -6 route show default; systemctl is-active radvd
|
||||
sudo journalctl -t vrrp-wan -t he-tunnel-follow -n 10 --no-pager'
|
||||
done; } > "$d/state.txt" 2>&1
|
||||
log "evidence -> ipv6-ha-evidence/$name/"
|
||||
}
|
||||
|
||||
# --- setup ------------------------------------------------------------------
|
||||
# Applied through REAL VyOS config, unlike the ISP-side scaffold, because "will
|
||||
# VyOS accept this?" is one of the questions being asked.
|
||||
vyos_apply() { # host, then set-lines on stdin
|
||||
local h="$1"
|
||||
{ printf '#!/bin/vbash\nsource /opt/vyatta/etc/functions/script-template\nconfigure\n'
|
||||
cat
|
||||
printf 'commit\nsave\nexit\n'
|
||||
} | timeout 90 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$h" \
|
||||
'cat > /tmp/v6-apply.sh && chmod +x /tmp/v6-apply.sh && sudo /tmp/v6-apply.sh' 2>&1 | tail -3
|
||||
# `vbash -c` never starts a config session and commit fails to stderr, which
|
||||
# a helper like this discards -- the T4 matrix in labsim-pppoe-ha-test.sh ran
|
||||
# its whole policy sweep against the default while printing the mode it
|
||||
# thought it was testing. Always read the value back.
|
||||
}
|
||||
|
||||
setup() {
|
||||
log "--setup: router-side IPv6, both routers"
|
||||
|
||||
# THE CLONED MAC, first and on its own. Production pins f0:9f:c2:12:9b:4f on
|
||||
# vif 53 so the 10 gig lease follows the VIP and the tunnel source is the
|
||||
# SAME address on either box. The sim never had it -- each router took its
|
||||
# own lease -- so the sim could not reproduce the one property the whole
|
||||
# IPv6-HA design leans on, and invariant 2 would have been untestable here.
|
||||
local h pref v9
|
||||
for h in "$R1" "$R2"; do
|
||||
log " $h: pinning the cloned WAN MAC $WAN_MAC"
|
||||
vyos_apply "$h" <<EOF
|
||||
set interfaces bonding bond0 vif 53 mac '$WAN_MAC'
|
||||
EOF
|
||||
done
|
||||
|
||||
# A new MAC means a NEW lease, so the tunnel source cannot be hardcoded --
|
||||
# discover it. Hardcoding the pre-change address here would have configured
|
||||
# every tunnel with a source neither router owns, i.e. the V4 blackhole, on
|
||||
# both boxes, while the matrix reported setup success.
|
||||
local src="" i
|
||||
for i in $(seq 1 30); do
|
||||
src="$(tengig_addr)"; [ -n "$src" ] && break
|
||||
sleep 5
|
||||
done
|
||||
[ -n "$src" ] || { fail "no router took a 10 gig lease after the MAC change -- cannot set a tunnel source"; return 1; }
|
||||
log " 10 gig lease under the cloned MAC: $src"
|
||||
|
||||
# Credentials pointing at the stub. HE_UPDATE_URL is read AFTER the secrets
|
||||
# file is sourced, so putting it here overrides the production default
|
||||
# without the script needing a sim-specific branch.
|
||||
for h in "$R1" "$R2"; do
|
||||
printf 'HE_USER=sim\nHE_UPDATE_KEY=sim\nHE_TUNNEL_ID=1\nHE_UPDATE_URL=http://%s/nic/update\n' "$HE_ADDR" \
|
||||
| timeout 30 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$h" \
|
||||
'cat > /tmp/he-secrets && sudo install -o root -g vyattacfg -m 0640 /tmp/he-secrets /config/he-secrets' >/dev/null
|
||||
done
|
||||
|
||||
for h in "$R1" "$R2"; do
|
||||
[ "$h" = "$R1" ] && { pref=high; v9=1; } || { pref=low; v9=2; }
|
||||
log " $h (RA preference $pref, bond0.9 ::${v9})"
|
||||
vyos_apply "$h" <<EOF
|
||||
set interfaces tunnel tun0 encapsulation 'sit'
|
||||
set interfaces tunnel tun0 source-address '$src'
|
||||
set interfaces tunnel tun0 remote '$HE_ADDR'
|
||||
set interfaces tunnel tun0 address '$RT_LINK6/64'
|
||||
set interfaces tunnel tun0 mtu '1480'
|
||||
set protocols static route6 ::/0 next-hop '$HE_LINK6'
|
||||
set interfaces bonding bond0 vif 9 address '${V9_PREFIX}::${v9}/64'
|
||||
set service router-advert interface bond0.9 prefix ${V9_PREFIX}::/64 preferred-lifetime '604800'
|
||||
set service router-advert interface bond0.9 link-mtu '1472'
|
||||
set service router-advert interface bond0.9 default-preference '$pref'
|
||||
set system task-scheduler task he-tunnel-follow executable path '/config/he-tunnel-follow'
|
||||
set system task-scheduler task he-tunnel-follow executable arguments 'run'
|
||||
set system task-scheduler task he-tunnel-follow interval '1m'
|
||||
EOF
|
||||
done
|
||||
# RA link-mtu is 1472, the PPPoE figure, on BOTH -- deliberately not 1480.
|
||||
# It cannot be reconciled at runtime (it needs a commit, and the tunnel plane
|
||||
# is commit-free on purpose), so advertise the lower of the two paths and be
|
||||
# correct on either WAN. Production pinned 1480 and was wrong whenever the
|
||||
# WAN fell back.
|
||||
log " setup done -- run V0 to check the baseline"
|
||||
}
|
||||
|
||||
# --- tests ------------------------------------------------------------------
|
||||
V0() { # baseline
|
||||
log "V0 baseline: the VIP holder owns the tunnel, the backup does not"
|
||||
local h o; h="$(holder)"; o=$([ "$h" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
[ "$h" = none ] && { fail "no VIP holder"; return; }
|
||||
case "$(tun_state "$h")" in UP|UNKNOWN) pass "master $h has tun0 up" ;;
|
||||
*) fail "master $h tun0 is $(tun_state "$h")" ;; esac
|
||||
case "$(tun_state "$o")" in DOWN|absent) pass "backup $o tun0 is $(tun_state "$o")" ;;
|
||||
*) fail "backup $o tun0 is $(tun_state "$o") -- it should be held down" ;; esac
|
||||
v6_online "$h" && pass "master reaches HE over v6" || fail "master has no IPv6"
|
||||
radvd_on "$h" && pass "master is advertising on VLAN 9" || fail "master radvd not running"
|
||||
radvd_on "$o" && fail "backup is ALSO advertising -- two default routers on VLAN 9" \
|
||||
|| pass "backup is not advertising"
|
||||
check_invariants
|
||||
save_evidence V0-baseline
|
||||
}
|
||||
|
||||
V1() { # clean failover: v6 follows, and HE is never called
|
||||
log "V1 clean failover: IPv6 follows, HE API untouched"
|
||||
local from to t0 before after i
|
||||
from="$(holder)"; to=$([ "$from" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
before="$(he_calls)"
|
||||
log " master=$from -> expecting $to (HE calls so far: ${before:-0})"
|
||||
t0=$(date +%s)
|
||||
r "$from" 'sudo mkdir -p /run/vrrp-wan && sudo touch /run/vrrp-wan/force-fault'
|
||||
local took=""
|
||||
for i in $(seq 1 36); do
|
||||
sleep 5
|
||||
[ "$(holder)" = "$to" ] && v6_online "$to" && { took=$(( $(date +%s) - t0 )); break; }
|
||||
done
|
||||
[ -n "$took" ] && pass "IPv6 reached $to in ${took}s" \
|
||||
|| fail "IPv6 never followed to $to within 180s"
|
||||
case "$(tun_state "$from")" in DOWN|absent) pass "$from released its tunnel" ;;
|
||||
*) fail "$from still has tun0 $(tun_state "$from") -- blackhole risk" ;; esac
|
||||
after="$(he_calls)"
|
||||
# THE invariant this test exists for.
|
||||
[ "${after:-0}" = "${before:-0}" ] \
|
||||
&& pass "HE API not called (${after:-0} total) -- the address followed the MAC" \
|
||||
|| fail "HE API called $(( ${after:-0} - ${before:-0} )) time(s) during a ROUTER failover"
|
||||
check_invariants
|
||||
save_evidence V1-clean-failover
|
||||
r "$from" 'sudo rm -f /run/vrrp-wan/force-fault'
|
||||
sleep 40
|
||||
}
|
||||
|
||||
V2() { # 10 gig down on the master: HE must be told, exactly once
|
||||
log "V2 10 gig down: he-tunnel-follow re-points the tunnel and tells HE"
|
||||
local h before after mtu i ok=no
|
||||
h="$(holder)"; before="$(he_calls)"
|
||||
local tengig; tengig="$(tengig_addr)"
|
||||
r "$h" 'sudo ip link set bond0.53 down'
|
||||
# he-tunnel-follow runs on a 1m task-scheduler with a 2-tick hysteresis, so
|
||||
# allow well past 2 minutes before calling it a failure.
|
||||
for i in $(seq 1 30); do
|
||||
sleep 10
|
||||
[ "$(tun_src "$h")" != "$tengig" ] && { ok=yes; break; }
|
||||
done
|
||||
[ "$ok" = yes ] && pass "tunnel source moved to $(tun_src "$h") after $((i*10))s" \
|
||||
|| fail "tunnel source never left the 10 gig address"
|
||||
mtu="$(tun_mtu "$h")"
|
||||
[ "$mtu" = 1472 ] && pass "MTU dropped to 1472 for the PPPoE path" \
|
||||
|| fail "MTU is $mtu, want 1472 -- large transfers will hang"
|
||||
after="$(he_calls)"
|
||||
[ "$(( ${after:-0} - ${before:-0} ))" -ge 1 ] \
|
||||
&& pass "HE API called $(( ${after:-0} - ${before:-0} )) time(s), as it must be here" \
|
||||
|| fail "HE was never told -- it still points at an address this box no longer has"
|
||||
v6_online "$h" && pass "IPv6 still up over PPPoE" || fail "IPv6 down on the PPPoE path"
|
||||
save_evidence V2-tengig-down
|
||||
r "$h" 'sudo ip link set bond0.53 up'
|
||||
sleep 60
|
||||
}
|
||||
|
||||
V3() { # a cold backup must not advertise, dial, or blackhole
|
||||
log "V3 cold backup: no tunnel, no RA, no HE call"
|
||||
local h o before after
|
||||
h="$(holder)"; o=$([ "$h" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
before="$(he_calls)"
|
||||
r "$o" 'sudo systemctl restart vrrp-wan-reconcile.service' >/dev/null
|
||||
sleep 20
|
||||
case "$(tun_state "$o")" in DOWN|absent) pass "backup tunnel stays $(tun_state "$o")" ;;
|
||||
*) fail "backup brought tun0 up while not holding the VIP" ;; esac
|
||||
radvd_on "$o" && fail "backup is advertising on VLAN 9" || pass "backup is silent on VLAN 9"
|
||||
after="$(he_calls)"
|
||||
[ "${after:-0}" = "${before:-0}" ] && pass "backup made no HE call" \
|
||||
|| fail "the BACKUP called the HE API -- it would point HE at its own idle line"
|
||||
save_evidence V3-cold-backup
|
||||
}
|
||||
|
||||
V4() { # the assumption the production override was built on
|
||||
log "V4 a tunnel whose source-address is absent: what does VyOS actually do?"
|
||||
# The production override says such a tunnel "would simply stay down", and
|
||||
# treats that as the reason it was safe to leave IPv6 single-homed. Measured
|
||||
# on the sim backup 2026-09-06: the commit SUCCEEDS and the link comes up
|
||||
# anyway -- it is a blackhole, not an inert node. That is why the runtime
|
||||
# gate is load-bearing rather than a nicety, exactly like the PPPoE gate.
|
||||
local o h; h="$(holder)"; o=$([ "$h" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
r "$o" 'sudo ip link set tun0 up' >/dev/null; sleep 3
|
||||
case "$(tun_state "$o")" in
|
||||
UP|UNKNOWN) pass "confirmed: VyOS leaves it UP with no source address (blackhole)" ;;
|
||||
*) warn "this VyOS version keeps it $(tun_state "$o") -- the override's assumption holds here; re-check the production version before relying on it" ;;
|
||||
esac
|
||||
v6_online "$o" && fail "the backup somehow reached HE -- two live tunnels" \
|
||||
|| pass "and it carries nothing, as expected"
|
||||
# Hand it straight back to the reconciler rather than leaving it up.
|
||||
r "$o" 'sudo systemctl restart vrrp-wan-reconcile.service' >/dev/null
|
||||
sleep 15
|
||||
case "$(tun_state "$o")" in DOWN|absent) pass "the reconciler put it back down" ;;
|
||||
*) fail "the reconciler did NOT re-close the gate -- this is the load-bearing bit" ;; esac
|
||||
save_evidence V4-absent-source-address
|
||||
}
|
||||
|
||||
V5() { # never two live tunnels, even mid-transition
|
||||
log "V5 both routers momentarily master: never two live tunnels"
|
||||
local from to i worst=0 n
|
||||
from="$(holder)"; to=$([ "$from" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
r "$from" 'sudo mkdir -p /run/vrrp-wan && sudo touch /run/vrrp-wan/force-fault'
|
||||
# Sample THROUGH the transition rather than at the ends. The interesting
|
||||
# window is the one where both boxes briefly think they are in charge.
|
||||
for i in $(seq 1 24); do
|
||||
n="$(tun_holders)"; [ "${n:-0}" -gt "$worst" ] && worst="$n"
|
||||
sleep 5
|
||||
done
|
||||
[ "$worst" -le 1 ] && pass "at most $worst live tunnel throughout the transition" \
|
||||
|| fail "saw $worst live tunnels at once -- HE would receive two claimants"
|
||||
r "$from" 'sudo rm -f /run/vrrp-wan/force-fault'
|
||||
save_evidence V5-transition-invariant
|
||||
sleep 40
|
||||
}
|
||||
|
||||
V6() { # RA deprecation: does a VLAN 9 host drop the dead gateway?
|
||||
log "V6 RA deprecation: the client must stop using a demoted router"
|
||||
local h before
|
||||
h="$(holder)"
|
||||
before="$(lan9 'ip -6 route show default 2>/dev/null | head -1')"
|
||||
if [ -z "$before" ]; then
|
||||
warn "VLAN 9 client has no IPv6 default route -- SLAAC may not have run; skipping"
|
||||
return
|
||||
fi
|
||||
log " client default was: $before"
|
||||
r "$h" 'sudo mkdir -p /run/vrrp-wan && sudo touch /run/vrrp-wan/force-fault'
|
||||
sleep 45
|
||||
local after; after="$(lan9 'ip -6 route show default 2>/dev/null | head -1')"
|
||||
log " client default now: ${after:-<none>}"
|
||||
# radvd emits a final RA with router-lifetime 0 on a graceful stop. Either
|
||||
# the client moved to the new master or it dropped the route entirely; both
|
||||
# are correct. Still pointing at the demoted box is not.
|
||||
if [ "$after" = "$before" ]; then
|
||||
fail "client still points at the demoted router -- the farewell RA did not land"
|
||||
else
|
||||
pass "client stopped using the demoted router"
|
||||
fi
|
||||
r "$h" 'sudo rm -f /run/vrrp-wan/force-fault'
|
||||
save_evidence V6-ra-deprecation
|
||||
sleep 40
|
||||
}
|
||||
|
||||
V7() { # replay the 2026-09-06 near-miss, but slowly
|
||||
log "V7 slow WAN restore: does the hysteresis still hold?"
|
||||
# On 2026-09-06 vif53-pin-boot-disable bounced the 10 gig, he-tunnel-follow
|
||||
# ticked once and saw the PPPoE address, and vyos-failover restored the 10
|
||||
# gig 22 SECONDS before the second tick would have pushed HE at an address
|
||||
# Vodafone reissues on every dial. 22s of margin is not a safety property.
|
||||
# Here the restore is deliberately slower than the hysteresis window.
|
||||
local h before after tengig
|
||||
h="$(holder)"; before="$(he_calls)"; tengig="$(tengig_addr)"
|
||||
r "$h" 'sudo ip link set bond0.53 down'
|
||||
sleep 200 # > 2 ticks of a 1m scheduler
|
||||
r "$h" 'sudo ip link set bond0.53 up'
|
||||
sleep 90
|
||||
after="$(he_calls)"
|
||||
if [ "$(( ${after:-0} - ${before:-0} ))" -ge 1 ]; then
|
||||
warn "HE was updated $(( ${after:-0} - ${before:-0} )) time(s) and then had to move back -- this is the 2026-09-06 shape, now reproduced deliberately. Either widen HYSTERESIS or make vif53-pin-boot-disable hold he-tunnel-follow off for the bounce."
|
||||
else
|
||||
pass "no HE churn across a slow WAN bounce"
|
||||
fi
|
||||
# Whatever happened, the tunnel must end up back on the 10 gig.
|
||||
local i
|
||||
for i in $(seq 1 24); do
|
||||
[ "$(tun_src "$h")" = "$tengig" ] && break
|
||||
sleep 10
|
||||
done
|
||||
[ "$(tun_src "$h")" = "$tengig" ] && pass "tunnel returned to the 10 gig address" \
|
||||
|| fail "tunnel stuck on $(tun_src "$h") after the 10 gig came back"
|
||||
save_evidence V7-slow-restore
|
||||
}
|
||||
|
||||
preflight() {
|
||||
log "preflight"
|
||||
local rc=0
|
||||
[ "$(timeout 60 "$HE" status | grep -c 'API bound : nohost')" = 1 ] \
|
||||
|| { fail "the fake HE endpoint is not answering -- run ./labsim-he-endpoint.sh up"; rc=1; }
|
||||
local h
|
||||
for h in "$R1" "$R2"; do
|
||||
[ "$(tun_state "$h")" = absent ] \
|
||||
&& { fail "$h has no tun0 -- run --setup first"; rc=1; }
|
||||
[ "$(r "$h" 'systemctl is-active vrrp-wan-reconcile.timer')" = active ] \
|
||||
|| { fail "$h vrrp-wan-reconcile.timer not active"; rc=1; }
|
||||
# The reconciler must be the version that knows about the v6 plane, or
|
||||
# every result below measures the OLD behaviour while printing the new
|
||||
# test names -- the failure mode this repo has already been bitten by.
|
||||
r "$h" 'grep -q v6_take /config/vrrp-wan-reconcile' \
|
||||
|| { fail "$h has a vrrp-wan-reconcile with no IPv6 plane -- run migration/vrrp-wan-install"; rc=1; }
|
||||
done
|
||||
# Line the sim's endpoint up with whoever actually holds the tunnel. The real
|
||||
# HE remembers where it was pointed; a rebuilt sim endpoint does not, and
|
||||
# he-tunnel-follow will not re-assert because from its side nothing changed.
|
||||
# Does NOT count as an API call, so the invariant-2 assertions stay honest.
|
||||
local m src
|
||||
m="$(holder)"; src="$(tun_src "$m")"
|
||||
if [ -n "$src" ]; then
|
||||
timeout 60 "$HE" point "$src" >/dev/null 2>&1 \
|
||||
|| { fail "could not point the sim HE endpoint at $src"; rc=1; }
|
||||
fi
|
||||
[ "$rc" -eq 0 ] && pass "HE endpoint answering and pointed at $src, both routers have tun0 and a v6-aware reconciler"
|
||||
return $rc
|
||||
}
|
||||
|
||||
case "${1:---all}" in
|
||||
--list) echo "V0 baseline | V1 clean failover | V2 10gig-down | V3 cold backup | V4 absent source-address | V5 transition invariant | V6 RA deprecation | V7 slow restore"; exit 0 ;;
|
||||
--setup) setup; exit 0 ;;
|
||||
--all) preflight || exit 1; V0; V1; V2; V3; V4; V5; V6; V7 ;;
|
||||
*) preflight || exit 1; "$1" ;;
|
||||
esac
|
||||
|
||||
echo
|
||||
[ "$FAILED" -eq 0 ] && { echo "ALL PASS"; exit 0; }
|
||||
echo "$FAILED check(s) FAILED"; exit 1
|
||||
279
labsim/labsim-k8s-etcd.sh
Executable file
279
labsim/labsim-k8s-etcd.sh
Executable file
@@ -0,0 +1,279 @@
|
||||
#!/bin/bash
|
||||
# A 3-SERVER embedded-etcd k3s cluster in labsim, configured the way PRODUCTION
|
||||
# is -- through /etc/rancher/k3s/config.yaml rendered by labctl's own generator.
|
||||
#
|
||||
# WHY THIS EXISTS, separate from k8s-up.sh. k8s-up.sh is 1 server + 2 agents and
|
||||
# drives k3s with inline INSTALL_K3S_EXEC flags. Production is 3 control-plane
|
||||
# servers with embedded etcd, configured by config.yaml from k3s-config.ts. The
|
||||
# dual-stack conversion touches etcd quorum on all three at once and rolls a
|
||||
# config.yaml change one server at a time -- a failure mode a single-server lab
|
||||
# structurally cannot show, driven by a mechanism k8s-up.sh does not use. So this
|
||||
# harness models the real shape and the real code path, or the rehearsal is
|
||||
# theatre.
|
||||
#
|
||||
# ./labsim-k8s-etcd.sh up build the 3-server cluster
|
||||
# ./labsim-k8s-etcd.sh kubeconfig fetch ./labsim-etcd.kubeconfig
|
||||
# ./labsim-k8s-etcd.sh status node + etcd health
|
||||
# ./labsim-k8s-etcd.sh down destroy the three VMs
|
||||
# ./labsim-k8s-etcd.sh render N print the config.yaml node N would get
|
||||
#
|
||||
# The config.yaml is produced by:
|
||||
# bastion .../k3s/bin/render-config.js (the production generator, exported)
|
||||
# so if that generator changes shape, this cluster moves with it.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
source "$SCRIPT_DIR/lib.sh"
|
||||
source "$SCRIPT_DIR/ovs.sh"
|
||||
|
||||
K8S_VLAN="${K8S_VLAN:-2}"
|
||||
NET="${NET:-172.31.2}"
|
||||
FIRST_OCTET="${FIRST_OCTET:-31}" # .31/.32/.33 -- clear of k8s-up.sh's .11-.13
|
||||
SERVERS="${SERVERS:-3}"
|
||||
MEM="${MEM:-4096}"; CPUS="${CPUS:-2}"; DISK_GB="${DISK_GB:-12}"
|
||||
TOKEN="${TOKEN:-labsim-etcd-token}"
|
||||
CILIUM_VERSION="${CILIUM_VERSION:-1.19.1}"
|
||||
|
||||
DEB_URL="${DEB_URL:-https://cloud.debian.org/images/cloud/trixie/latest/debian-13-genericcloud-amd64.qcow2}"
|
||||
DEB_BASE="${DEB_BASE:-$IMG_DIR/debian-13-genericcloud-amd64.qcow2}"
|
||||
|
||||
# The production generator, compiled. Built by `npm --prefix bastion/src/modules run build`.
|
||||
RENDER="${RENDER:-$SCRIPT_DIR/../bastion/src/modules/dist/modules/k3s/bin/render-config.js}"
|
||||
|
||||
node_name() { echo "labsim-etcd$1"; }
|
||||
node_ip() { echo "${NET}.$((FIRST_OCTET + $1 - 1))"; }
|
||||
|
||||
# The audit policy the generated config.yaml references. Without the file the
|
||||
# apiserver refuses to start (audit-policy-file points at a missing path), which
|
||||
# is a silent-looking crash loop. Kept byte-identical to labctl's audit-policy.ts.
|
||||
AUDIT_POLICY='apiVersion: audit.k8s.io/v1
|
||||
kind: Policy
|
||||
rules:
|
||||
- level: Metadata
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["secrets", "configmaps"]
|
||||
- level: RequestResponse
|
||||
verbs: ["create", "update", "patch", "delete"]
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["pods", "services", "deployments"]
|
||||
- level: None
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["endpoints", "events"]
|
||||
users: ["system:kube-proxy", "system:apiserver"]
|
||||
- level: Metadata
|
||||
omitStages:
|
||||
- "RequestReceived"'
|
||||
|
||||
# Render node N's config.yaml with the production generator. Node 1 is
|
||||
# cluster-init; 2..N join as SERVERS (not agents) -- role=infra with a server
|
||||
# URL is exactly a joining etcd member, the same as production worker1/worker2.
|
||||
render_config() {
|
||||
local n="$1"
|
||||
local ip; ip="$(node_ip "$n")"
|
||||
local server1; server1="$(node_ip 1)"
|
||||
if [ "$n" -eq 1 ]; then
|
||||
ROLE=infra HOSTNAME="$(node_name 1)" IP="$ip" TLS_SANS="$ip" \
|
||||
node "$RENDER"
|
||||
else
|
||||
ROLE=infra HOSTNAME="$(node_name "$n")" IP="$ip" TLS_SANS="$ip" \
|
||||
K3S_SERVER_URL="https://${server1}:6443" K3S_TOKEN="$TOKEN" \
|
||||
node "$RENDER"
|
||||
fi
|
||||
}
|
||||
|
||||
ensure_base_image() {
|
||||
[ -f "$DEB_BASE" ] && { log "base image present"; return; }
|
||||
log "fetching Debian cloud image -> $DEB_BASE"
|
||||
sudo mkdir -p "$IMG_DIR"
|
||||
sudo curl -fsSL --retry 3 -o "${DEB_BASE}.tmp" "$DEB_URL" || die "fetch failed"
|
||||
sudo mv "${DEB_BASE}.tmp" "$DEB_BASE"
|
||||
}
|
||||
|
||||
build_seed() {
|
||||
local iso="$1" n="$2" pubkey="$3"
|
||||
local vm; vm="$(node_name "$n")"; local ip; ip="$(node_ip "$n")"
|
||||
local server1; server1="$(node_ip 1)"
|
||||
local tmp; tmp="$(mktemp -d)"
|
||||
local cfg; cfg="$(render_config "$n")"
|
||||
|
||||
cat > "$tmp/meta-data" <<EOF
|
||||
instance-id: $vm
|
||||
local-hostname: $vm
|
||||
EOF
|
||||
cat > "$tmp/network-config" <<EOF
|
||||
version: 2
|
||||
ethernets:
|
||||
enp1s0:
|
||||
match: { name: "en*" }
|
||||
addresses: [$ip/24]
|
||||
routes: [{ to: default, via: ${NET}.1 }]
|
||||
nameservers: { addresses: [8.8.8.8, 1.1.1.1] }
|
||||
EOF
|
||||
|
||||
# config.yaml and audit policy embedded via write_files, indented for YAML.
|
||||
local cfg_ind audit_ind
|
||||
cfg_ind="$(printf '%s\n' "$cfg" | sed 's/^/ /')"
|
||||
audit_ind="$(printf '%s\n' "$AUDIT_POLICY" | sed 's/^/ /')"
|
||||
|
||||
cat > "$tmp/user-data" <<EOF
|
||||
#cloud-config
|
||||
hostname: $vm
|
||||
users:
|
||||
- name: debian
|
||||
groups: [sudo]
|
||||
shell: /bin/bash
|
||||
sudo: ["ALL=(ALL) NOPASSWD:ALL"]
|
||||
lock_passwd: false
|
||||
plain_text_passwd: labsim
|
||||
ssh_authorized_keys: [ $pubkey ]
|
||||
ssh_pwauth: true
|
||||
ssh_authorized_keys: [ $pubkey ]
|
||||
|
||||
package_update: true
|
||||
packages: [curl, jq, iproute2, tcpdump, etcd-client]
|
||||
|
||||
write_files:
|
||||
- path: /etc/rancher/k3s/config.yaml
|
||||
content: |
|
||||
$cfg_ind
|
||||
- path: /etc/rancher/k3s/audit-policy.yaml
|
||||
content: |
|
||||
$audit_ind
|
||||
- path: /etc/modules-load.d/cilium.conf
|
||||
content: |
|
||||
br_netfilter
|
||||
overlay
|
||||
# The CIS sysctls the generated config's `protect-kernel-defaults: true`
|
||||
# REQUIRES -- byte-for-byte from labctl's sysctl.ts (applyCisHardening), plus
|
||||
# v6 forwarding. Without vm.overcommit_memory=1 / kernel.panic=10 /
|
||||
# kernel.panic_on_oops=1 the kubelet REFUSES to start ("invalid kernel flag"),
|
||||
# k3s exits 1 and crash-loops -- which presents downstream as etcd
|
||||
# re-initialising and the apiserver flapping, i.e. it looks like an etcd/CPU
|
||||
# problem when it is not. In production these come from install.ks.ts + this
|
||||
# operation; the sim must set them too.
|
||||
- path: /etc/sysctl.d/90-k3s-cis.conf
|
||||
content: |
|
||||
net.bridge.bridge-nf-call-iptables = 1
|
||||
net.bridge.bridge-nf-call-ip6tables = 1
|
||||
net.ipv4.ip_forward = 1
|
||||
net.ipv6.conf.all.forwarding = 1
|
||||
vm.panic_on_oom = 0
|
||||
vm.overcommit_memory = 1
|
||||
kernel.panic = 10
|
||||
kernel.panic_on_oops = 1
|
||||
fs.inotify.max_user_instances = 524288
|
||||
fs.inotify.max_user_watches = 524288
|
||||
|
||||
runcmd:
|
||||
- [ modprobe, br_netfilter ]
|
||||
- [ modprobe, overlay ]
|
||||
- [ sysctl, --system ]
|
||||
- |
|
||||
# Joining servers must wait for the cluster-init server's API, or the join
|
||||
# races etcd bootstrap and the unit backs off for minutes.
|
||||
if [ "$n" -ne 1 ]; then
|
||||
for i in \$(seq 1 90); do
|
||||
curl -sk --max-time 3 https://${server1}:6443/ping >/dev/null 2>&1 && break
|
||||
sleep 5
|
||||
done
|
||||
fi
|
||||
- |
|
||||
# INSTALL_K3S_EXEC=server (bare) -- everything else comes from config.yaml,
|
||||
# exactly as production. The token is passed via env for the join; on node 1
|
||||
# it seeds the cluster token.
|
||||
#
|
||||
# The etcd-arg tuning is LAB-ONLY (not in production's config): relaxed
|
||||
# heartbeat/election timers so etcd tolerates nested-virt scheduling jitter.
|
||||
# NB: this was NOT what fixed the first run's failure -- that was the missing
|
||||
# protect-kernel-defaults sysctls above, which crash-looped the kubelet and
|
||||
# only LOOKED like etcd instability. The tuning is kept as cheap defensive
|
||||
# insurance for a busy host; it changes nothing the conversion test
|
||||
# exercises.
|
||||
curl -sfL https://get.k3s.io | INSTALL_K3S_EXEC="server --etcd-arg=heartbeat-interval=500 --etcd-arg=election-timeout=5000" K3S_TOKEN="$TOKEN" sh -
|
||||
EOF
|
||||
|
||||
# Guard the generated YAML before building the ISO -- a bad indent in the
|
||||
# embedded config.yaml would fail on the node, minutes later and opaquely.
|
||||
python3 -c "import yaml,sys; yaml.safe_load(open(sys.argv[1]))" "$tmp/user-data" \
|
||||
|| die "generated user-data is not valid YAML for node $n"
|
||||
|
||||
sudo mkdir -p "$(dirname "$iso")"
|
||||
sudo genisoimage -quiet -output "$iso" -volid cidata -joliet -rock \
|
||||
"$tmp/user-data" "$tmp/meta-data" "$tmp/network-config"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
create_node() {
|
||||
local n="$1" pubkey="$2"
|
||||
local vm; vm="$(node_name "$n")"; local ip; ip="$(node_ip "$n")"
|
||||
if virsh_q dominfo "$vm" >/dev/null 2>&1; then
|
||||
local st; st="$(virsh_q domstate "$vm" 2>/dev/null | head -1 | tr -d '\n')"
|
||||
[ "$st" = running ] && { log "$vm already running ($ip)"; return; }
|
||||
log "$vm is $st -- starting"; virsh_q start "$vm" >/dev/null; return
|
||||
fi
|
||||
local disk="$IMG_DIR/${vm}.qcow2" seed="$IMG_DIR/${vm}-seed.iso"
|
||||
log "creating $vm ($ip, server ${n}, ${MEM}MB/${CPUS}cpu)"
|
||||
sudo qemu-img create -q -f qcow2 -F qcow2 -b "$DEB_BASE" "$disk" "${DISK_GB}G" >/dev/null
|
||||
build_seed "$seed" "$n" "$pubkey"
|
||||
sudo virt-install --connect "$LIBVIRT_URI" --name "$vm" \
|
||||
--memory "$MEM" --vcpus "$CPUS" \
|
||||
--disk "path=$disk,format=qcow2,bus=virtio" \
|
||||
--disk "path=$seed,device=cdrom" \
|
||||
--network "network=$OVS_NET,portgroup=vlan${K8S_VLAN},model=virtio" \
|
||||
--os-variant debian12 --graphics none --noautoconsole --import >/dev/null
|
||||
}
|
||||
|
||||
ssh_node() { ssh -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
|
||||
-o LogLevel=ERROR -o ConnectTimeout=8 "debian@$1" "$2" 2>/dev/null; }
|
||||
|
||||
cmd_up() {
|
||||
require_tools
|
||||
command -v genisoimage >/dev/null || die "genisoimage missing"
|
||||
[ -f "$RENDER" ] || die "render CLI not built: $RENDER (run: npm --prefix bastion/src/modules run build)"
|
||||
local pubkey; pubkey="$(find_ssh_pubkey)"
|
||||
ensure_base_image
|
||||
selected_vlans; log "ensuring OVS fabric"; ovs_up
|
||||
local n; for n in $(seq 1 "$SERVERS"); do create_node "$n" "$pubkey"; done
|
||||
echo
|
||||
log "3 servers booting. Node 1 cluster-inits; 2/3 join as etcd members."
|
||||
log "watch: ssh debian@$(node_ip 1) 'sudo k3s kubectl get nodes'"
|
||||
log "then: $0 kubeconfig && $0 status"
|
||||
}
|
||||
|
||||
cmd_kubeconfig() {
|
||||
local s1; s1="$(node_ip 1)"; local out="$SCRIPT_DIR/labsim-etcd.kubeconfig"
|
||||
ssh_node "$s1" "sudo cat /etc/rancher/k3s/k3s.yaml" | sed "s|127.0.0.1|$s1|" > "$out"
|
||||
chmod 600 "$out"; log "wrote $out"
|
||||
}
|
||||
|
||||
cmd_status() {
|
||||
local s1; s1="$(node_ip 1)"
|
||||
echo "=== nodes ==="
|
||||
ssh_node "$s1" "sudo k3s kubectl get nodes -o wide 2>/dev/null" | sed 's/^/ /'
|
||||
echo "=== etcd members (quorum needs 2 of 3) ==="
|
||||
ssh_node "$s1" 'sudo k3s kubectl get nodes -l node-role.kubernetes.io/etcd=true --no-headers 2>/dev/null | wc -l' | sed 's/^/ etcd nodes: /'
|
||||
echo "=== servicecidr (both families once dual-stack) ==="
|
||||
ssh_node "$s1" "sudo k3s kubectl get servicecidr -o jsonpath='{range .items[*]}{.metadata.name}={.spec.cidrs}{\"\\n\"}{end}' 2>/dev/null" | sed 's/^/ /'
|
||||
}
|
||||
|
||||
cmd_down() {
|
||||
local n vm
|
||||
for n in $(seq 1 "$SERVERS"); do
|
||||
vm="$(node_name "$n")"
|
||||
virsh_q destroy "$vm" >/dev/null 2>&1 || true
|
||||
virsh_q undefine "$vm" --remove-all-storage >/dev/null 2>&1 || true
|
||||
log "removed $vm"
|
||||
done
|
||||
}
|
||||
|
||||
case "${1:-up}" in
|
||||
up) cmd_up ;;
|
||||
kubeconfig) cmd_kubeconfig ;;
|
||||
status) cmd_status ;;
|
||||
down) cmd_down ;;
|
||||
render) render_config "${2:-1}" ;;
|
||||
*) die "usage: $0 {up|kubeconfig|status|down|render N}" ;;
|
||||
esac
|
||||
381
labsim/labsim-pppoe-ha-test.sh
Executable file
381
labsim/labsim-pppoe-ha-test.sh
Executable file
@@ -0,0 +1,381 @@
|
||||
#!/bin/bash
|
||||
# Does the WAN follow VRRP mastership, and does exactly ONE router ever hold the
|
||||
# ISP session?
|
||||
#
|
||||
# The question is not "did a client get internet". A client can be answered by
|
||||
# the wrong path entirely -- for months labsim-vyos's only default route was the
|
||||
# libvirt-NAT scaffold on eth2, so every "the LAN still has internet" verdict was
|
||||
# answered by eth2 rather than by the WAN under test. This script therefore
|
||||
# refuses to run while that is true, and asks its questions of the ROUTERS and
|
||||
# the ACCESS CONCENTRATOR, which cannot be answered by accident.
|
||||
#
|
||||
# The invariant, checked continuously and independently of any individual test:
|
||||
#
|
||||
# the AC never reports two `simdsl` sessions, and no two routers ever have a
|
||||
# pppoe0 interface at the same time
|
||||
#
|
||||
# A run that violates it FAILS regardless of its own verdict, because a single
|
||||
# consumer credential is the whole constraint the design exists to satisfy.
|
||||
#
|
||||
# ./labsim-pppoe-ha-test.sh --list
|
||||
# ./labsim-pppoe-ha-test.sh T3
|
||||
# ./labsim-pppoe-ha-test.sh --all
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
R1="${R1:-172.31.1.252}"; R2="${R2:-172.31.1.253}"
|
||||
ISP="${ISP:-192.168.122.63}" # the fake access concentrator
|
||||
LANVM="${LANVM:-172.31.10.10}"
|
||||
VIP="${VIP:-172.31.1.1}"
|
||||
PW="${VYOS_PW:-vyos}"; LANPW="${LANPW:-labsim}"
|
||||
EVID="$SCRIPT_DIR/wan-failover-evidence"
|
||||
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6 -o PreferredAuthentications=password)
|
||||
r() { timeout 45 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$1" "${@:2}" 2>/dev/null; }
|
||||
isp() { timeout 30 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$ISP" "$@" 2>/dev/null; }
|
||||
|
||||
# Set the AC's session policy, and PROVE it landed.
|
||||
#
|
||||
# `vbash -c 'source script-template; configure; ...; commit'` does NOT work: the
|
||||
# config session never starts and commit dies with "Invalid command: [commit]",
|
||||
# on stderr, which the isp() helper discards. The whole T4 matrix therefore ran
|
||||
# all three iterations against the accel-ppp DEFAULT while printing
|
||||
# "--- session-control=deny ---" -- it reported coverage it did not have, which
|
||||
# is worse than reporting a failure. Drive it from a real script FILE, then read
|
||||
# the value back and abort the run if it disagrees.
|
||||
isp_session_control() {
|
||||
local mode="$1"
|
||||
printf '#!/bin/vbash\nsource /opt/vyatta/etc/functions/script-template\nconfigure\nset service pppoe-server session-control %s\ncommit\nsave\nexit\n' "$mode" \
|
||||
| timeout 30 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$ISP" 'cat > /tmp/set-sc.sh && chmod +x /tmp/set-sc.sh && sudo /tmp/set-sc.sh' >/dev/null 2>&1
|
||||
local got
|
||||
got="$(isp '/opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands' \
|
||||
| sed -n "s/.*session-control '\\(.*\\)'/\\1/p")"
|
||||
if [ "$got" = "$mode" ]; then
|
||||
log " AC session-control=$mode (verified)"
|
||||
return 0
|
||||
fi
|
||||
fail "could not set AC session-control=$mode (reads '${got:-unset}') -- results would be fiction"
|
||||
return 1
|
||||
}
|
||||
# The LAN VMs are Alpine and their sshd offers keyboard-interactive, not
|
||||
# `password`. Reusing the routers' option set here made ssh exit 255 BEFORE
|
||||
# running anything, and T5 read that as "the LAN lost the internet" while a
|
||||
# tcpdump on the router showed the pings flowing out pppoe0 and the replies
|
||||
# coming back. An exit code that can mean "the network is broken" or "I could
|
||||
# not log in" is not a connectivity test.
|
||||
LAN_SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=6)
|
||||
lan() { timeout 45 sshpass -p "$LANPW" ssh "${LAN_SSH[@]}" "root@$LANVM" "$@" 2>/dev/null; }
|
||||
|
||||
# Assert on what the guest actually reported, not on ssh's exit status.
|
||||
lan_online() { [ "$(lan 'ping -c2 -W3 9.9.9.9 >/dev/null 2>&1 && echo ONLINE')" = ONLINE ]; }
|
||||
|
||||
log() { printf '\033[36m==>\033[0m %s\n' "$*"; }
|
||||
pass() { printf ' \033[32mPASS\033[0m %s\n' "$*"; }
|
||||
fail() { printf ' \033[31mFAIL\033[0m %s\n' "$*"; FAILED=$((FAILED+1)); }
|
||||
FAILED=0
|
||||
|
||||
# --- observations ----------------------------------------------------------
|
||||
ac_sessions() { isp '/opt/vyatta/bin/vyatta-op-cmd-wrapper show pppoe-server sessions' \
|
||||
| grep -c ' simdsl ' || true; }
|
||||
ac_detail() { isp '/opt/vyatta/bin/vyatta-op-cmd-wrapper show pppoe-server sessions'; }
|
||||
# A destroyed or unreachable router is emphatically NOT holding pppoe0, but ssh
|
||||
# returns an EMPTY string rather than 0 -- and `[ "" = 0 ]` is false, so a
|
||||
# hard-failover test waited for the dead box to "report" zero and hung until its
|
||||
# timeout, long after the survivor had taken over correctly. Default to 0.
|
||||
ppp_on() { local v; v="$(r "$1" 'ip -4 addr show pppoe0 2>/dev/null | grep -c inet' | tr -d ' \n')"
|
||||
echo "${v:-0}"; }
|
||||
holder() { for h in "$R1" "$R2"; do
|
||||
[ "$(r "$h" "ip -4 -o addr show | grep -c ' ${VIP}/'" | tr -d ' \n')" != 0 ] \
|
||||
&& { echo "$h"; return; }; done; echo none; }
|
||||
status() { r "$1" 'sudo /config/vrrp-wan-reconcile --status'; }
|
||||
|
||||
# How many routers currently hold a PPPoE interface. The invariant's other half.
|
||||
ppp_holders() { n=0; for h in "$R1" "$R2"; do
|
||||
[ "$(ppp_on "$h")" != 0 ] && n=$((n+1)); done; echo "$n"; }
|
||||
|
||||
warn() { printf ' \033[33mWARN\033[0m %s\n' "$*"; }
|
||||
|
||||
check_invariant() {
|
||||
local s p ok=0
|
||||
s="$(ac_sessions)"; p="$(ppp_holders)"
|
||||
|
||||
# THE invariant. Two of OUR routers dialled at once is the failure that
|
||||
# matters: one ISP account, and against a real ISP that is how you get
|
||||
# rate-limited or locked out.
|
||||
[ "${p:-0}" -le 1 ] || { fail "INVARIANT: $p routers hold pppoe0"; ok=1; }
|
||||
|
||||
# The AC's session count is a PROXY for the above, and only a valid one
|
||||
# while the AC enforces single-session. Under session-control=disable it
|
||||
# does not, so a destroyed router's session simply stays in the table and
|
||||
# the count reads 2 while exactly one live router is dialled -- which is AC
|
||||
# bookkeeping, not a double dial. Attribute it rather than failing blind:
|
||||
# only call it a violation when more than one router is ACTUALLY dialled.
|
||||
#
|
||||
# Do not silence it either. An orphaned session still occupies the single
|
||||
# slot at a real ISP, and that is exactly what made session-control=deny
|
||||
# take 148s while the survivor's dial attempts were refused.
|
||||
if [ "${s:-0}" -gt 1 ]; then
|
||||
if [ "${p:-0}" -gt 1 ]; then
|
||||
fail "INVARIANT: AC reports $s simdsl sessions AND $p routers are dialled"
|
||||
ok=1
|
||||
else
|
||||
warn "AC reports $s simdsl sessions but only ${p:-0} router is dialled -- stale session from the destroyed peer (expected where the AC does not enforce single-session; it is what a hostile AC holds against the survivor)"
|
||||
fi
|
||||
fi
|
||||
return $ok
|
||||
}
|
||||
|
||||
# --- preconditions ---------------------------------------------------------
|
||||
# The scaffold check is a hard gate, not a warning. A default route via eth2
|
||||
# means the box can reach the internet without the WAN working at all, and every
|
||||
# connectivity verdict below would be a lie.
|
||||
preflight() {
|
||||
log "preflight"
|
||||
local rc=0
|
||||
for h in "$R1" "$R2"; do
|
||||
if r "$h" 'ip route show default' | grep -q 'dev eth2'; then
|
||||
fail "$h still routes via eth2 (libvirt-NAT scaffold) -- run sim-net-config.py --drop-scaffold"
|
||||
rc=1
|
||||
fi
|
||||
if [ "$(r "$h" '[ -f /etc/systemd/system/ppp@pppoe0.service.d/10-vrrp-wan-gate.conf ] && echo y')" != y ]; then
|
||||
fail "$h is missing the ppp gate drop-in -- run migration/vrrp-wan-install"
|
||||
rc=1
|
||||
fi
|
||||
[ "$(r "$h" 'systemctl is-active vrrp-wan-guard.timer')" = active ] \
|
||||
|| { fail "$h vrrp-wan-guard.timer not active"; rc=1; }
|
||||
# A router with no peers file CANNOT dial, and says so only once in the
|
||||
# journal. Every failover result in the run would then be a false
|
||||
# negative blamed on the ISP. Check both, and check the config that
|
||||
# renders it -- an unsaved commit reverts on reboot and takes pppoe0
|
||||
# with it, which is how the sim secondary silently stopped dialling.
|
||||
if [ "$(r "$h" '[ -f /etc/ppp/peers/pppoe0 ] && echo y')" != y ]; then
|
||||
fail "$h has no /etc/ppp/peers/pppoe0 -- it cannot dial; re-commit the pppoe subtree"
|
||||
rc=1
|
||||
fi
|
||||
# The op-mode WRAPPER, not a bare `show`: over non-interactive ssh the
|
||||
# bare form is not on PATH, so this silently matched nothing and failed
|
||||
# both routers that were in fact configured correctly.
|
||||
if ! r "$h" '/opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands' 2>/dev/null | grep -q 'interfaces pppoe pppoe0 source-interface'; then
|
||||
fail "$h has no pppoe0 in config (unsaved commit lost on reboot?)"
|
||||
rc=1
|
||||
fi
|
||||
done
|
||||
[ "$rc" -eq 0 ] && pass "scaffold dropped, gate present, guard running, both can dial"
|
||||
return $rc
|
||||
}
|
||||
|
||||
settle() { # wait until exactly one router holds pppoe0, or give up
|
||||
local i
|
||||
for i in $(seq 1 "${1:-24}"); do
|
||||
[ "$(ppp_holders)" = 1 ] && return 0
|
||||
sleep 5
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
# Wait until a SPECIFIC router holds pppoe0 and the other does not.
|
||||
#
|
||||
# The obvious `settle` is wrong for a failover: "exactly one holder" is already
|
||||
# true before the handover starts, so it returns instantly and the test reports
|
||||
# that nothing moved while the handover is still in flight. Asking who holds it
|
||||
# is the only useful form of the question.
|
||||
settle_on() {
|
||||
local want="$1" other i
|
||||
other=$([ "$want" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
for i in $(seq 1 "${2:-30}"); do
|
||||
[ "$(ppp_on "$want")" != 0 ] && [ "$(ppp_on "$other")" = 0 ] && return 0
|
||||
sleep 5
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
save_evidence() {
|
||||
local name="$1"; local d="$EVID/$name"; mkdir -p "$d"
|
||||
{ echo "=== $(date -Is) ==="; echo "--- AC sessions ---"; ac_detail
|
||||
for h in "$R1" "$R2"; do echo "--- $h ---"; status "$h"
|
||||
r "$h" 'ip -4 -br addr show pppoe0 2>/dev/null; ip route show default; sudo journalctl -t vrrp-wan -n 8 --no-pager'
|
||||
done; } > "$d/state.txt" 2>&1
|
||||
log "evidence -> wan-failover-evidence/$name/"
|
||||
}
|
||||
|
||||
# --- tests -----------------------------------------------------------------
|
||||
T0() { # baseline
|
||||
log "T0 baseline: exactly one session, held by the VIP holder"
|
||||
local h s; h="$(holder)"; s="$(ac_sessions)"
|
||||
[ "$s" = 1 ] && pass "AC reports 1 session" || fail "AC reports $s sessions"
|
||||
[ "$(ppp_on "$h")" != 0 ] && pass "the VIP holder ($h) is the one dialled" \
|
||||
|| fail "VIP holder $h has no pppoe0"
|
||||
local other; other=$([ "$h" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
[ "$(ppp_on "$other")" = 0 ] && pass "the backup ($other) is not dialled" \
|
||||
|| fail "backup $other also holds pppoe0"
|
||||
save_evidence T0-baseline
|
||||
}
|
||||
|
||||
T3() { # clean, deliberate failover
|
||||
log "T3 clean failover via force-fault"
|
||||
local from to t0 t1; from="$(holder)"
|
||||
to=$([ "$from" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
log " master=$from -> expecting $to"
|
||||
t0=$(date +%s)
|
||||
r "$from" 'sudo mkdir -p /run/vrrp-wan && sudo touch /run/vrrp-wan/force-fault'
|
||||
if settle_on "$to" 30; then
|
||||
t1=$(date +%s)
|
||||
[ "$(ppp_on "$to")" != 0 ] && pass "pppoe0 moved to $to in $((t1-t0))s" \
|
||||
|| fail "pppoe0 did not move to $to"
|
||||
[ "$(ppp_on "$from")" = 0 ] && pass "$from released pppoe0" \
|
||||
|| fail "$from still holds pppoe0"
|
||||
else
|
||||
fail "never settled to exactly one pppoe0 holder"
|
||||
fi
|
||||
check_invariant
|
||||
save_evidence T3-clean-failover
|
||||
r "$from" 'sudo rm -f /run/vrrp-wan/force-fault'
|
||||
settle 30 >/dev/null
|
||||
}
|
||||
|
||||
T5() { # 10gig down -> PPPoE carries traffic
|
||||
log "T5 10 gig down on the master: traffic must survive on pppoe0"
|
||||
local h; h="$(holder)"
|
||||
r "$h" 'sudo ip link set bond0.53 down'
|
||||
sleep 20
|
||||
local via; via="$(r "$h" 'ip route show default' | head -1)"
|
||||
if echo "$via" | grep -q pppoe0; then
|
||||
pass "default route moved to pppoe0: $via"
|
||||
else
|
||||
fail "default route did not move to pppoe0: ${via:-<none>}"
|
||||
fi
|
||||
# Poll, do not sample. Judging connectivity on one ping 20s after the link
|
||||
# dropped failed while the path was still reconverging, and reported "the LAN
|
||||
# lost the internet" for a path that came back moments later. A single
|
||||
# negative sample is the least trustworthy verdict this harness can produce.
|
||||
local ok=no i
|
||||
for i in $(seq 1 12); do
|
||||
lan_online && { ok=yes; break; }
|
||||
sleep 5
|
||||
done
|
||||
[ "$ok" = yes ] && pass "LAN reaches the internet over pppoe0 (after $((i*5))s)" \
|
||||
|| fail "LAN never regained the internet with only pppoe0 up (60s)"
|
||||
save_evidence T5-tengig-down
|
||||
r "$h" 'sudo ip link set bond0.53 up'
|
||||
sleep 20
|
||||
}
|
||||
|
||||
T11() { # a blessed box with no peers file must not restart-loop
|
||||
log "T11 missing peers file must not restart-loop"
|
||||
local h; h="$(holder)"
|
||||
# COPY then remove, never move: only a commit touching the pppoe subtree
|
||||
# re-renders this file, so losing it strands the box permanently -- the gate
|
||||
# blocks every dial, systemd says "skipped because of an unmet condition
|
||||
# check" exactly once, and nothing else complains. An earlier `mv` pair did
|
||||
# exactly that to the sim secondary and cost a debugging session.
|
||||
r "$h" 'sudo cp -a /etc/ppp/peers/pppoe0 /run/peers.bak && sudo rm -f /etc/ppp/peers/pppoe0; sudo systemctl restart ppp@pppoe0'
|
||||
sleep 12
|
||||
local n; n="$(r "$h" 'systemctl show ppp@pppoe0 -p NRestarts --value')"
|
||||
[ "${n:-99}" -le 1 ] && pass "NRestarts=$n (gate refused the start)" \
|
||||
|| fail "NRestarts=$n -- restart loop is back"
|
||||
# Restore on the SAME host we broke, and prove it landed. Do not trust the
|
||||
# copy back: if it silently failed, every later test in the run would be
|
||||
# measuring a router that physically cannot dial.
|
||||
r "$h" 'sudo cp -a /run/peers.bak /etc/ppp/peers/pppoe0'
|
||||
if r "$h" 'test -f /etc/ppp/peers/pppoe0'; then
|
||||
pass "peers file restored on $h"
|
||||
else
|
||||
fail "peers file NOT restored on $h -- that router can no longer dial"
|
||||
fi
|
||||
save_evidence T11-no-peers-file
|
||||
settle 24 >/dev/null
|
||||
}
|
||||
|
||||
T8() { # lease expiry: the guard must hang up a demoted-but-unreconciled box
|
||||
log "T8 lease expiry revokes the session"
|
||||
local h; h="$(holder)"
|
||||
r "$h" 'sudo systemctl stop vrrp-wan-reconcile.timer'
|
||||
r "$h" 'sudo touch -d "-200 seconds" /run/vrrp-wan/may-dial'
|
||||
sleep 12
|
||||
[ "$(ppp_on "$h")" = 0 ] && pass "guard hung up on a stale lease" \
|
||||
|| fail "stale lease did not revoke the session"
|
||||
r "$h" 'sudo systemctl start vrrp-wan-reconcile.timer'
|
||||
save_evidence T8-lease-expiry
|
||||
settle 24 >/dev/null
|
||||
}
|
||||
|
||||
|
||||
T4() { # hard failover across all three AC session-control policies
|
||||
log "T4 hard failover (destroy the master) x session-control"
|
||||
# Vodafone's policy is unknowable from here, so prove the design survives
|
||||
# every one VyOS can express. `replace` is the accel-ppp default and the
|
||||
# friendly case; `deny` is the hostile one, where the AC refuses the second
|
||||
# session until its own dead-peer timer (lcp-echo-interval 30 x failure 3 =
|
||||
# 90s) frees the first -- which is exactly why GRACE is no longer 90.
|
||||
local mode from to vm t0 t1
|
||||
for mode in replace deny disable; do
|
||||
log " --- session-control=$mode ---"
|
||||
# Skip the iteration rather than measure the wrong policy.
|
||||
isp_session_control "$mode" || continue
|
||||
sleep 5
|
||||
from="$(holder)"; to=$([ "$from" = "$R1" ] && echo "$R2" || echo "$R1")
|
||||
vm=$([ "$from" = "$R1" ] && echo labsim-vyos || echo labsim-vyos2)
|
||||
[ "$from" = none ] && { fail "no master before $mode run"; continue; }
|
||||
log " destroying $vm (master=$from), expecting $to"
|
||||
t0=$(date +%s)
|
||||
sudo virsh destroy "$vm" >/dev/null 2>&1
|
||||
if settle_on "$to" 48; then
|
||||
t1=$(date +%s)
|
||||
pass "$mode: pppoe0 reached $to in $((t1-t0))s"
|
||||
else
|
||||
fail "$mode: $to never dialled within 240s"
|
||||
fi
|
||||
check_invariant
|
||||
save_evidence "T4-hard-failover-$mode"
|
||||
sudo virsh start "$vm" >/dev/null 2>&1
|
||||
# Re-bond. A VM restart recreates its taps under NEW names, and the OVS
|
||||
# bond keeps the old ones -- lacp dies, VLAN 1 goes with it, and the box
|
||||
# comes back reachable on some VLANs but not others. ovs_bond_router
|
||||
# detects the stale membership and rebuilds, but nothing runs it
|
||||
# automatically, so a destroy/start test must do it or the survivor
|
||||
# looks like a failover failure.
|
||||
( source "$SCRIPT_DIR/lib.sh"; source "$SCRIPT_DIR/ovs.sh"; selected_vlans
|
||||
LAG_NAME=$([ "$vm" = labsim-vyos ] && echo lag-vyos || echo lag-vyos2)
|
||||
ovs_bond_router "$vm" ) >/dev/null 2>&1
|
||||
# Give the returning box time to boot and settle as BACKUP before the
|
||||
# next iteration; it must NOT dial on the way up.
|
||||
sleep 90
|
||||
[ "$(ppp_on "$from")" = 0 ] && pass "$mode: $from did not dial on reboot" \
|
||||
|| fail "$mode: $from dialled on reboot (gate failed)"
|
||||
done
|
||||
isp_session_control replace >/dev/null
|
||||
log " AC restored to session-control=replace"
|
||||
}
|
||||
|
||||
T12() { # the flap damper must never tear down an ESTABLISHED session
|
||||
log "T12 flap holdoff must not kill a live session"
|
||||
local h; h="$(holder)"
|
||||
[ "$h" = none ] && { fail "no master to test"; return; }
|
||||
# Forge a holdoff far in the future, as a dial storm would. Before the fix
|
||||
# ppp_dial() returned here BEFORE renewing may-dial, the lease went stale,
|
||||
# and vrrp-wan-guard hung up the master's working WAN ~80s later.
|
||||
r "$h" 'sudo sh -c "echo $(( $(date +%s) + 900 )) > /run/vrrp-wan/holdoff"'
|
||||
# Sleep past LEASE_TTL (75s) so a non-renewed lease would definitely expire.
|
||||
sleep 100
|
||||
if [ "$(ppp_on "$h")" = 1 ]; then
|
||||
pass "session survived a 900s holdoff (lease still renewed)"
|
||||
else
|
||||
fail "holdoff killed the live session -- damper is tearing down the WAN"
|
||||
fi
|
||||
r "$h" 'sudo rm -f /run/vrrp-wan/holdoff /run/vrrp-wan/dials'
|
||||
save_evidence T12-holdoff-keeps-session
|
||||
settle 24 >/dev/null
|
||||
}
|
||||
|
||||
case "${1:---all}" in
|
||||
--list) echo "T0 baseline | T3 clean failover | T4 hard failover x policy | T5 10gig-down | T8 lease expiry | T11 no-peers-file | T12 holdoff-keeps-session"; exit 0 ;;
|
||||
--all) preflight || exit 1; T0; T3; T5; T8; T11; T12 ;;
|
||||
--hard) preflight || exit 1; T4 ;;
|
||||
*) preflight || exit 1; "$1" ;;
|
||||
esac
|
||||
|
||||
echo
|
||||
[ "$FAILED" -eq 0 ] && { echo "ALL PASS"; exit 0; }
|
||||
echo "$FAILED check(s) FAILED"; exit 1
|
||||
178
labsim/labsim-vlan-leak-test.sh
Executable file
178
labsim/labsim-vlan-leak-test.sh
Executable file
@@ -0,0 +1,178 @@
|
||||
#!/bin/bash
|
||||
# Does the router offer an address from the WRONG VLAN's pool?
|
||||
#
|
||||
# The fault (ISC Kea #1117, "Mix of physical and virtual interfaces (VLAN) does
|
||||
# not work"): with `dhcp-socket-type: raw`, a frame tagged for a sub-interface is
|
||||
# ALSO delivered to the PARENT's AF_PACKET socket. Kea then selects a subnet from
|
||||
# the parent's own address and answers a second time from the wrong pool. Both
|
||||
# offers race to the client and the CLIENT decides which one wins -- which is why
|
||||
# the symptom looks device-dependent and unreproducible.
|
||||
#
|
||||
# Production and this sim have the identical shape that triggers it: Management
|
||||
# is the NATIVE/untagged VLAN on `bond0` and therefore has a subnet on the
|
||||
# parent, while every other VLAN is a `bond0.<vif>` sub-interface of that same
|
||||
# bond.
|
||||
#
|
||||
# Method: make one DHCP client on a TAGGED VLAN send a DISCOVER, and capture
|
||||
# simultaneously on the parent and on the sub-interface. The verdict is not
|
||||
# "did the client get the right address" -- the client picking correctly is
|
||||
# exactly how this hid for weeks. The verdict is how many OFFERs the SERVER
|
||||
# emitted and which source addresses they carried.
|
||||
#
|
||||
# ./labsim-vlan-leak-test.sh test VLAN 3
|
||||
# ./labsim-vlan-leak-test.sh --vlan 9 test another VLAN
|
||||
# ./labsim-vlan-leak-test.sh --save before also write the raw captures to
|
||||
# vlan-leak-evidence/before/
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
ROUTER_IP="${ROUTER_IP:-172.31.1.1}"
|
||||
ROUTER_PW="${ROUTER_PW:-vyos}"
|
||||
CLIENT_PW="${CLIENT_PW:-labsim}"
|
||||
VLAN=3
|
||||
CLIENT=""
|
||||
SAVE=""
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--vlan) VLAN="$2"; shift 2 ;;
|
||||
--client) CLIENT="$2"; shift 2 ;;
|
||||
--save) SAVE="$2"; shift 2 ;;
|
||||
*) echo "usage: $0 [--vlan N] [--client IP] [--save LABEL]" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
: "${CLIENT:=172.31.${VLAN}.10}"
|
||||
|
||||
log() { printf '\033[36m==>\033[0m %s\n' "$*"; }
|
||||
die() { printf '\033[31merror:\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
command -v sshpass >/dev/null || die "sshpass required"
|
||||
|
||||
# A silent router is the one verdict worth double-checking before reporting.
|
||||
#
|
||||
# Kea can be `is-active` and answering nothing -- it reopens sockets on a retry
|
||||
# loop, and some configurations (`listen-interface`, notably) leave individual
|
||||
# VLANs dead while the rest work. Both look identical to a one-shot test: "no
|
||||
# reply at all". Two opposite and equally wrong conclusions about
|
||||
# `listen-interface` came out of believing a single negative run, in both
|
||||
# directions, before a retry made the real pattern obvious.
|
||||
#
|
||||
# Kea's fallback UDP socket appearing is NOT a readiness signal -- it is bound
|
||||
# well before the server actually answers. Checked, and it does not work.
|
||||
RETRIED="${RETRIED:-0}"
|
||||
|
||||
router() {
|
||||
timeout 40 sshpass -p "$ROUTER_PW" ssh -o StrictHostKeyChecking=no \
|
||||
-o ConnectTimeout=8 "vyos@$ROUTER_IP" "$@" 2>/dev/null
|
||||
}
|
||||
# VyOS's login shell is vbash, which returns 255 on anything it does not like --
|
||||
# in particular a backgrounded job. Feeding the script to `bash -s` on stdin
|
||||
# sidesteps vbash entirely and is the only reliable way to leave a daemon behind.
|
||||
router_sh() {
|
||||
timeout 40 sshpass -p "$ROUTER_PW" ssh -o StrictHostKeyChecking=no \
|
||||
-o ConnectTimeout=8 "vyos@$ROUTER_IP" 'bash -s' 2>/dev/null
|
||||
}
|
||||
client() {
|
||||
timeout 60 sshpass -p "$CLIENT_PW" ssh -o StrictHostKeyChecking=no \
|
||||
-o ConnectTimeout=8 "root@$CLIENT" "$@" 2>/dev/null
|
||||
}
|
||||
|
||||
# Which interfaces to watch. The parent is the whole point: after the fix it
|
||||
# should carry no DHCP traffic of its own at all.
|
||||
PARENT="bond0"
|
||||
VIF="bond0.${VLAN}"
|
||||
|
||||
log "router $ROUTER_IP -- capturing on $PARENT and $VIF"
|
||||
# Kill EVERY tcpdump first, not just ones matching this run's pattern, and count
|
||||
# only afterwards. Counting `pgrep -f 'tcpdump -i bond0'` while a stray tcpdump
|
||||
# from an earlier session was still running satisfied the >=2 guard with zero of
|
||||
# THIS run's captures alive -- and a capture that records nothing reports
|
||||
# "the router sent no reply at all", which reads as a DHCP outage. That sent me
|
||||
# chasing a fault in the router that was entirely in the test harness.
|
||||
started="$(router_sh <<EOF
|
||||
sudo pkill -x tcpdump >/dev/null 2>&1
|
||||
sleep 1
|
||||
sudo rm -f /tmp/leak-*.txt
|
||||
sudo nohup tcpdump -i $PARENT -e -nn -l 'udp port 67 or udp port 68' > /tmp/leak-parent.txt 2>/dev/null &
|
||||
sudo nohup tcpdump -i $VIF -e -nn -l 'udp port 67 or udp port 68' > /tmp/leak-vif.txt 2>/dev/null &
|
||||
sleep 3
|
||||
pgrep -c -x tcpdump
|
||||
EOF
|
||||
)"
|
||||
[ "${started:-0}" -eq 2 ] || die "capture did not start on the router (got ${started:-0}, expected exactly 2)"
|
||||
|
||||
# -s /bin/true: ask, observe the answer, apply nothing. The client's existing
|
||||
# static address is left alone, so this is safe to run against a live sim VM.
|
||||
log "client $CLIENT -- sending DISCOVER on VLAN $VLAN"
|
||||
client_out="$(client "udhcpc -n -q -f -i eth0 -s /bin/true -t 3 -T 3 2>&1")"
|
||||
[ -n "$client_out" ] || die "no response from client $CLIENT"
|
||||
|
||||
sleep 2
|
||||
router "sudo pkill -x tcpdump" >/dev/null
|
||||
parent="$(router 'sudo cat /tmp/leak-parent.txt')"
|
||||
vif="$(router 'sudo cat /tmp/leak-vif.txt')"
|
||||
|
||||
echo
|
||||
echo "--- client ---"
|
||||
echo "$client_out" | sed 's/^/ /'
|
||||
echo
|
||||
echo "--- $PARENT (parent) ---"
|
||||
echo "${parent:- (nothing)}" | sed 's/^/ /'
|
||||
echo
|
||||
echo "--- $VIF (sub-interface) ---"
|
||||
echo "${vif:- (nothing)}" | sed 's/^/ /'
|
||||
echo
|
||||
|
||||
if [ -n "$SAVE" ]; then
|
||||
d="$SCRIPT_DIR/vlan-leak-evidence/$SAVE"
|
||||
mkdir -p "$d"
|
||||
printf '%s\n' "$client_out" > "$d/client.txt"
|
||||
printf '%s\n' "$parent" > "$d/capture-parent.txt"
|
||||
printf '%s\n' "$vif" > "$d/capture-vif.txt"
|
||||
router '/opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands' \
|
||||
| grep -E 'interfaces bonding|vrrp group' > "$d/router-config.txt"
|
||||
log "evidence saved to vlan-leak-evidence/$SAVE/"
|
||||
fi
|
||||
|
||||
# --- verdict ---------------------------------------------------------------
|
||||
# Every BOOTP Reply seen anywhere, reduced to its source address. A reply whose
|
||||
# source is not this VLAN's router leg is an offer from the wrong subnet.
|
||||
replies="$(printf '%s\n%s\n' "$parent" "$vif" \
|
||||
| grep -o '[0-9.]*\.67 > [0-9.]*\.68' | awk '{print $1}' | sed 's/\.67$//' \
|
||||
| sort -u)"
|
||||
want_prefix="172.31.${VLAN}."
|
||||
|
||||
echo "=== verdict ==="
|
||||
if [ -z "$replies" ]; then
|
||||
if [ "$RETRIED" -eq 0 ]; then
|
||||
log "no reply -- retrying once in 20s before calling DHCP down"
|
||||
sleep 20; RETRIED=1 exec "$0" --vlan "$VLAN" --client "$CLIENT" ${SAVE:+--save "$SAVE"}
|
||||
fi
|
||||
echo "INCONCLUSIVE: the router sent no reply at all, twice -- DHCP is down on VLAN $VLAN"
|
||||
exit 2
|
||||
fi
|
||||
|
||||
bad=0
|
||||
while read -r src; do
|
||||
[ -z "$src" ] && continue
|
||||
case "$src" in
|
||||
"$want_prefix"*) printf ' ok offer from %s (this VLAN)\n' "$src" ;;
|
||||
*) printf ' LEAK offer from %s (WRONG subnet)\n' "$src"; bad=1 ;;
|
||||
esac
|
||||
done <<<"$replies"
|
||||
|
||||
# The parent carrying any DHCP of its own is the mechanism, not just a symptom:
|
||||
# it means the parent still has a subnet kea can match a tagged frame against.
|
||||
if printf '%s' "$parent" | grep -q 'ethertype IPv4' \
|
||||
&& printf '%s' "$parent" | grep -v 'vlan ' | grep -q '\.67 > '; then
|
||||
echo " note $PARENT emitted an UNTAGGED reply -- the parent still serves a subnet"
|
||||
fi
|
||||
|
||||
echo
|
||||
if [ "$bad" -eq 0 ]; then
|
||||
echo "PASS: only this VLAN's pool answered."
|
||||
exit 0
|
||||
fi
|
||||
echo "FAIL: the router answered from another VLAN's pool (kea #1117)."
|
||||
exit 1
|
||||
141
labsim/ovs.sh
141
labsim/ovs.sh
@@ -17,6 +17,21 @@ OVS_BR="${OVS_BR:-ovs-labsim}"
|
||||
OVS_NET="${OVS_NET:-labsim-ovs}" # libvirt network wrapping the bridge
|
||||
LAG_NAME="${LAG_NAME:-lag-vyos}"
|
||||
|
||||
# Native (untagged) VLAN on the trunks to the routers. Empty means NONE: every
|
||||
# VLAN, Management included, is tagged.
|
||||
#
|
||||
# This is not a style choice. A native VLAN is what puts a subnet on the bond
|
||||
# PARENT (`bond0`) while every other VLAN lives on a sub-interface of it. With
|
||||
# `dhcp-socket-type: raw`, kea then receives each tagged frame TWICE -- once on
|
||||
# the sub-interface and once on the parent -- and answers from the parent's pool
|
||||
# as well, so a client on VLAN 3 is offered a Management address and picks
|
||||
# whichever reply arrives first (ISC Kea #1117).
|
||||
#
|
||||
# Set LABSIM_NATIVE_VLAN=1 to restore the old shape and reproduce the bug:
|
||||
# LABSIM_NATIVE_VLAN=1 ./router-up.sh && ./labsim-vlan-leak-test.sh # FAIL
|
||||
# ./router-up.sh && ./labsim-vlan-leak-test.sh # PASS
|
||||
NATIVE_VLAN="${LABSIM_NATIVE_VLAN:-}"
|
||||
|
||||
ovs() { sudo ovs-vsctl "$@"; }
|
||||
|
||||
ovs_require() {
|
||||
@@ -25,6 +40,10 @@ ovs_require() {
|
||||
|| die "could not start openvswitch"
|
||||
}
|
||||
|
||||
# A comma-separated VLAN list, numerically sorted, for comparing two lists that
|
||||
# came from different places and need not agree on order.
|
||||
vlan_sorted() { echo "$1" | tr ',' '\n' | grep -v '^$' | sort -n | paste -sd, -; }
|
||||
|
||||
# All VLAN ids from the config, comma separated — used for trunk ports.
|
||||
vlan_id_list() {
|
||||
local ids=()
|
||||
@@ -72,18 +91,23 @@ ovs_define_libvirt_net() {
|
||||
"
|
||||
done
|
||||
|
||||
# Trunk: VLAN 1 native/untagged, everything else tagged — the production
|
||||
# shape. libvirt expresses this declaratively via nativeMode='untagged'
|
||||
# (see libvirt formatnetwork.html), so it does not need fixing up by hand.
|
||||
# It also matters functionally: LACPDUs are untagged, and a trunk with no
|
||||
# native VLAN has nowhere to put them.
|
||||
# Trunk: every VLAN tagged, and by default NO native VLAN (see NATIVE_VLAN at
|
||||
# the top of this file for why -- it is the kea #1117 fix, not tidiness).
|
||||
# libvirt expresses a native VLAN declaratively via nativeMode='untagged'
|
||||
# (see libvirt formatnetwork.html), so it needs no fixing up by hand.
|
||||
#
|
||||
# The worry that a trunk with no native VLAN has nowhere to put LACPDUs is
|
||||
# unfounded, and was tested rather than reasoned about: with vlan_mode=trunk
|
||||
# and no tag, `ovs-appctl bond/show` still reports lacp_status: negotiated
|
||||
# with both members enabled. LACPDUs are slow-protocol frames handled per
|
||||
# member, below the VLAN layer.
|
||||
local trunk=" <portgroup name='trunk'>
|
||||
<vlan trunk='yes'>
|
||||
"
|
||||
for entry in "${SELECTED[@]}"; do
|
||||
IFS=: read -r vid _n _p _r <<<"$entry"
|
||||
if [ "$vid" = "1" ]; then
|
||||
trunk+=" <tag id='1' nativeMode='untagged'/>
|
||||
if [ -n "$NATIVE_VLAN" ] && [ "$vid" = "$NATIVE_VLAN" ]; then
|
||||
trunk+=" <tag id='${vid}' nativeMode='untagged'/>
|
||||
"
|
||||
else
|
||||
trunk+=" <tag id='${vid}'/>
|
||||
@@ -116,29 +140,59 @@ ${pg}${trunk}</network>"
|
||||
ovs_bond_router() {
|
||||
local vm="$1"
|
||||
local taps
|
||||
# NB: domiflist indents its rows, so anchor on the FIELD not the line —
|
||||
# /^vnet/ silently matches nothing and the bond never gets built.
|
||||
taps="$(virsh_q domiflist "$vm" 2>/dev/null | awk '$1 ~ /^vnet/ {print $1}')"
|
||||
# Two filters, both load-bearing:
|
||||
#
|
||||
# $1 ~ /^vnet/ -- domiflist indents its rows, so anchor on the FIELD, not
|
||||
# the line. /^vnet/ silently matches nothing and the bond never gets built.
|
||||
#
|
||||
# $3 == OVS_NET -- count only the taps on the sim fabric. The primary also
|
||||
# carries a libvirt-NAT scaffold NIC (see --drop-scaffold in the README), so
|
||||
# an unfiltered count is 3, and this function's "expected 2" guard then
|
||||
# skipped the primary's bond entirely while reporting only a warning.
|
||||
taps="$(virsh_q domiflist "$vm" 2>/dev/null \
|
||||
| awk -v net="$OVS_NET" '$1 ~ /^vnet/ && $3 == net {print $1}')"
|
||||
local count; count="$(echo "$taps" | grep -c .)"
|
||||
[ "$count" -eq 2 ] || { warn "router $vm has $count tap(s), expected 2 — skipping bond"; return 1; }
|
||||
|
||||
# Already bonded? Re-runs must still reconcile the VLAN list: adding a VLAN to
|
||||
# vlans.conf and finding the bond unchanged is exactly how a VLAN silently
|
||||
# fails to reach a router -- interface present, tag missing, frames dropped by
|
||||
# the switch. Returning early here once cost real debugging time.
|
||||
if ovs list-ports "$OVS_BR" 2>/dev/null | grep -qx "$LAG_NAME"; then
|
||||
local want; want="$(vlan_id_list | tr ',' '\n' | grep -vx 1 | paste -sd, -)"
|
||||
local have; have="$(ovs get port "$LAG_NAME" trunks 2>/dev/null | tr -d '[] ')"
|
||||
if [ "$want" != "$have" ]; then
|
||||
log "bond $LAG_NAME trunk drift: [$have] -> [$want]; updating"
|
||||
ovs set port "$LAG_NAME" trunks="$want"
|
||||
else
|
||||
log "LACP bond $LAG_NAME already present, trunk correct"
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
[ "$count" -eq 2 ] \
|
||||
|| { warn "router $vm has $count tap(s) on $OVS_NET, expected 2 — skipping bond"; return 1; }
|
||||
|
||||
local t1 t2; t1="$(echo "$taps" | sed -n 1p)"; t2="$(echo "$taps" | sed -n 2p)"
|
||||
local want; want="$(vlan_id_list)"
|
||||
|
||||
# Already bonded? Re-runs must still reconcile BOTH the VLAN list and the
|
||||
# membership, and each has drawn blood:
|
||||
#
|
||||
# VLANs -- adding a VLAN to vlans.conf and finding the bond unchanged is how
|
||||
# a VLAN silently fails to reach a router: interface present, tag missing,
|
||||
# frames dropped by the switch.
|
||||
#
|
||||
# MEMBERS -- restarting the VM recreates its taps with NEW names, leaving the
|
||||
# bond holding two interfaces that no longer exist. `list-ports` still shows
|
||||
# the bond, so this early return declared success while the router's real
|
||||
# taps sat in the bridge as two INDEPENDENT ports, each carrying libvirt's
|
||||
# own portgroup VLAN config. That is how labsim-vyos2 ran for weeks with no
|
||||
# LACP at all and a native VLAN nobody had asked for -- and it is invisible
|
||||
# until you change the trunk and only one router follows.
|
||||
if ovs list-ports "$OVS_BR" 2>/dev/null | grep -qx "$LAG_NAME"; then
|
||||
local members; members="$(ovs-appctl-members)"
|
||||
if [ "$members" != "$(printf '%s\n%s' "$t1" "$t2" | sort | paste -sd, -)" ]; then
|
||||
warn "bond $LAG_NAME holds stale members [$members], VM has [$t1,$t2] — rebuilding"
|
||||
ovs --if-exists del-port "$OVS_BR" "$LAG_NAME"
|
||||
else
|
||||
# Compare as SETS. vlan_id_list yields config order (1,2,3,9,10,200,51,53)
|
||||
# while OVS returns its own sorted order, so a raw string compare reports
|
||||
# drift on every run and rewrites a trunk that was already correct.
|
||||
local have; have="$(ovs get port "$LAG_NAME" trunks 2>/dev/null | tr -d '[] ')"
|
||||
if [ "$(vlan_sorted "$want")" != "$(vlan_sorted "$have")" ]; then
|
||||
log "bond $LAG_NAME trunk drift: [$have] -> [$want]; updating"
|
||||
ovs set port "$LAG_NAME" trunks="$want"
|
||||
else
|
||||
log "LACP bond $LAG_NAME already present, trunk correct"
|
||||
fi
|
||||
ovs_set_native "$LAG_NAME"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
|
||||
log "bonding $t1 + $t2 into $LAG_NAME (LACP active, balance-tcp)"
|
||||
ovs del-port "$OVS_BR" "$t1" 2>/dev/null || true
|
||||
ovs del-port "$OVS_BR" "$t2" 2>/dev/null || true
|
||||
@@ -152,15 +206,36 @@ ovs_bond_router() {
|
||||
# LACPDUs. Falling back to active-backup brings the links up so negotiation
|
||||
# can start.
|
||||
#
|
||||
# native-untagged + tag=1 carries the untagged LACPDUs and the management
|
||||
# VLAN, matching production. libvirt's portgroup VLAN config does NOT apply
|
||||
# here — the bond is a port libvirt never created — so set it inline.
|
||||
local tagged; tagged="$(vlan_id_list | tr ',' '\n' | grep -vx 1 | paste -sd, -)"
|
||||
# The VLAN mode is set inline: libvirt's portgroup config does NOT apply here,
|
||||
# because the bond is a port libvirt never created.
|
||||
ovs add-bond "$OVS_BR" "$LAG_NAME" "$t1" "$t2" \
|
||||
lacp=active bond_mode=balance-tcp \
|
||||
vlan_mode=native-untagged tag=1 trunks="$tagged" \
|
||||
lacp=active bond_mode=balance-tcp trunks="$want" \
|
||||
-- set port "$LAG_NAME" other_config:lacp-time=fast \
|
||||
-- set port "$LAG_NAME" other_config:lacp-fallback-ab=true
|
||||
ovs_set_native "$LAG_NAME"
|
||||
}
|
||||
|
||||
# The bond's current members, sorted and comma-joined, or empty if the bond does
|
||||
# not resolve at all (which is itself the stale case worth rebuilding for).
|
||||
ovs-appctl-members() {
|
||||
sudo ovs-appctl bond/show "$LAG_NAME" 2>/dev/null \
|
||||
| awk '/^member /{gsub(/:/,"",$2); print $2}' | sort | paste -sd, -
|
||||
}
|
||||
|
||||
# Apply NATIVE_VLAN to a trunk port.
|
||||
#
|
||||
# `tag` MUST be removed, not merely left alone, when there is no native VLAN.
|
||||
# Setting vlan_mode=trunk while a stale `tag` remains looks correct in
|
||||
# `ovs-vsctl list port` -- it prints vlan_mode: trunk right next to tag: 1 --
|
||||
# but the port keeps egressing that VLAN untagged. Half an hour went into
|
||||
# "the router is ignoring the trunk change" before the tag was the answer.
|
||||
ovs_set_native() {
|
||||
local port="$1"
|
||||
if [ -n "$NATIVE_VLAN" ]; then
|
||||
ovs set port "$port" vlan_mode=native-untagged tag="$NATIVE_VLAN"
|
||||
else
|
||||
ovs set port "$port" vlan_mode=trunk -- clear port "$port" tag
|
||||
fi
|
||||
}
|
||||
|
||||
ovs_bond_status() {
|
||||
|
||||
@@ -45,6 +45,12 @@ VLANS = {
|
||||
10: ("172.31.10", 23),
|
||||
200: ("172.31.200", 24),
|
||||
}
|
||||
# VLAN 2 (k8s) also gets a ULA IPv6, so the routers can peer eBGP with the nodes
|
||||
# over IPv6 (the neighbors in sim-net-config.py K8S_NODES_V6 = fd00:2::1x). Only
|
||||
# VLAN 2 needs it for the BGP rehearsal; a ULA keeps sim traffic out of the real
|
||||
# HE /48. Routers hold ::252 / ::253 (no v6 VRRP VIP -- BGP peers the real per-box
|
||||
# address, exactly as production).
|
||||
VLANS6 = {2: ("fd00:2", 64)}
|
||||
DHCP_HA_NAME = "labsim-dhcp-pair" # must not equal either host-name
|
||||
|
||||
|
||||
@@ -60,14 +66,22 @@ def build(role: str) -> list[str]:
|
||||
|
||||
for vlan, (pfx, cidr) in VLANS.items():
|
||||
g = group(vlan)
|
||||
iface = "bond0" if vlan == 1 else f"bond0 vif {vlan}"
|
||||
# EVERY VLAN is a sub-interface, Management (VLAN 1) included. Putting
|
||||
# Management on the bare `bond0` is what gives the parent a subnet, and
|
||||
# kea then answers tagged frames from it as well as from the correct
|
||||
# sub-interface -- clients on other VLANs get offered a Management
|
||||
# address (ISC Kea #1117). See NATIVE_VLAN in ovs.sh; proven by
|
||||
# labsim-vlan-leak-test.sh.
|
||||
iface = f"bond0 vif {vlan}"
|
||||
out += [
|
||||
f"# VLAN {vlan}",
|
||||
# The node's own address replaces the .1 it used to hold directly;
|
||||
# .1 becomes the floating VIP, exactly as production will be.
|
||||
f"delete interfaces bonding {iface} address",
|
||||
f"set interfaces bonding {iface} address '{pfx}.{self_o}/{cidr}'",
|
||||
f"set high-availability vrrp group {g} interface bond0{'' if vlan == 1 else f'.{vlan}'}",
|
||||
*([f"set interfaces bonding {iface} address '{VLANS6[vlan][0]}::{self_o}/{VLANS6[vlan][1]}'"]
|
||||
if vlan in VLANS6 else []),
|
||||
f"set high-availability vrrp group {g} interface bond0.{vlan}",
|
||||
f"set high-availability vrrp group {g} vrid {vlan}",
|
||||
f"set high-availability vrrp group {g} address {pfx}.1/{cidr}",
|
||||
f"set high-availability vrrp group {g} priority {prio}",
|
||||
@@ -79,6 +93,27 @@ def build(role: str) -> list[str]:
|
||||
]
|
||||
|
||||
out += [
|
||||
"# --- WAN follows VRRP mastership ---",
|
||||
# These four hooks and the health check existed on both live sim VMs but
|
||||
# in NEITHER generator, so `sim-net-apply.sh check` reported "in sync"
|
||||
# while the mechanism under test was pure undetected drift -- exactly
|
||||
# the failure mode this file was written to end.
|
||||
#
|
||||
# The check goes on the SYNC GROUP, not per group: VyOS rejects a
|
||||
# per-group check while the group is in a sync group ("Only sync group
|
||||
# health check will be used").
|
||||
"set high-availability vrrp sync-group MAIN health-check script '/config/vrrp-wan-health'",
|
||||
"set high-availability vrrp sync-group MAIN health-check interval '5'",
|
||||
"set high-availability vrrp sync-group MAIN health-check failure-count '3'",
|
||||
# take and release both exec vrrp-wan-reconcile: one code path, asked at
|
||||
# different moments. `stop` matters as much as `backup` -- a stopped
|
||||
# keepalived is a demotion too, and without it the box would keep the
|
||||
# WAN while holding no VIPs.
|
||||
"set high-availability vrrp sync-group MAIN transition-script master '/config/vrrp-wan-take'",
|
||||
"set high-availability vrrp sync-group MAIN transition-script backup '/config/vrrp-wan-release'",
|
||||
"set high-availability vrrp sync-group MAIN transition-script fault '/config/vrrp-wan-release'",
|
||||
"set high-availability vrrp sync-group MAIN transition-script stop '/config/vrrp-wan-release'",
|
||||
"",
|
||||
"# --- DHCP high-availability ---",
|
||||
"# The thing under test: active-passive should mean exactly one OFFER.",
|
||||
"set service dhcp-server high-availability mode active-passive",
|
||||
|
||||
71
labsim/sim-net-apply.sh
Executable file
71
labsim/sim-net-apply.sh
Executable file
@@ -0,0 +1,71 @@
|
||||
#!/usr/bin/env bash
|
||||
# Apply -- or drift-check -- the labsim routing config on all four VMs.
|
||||
#
|
||||
# ./sim-net-apply.sh check what the VMs run vs what sim-net-config.py says
|
||||
# ./sim-net-apply.sh apply push the generated config over the serial console
|
||||
#
|
||||
# `check` is the one you want most of the time. The whole failure mode this
|
||||
# guards against is somebody (including me) fixing something on a VM over SSH
|
||||
# and never writing it down, so the next rebuild silently loses it.
|
||||
#
|
||||
# Applied over the serial console rather than SSH because a freshly installed
|
||||
# sim router holds the same addresses as its peer -- there is a window where it
|
||||
# is not safely reachable over the network at all. See console-apply.py.
|
||||
set -uo pipefail
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
ACTION="${1:-check}"
|
||||
WORK="$(mktemp -d)"; trap 'rm -rf "$WORK"' EXIT
|
||||
|
||||
# role : vm : address : regex selecting the subtrees this generator owns
|
||||
#
|
||||
# primary and secondary now own the SAME subtrees. The secondary's used to omit
|
||||
# `interfaces pppoe`, `interfaces bonding`, `nat source` and `protocols
|
||||
# failover|static`, so `check` was blind to precisely the WAN config the
|
||||
# failover mechanism depends on -- it reported "in sync" for a box that had no
|
||||
# WAN at all.
|
||||
TARGETS=(
|
||||
"primary:labsim-vyos:172.31.1.252:^set (protocols (bgp|failover|static)|policy (prefix-list|route-map)|nat source rule 1[12]0|interfaces (pppoe|bonding bond0 vif 5[13])|firewall (group interface-group LAN|ipv4|ipv6))"
|
||||
"secondary:labsim-vyos2:172.31.1.253:^set (protocols (bgp|failover|static)|policy (prefix-list|route-map)|nat source rule 1[12]0|interfaces (pppoe|bonding bond0 vif 5[13])|firewall (group interface-group LAN|ipv4|ipv6))"
|
||||
"isp-dhcp:labsim-isp-dhcp:192.168.122.136:^set (interfaces ethernet|nat source|service dhcp-server|firewall ipv4 forward|system host-name)"
|
||||
"isp-pppoe:labsim-isp-pppoe:192.168.122.63:^set (interfaces ethernet|nat source|service pppoe-server|firewall ipv4 forward|system host-name)"
|
||||
)
|
||||
# Sim-only credential; these VMs hold nothing real and are not reachable from
|
||||
# outside the hypervisor.
|
||||
SSH_OPTS=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o PreferredAuthentications=password -o ConnectTimeout=5)
|
||||
live() { timeout 30 sshpass -p vyos ssh "${SSH_OPTS[@]}" "vyos@$1" \
|
||||
"/opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands" 2>/dev/null; }
|
||||
# `vif 53 disable` is RUNTIME state, not config drift. The generators declare it
|
||||
# on both routers as the safe resting state (only one box may hold the cloned
|
||||
# MAC), and vrrp-wan-reconcile removes it on whichever box currently holds the
|
||||
# VIP. Comparing it would therefore report drift on the master for ever, and a
|
||||
# check that always cries wolf is a check nobody reads.
|
||||
norm() { sed "s/'//g" | grep -v 'hw-id\|offload' \
|
||||
| grep -v 'interfaces bonding bond0 vif 53 disable' | sort -u; }
|
||||
|
||||
rc=0
|
||||
for t in "${TARGETS[@]}"; do
|
||||
IFS=: read -r role vm addr rx <<<"$t"
|
||||
"$HERE/sim-net-config.py" --role "$role" >"$WORK/$role.conf" 2>/dev/null || {
|
||||
printf ' %-11s GENERATE FAILED\n' "$role"; rc=1; continue; }
|
||||
|
||||
if [ "$ACTION" = apply ]; then
|
||||
printf ' %-11s applying to %s over console...\n' "$role" "$vm"
|
||||
"$HERE/console-apply.py" --vm "$vm" --config "$WORK/$role.conf" || rc=1
|
||||
continue
|
||||
fi
|
||||
|
||||
if ! live "$addr" >"$WORK/$role.live" || [ ! -s "$WORK/$role.live" ]; then
|
||||
printf ' %-11s UNREACHABLE (%s)\n' "$role" "$addr"; rc=1; continue
|
||||
fi
|
||||
grep -E '^set ' "$WORK/$role.conf" | norm >"$WORK/$role.g"
|
||||
grep -E "$rx" "$WORK/$role.live" | norm >"$WORK/$role.l"
|
||||
if d="$(diff "$WORK/$role.g" "$WORK/$role.l")" && [ -z "$d" ]; then
|
||||
printf ' %-11s in sync (%s commands)\n' "$role" "$(wc -l <"$WORK/$role.g")"
|
||||
else
|
||||
printf ' %-11s DRIFT — "<" only in code, ">" only on the VM:\n' "$role"
|
||||
printf '%s\n' "$d" | sed 's/^/ /'
|
||||
rc=1
|
||||
fi
|
||||
done
|
||||
exit $rc
|
||||
481
labsim/sim-net-config.py
Executable file
481
labsim/sim-net-config.py
Executable file
@@ -0,0 +1,481 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate the labsim routing + WAN config: BGP, dual WAN, and the two ISP VMs.
|
||||
|
||||
`sim-ha-config.py` covers the LAN side of the sim routers (addresses, VRRP,
|
||||
conntrack-sync, DHCP). This covers everything that makes the sim a rehearsal for
|
||||
production routing rather than just a LAN:
|
||||
|
||||
* eBGP between the sim routers and the k3s nodes, carrying the service range
|
||||
* dual WAN -- DHCP on VLAN 53, PPPoE on VLAN 51 -- with health-checked failover
|
||||
* the two ISP VMs that terminate those WANs and NAT to the real internet
|
||||
|
||||
All of it previously existed only as running state on the VMs, applied by hand
|
||||
over SSH. Rebuilding a VM lost the rehearsal, and nothing recorded *why* any of
|
||||
it was shaped the way it is. That is the entire reason this file exists.
|
||||
|
||||
./sim-net-config.py --role primary > r1-net.conf
|
||||
./console-apply.py --vm labsim-vyos --config r1-net.conf
|
||||
|
||||
or apply all four at once with ./sim-net-apply.sh.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# BGP. Numbers match production so what is proven here ports over unchanged.
|
||||
# ---------------------------------------------------------------------------
|
||||
ROUTER_AS = 65000
|
||||
CLUSTER_AS = 65001
|
||||
# The service range Cilium advertises. Chosen against a survey of third-party
|
||||
# RFC1918 defaults (docker-desktop, tailscale, k3s, EKS...) so it cannot collide
|
||||
# with something we adopt later. Production uses the same /22 -- keep them equal.
|
||||
SERVICE_CIDR = "10.61.0.0/22"
|
||||
K8S_VLAN = 2
|
||||
K8S_NODES = ["172.31.2.11", "172.31.2.12", "172.31.2.13"]
|
||||
PEER_GROUP = "K8S"
|
||||
PFX_LIST = "K8S-SERVICE-IPS"
|
||||
RM_IN, RM_OUT = "K8S-IN", "K8S-OUT"
|
||||
|
||||
# IPv6 unicast. Production advertises a public Gateway LoadBalancer /64 out of
|
||||
# the HE /48 (2001:470:187e:1e00::/64) and peers over VLAN 2 IPv6. The sim proves
|
||||
# the MECHANISM -- ipv6-unicast eBGP, prefix-list6, /128 host routes, ECMP -- with
|
||||
# a ULA so nothing here can leak into the real /48. Peering is over VLAN 2 v6
|
||||
# (nodes' fd00:2::1x <-> routers' fd00:2::25x, set in sim-ha-config.py /
|
||||
# k8s-up.sh), directly connected exactly as v4.
|
||||
SERVICE_CIDR_V6 = "fd61:1e00::/64" # sim analog of prod :1e00::/64 LB pool
|
||||
K8S_NODES_V6 = ["fd00:2::11", "fd00:2::12", "fd00:2::13"]
|
||||
PEER_GROUP_V6 = "K8S6"
|
||||
PFX_LIST_V6 = "K8S-SERVICE-IPS-V6"
|
||||
RM_IN_V6, RM_OUT_V6 = "K8S-IN6", "K8S-OUT6"
|
||||
# One route per node; ECMP across all three. 4 leaves headroom for a fourth node
|
||||
# without a config change.
|
||||
MAX_PATHS = 4
|
||||
# A safety valve, not a capacity plan: a misconfigured Cilium that starts
|
||||
# advertising pod CIDRs should tear the session down, not quietly fill the FIB.
|
||||
MAX_PREFIX = 100
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Dual WAN. The sim ISPs deliberately use TEST-NET-3 (203.0.113.0/24) and
|
||||
# TEST-NET-2 (198.51.100.0/24) from RFC 5737: documentation ranges that are
|
||||
# guaranteed never to be real destinations, so a leaked sim route cannot
|
||||
# blackhole something that matters.
|
||||
# ---------------------------------------------------------------------------
|
||||
WAN_DHCP_VLAN = 53 # "10gig-equivalent" -- the primary in production
|
||||
WAN_PPPOE_VLAN = 51 # "Vodafone-equivalent" -- the backup
|
||||
ISP_DHCP_NET = "203.0.113.0/24"
|
||||
ISP_DHCP_GW = "203.0.113.1"
|
||||
ISP_DHCP_POOL = ("203.0.113.100", "203.0.113.150")
|
||||
ISP_PPPOE_NET = "198.51.100.0/24"
|
||||
ISP_PPPOE_GW = "198.51.100.1"
|
||||
ISP_PPPOE_POOL = ("198.51.100.100", "198.51.100.150")
|
||||
# Sim-only fake credentials. Both ends are in this file on purpose: they
|
||||
# authenticate nothing real, and splitting them across a secret store would make
|
||||
# the sim unreproducible for no security gain. The PRODUCTION PPPoE password
|
||||
# lives in /config/wan-secrets on the router and is never in git.
|
||||
PPPOE_USER, PPPOE_PASS = "simdsl", "simpass"
|
||||
PPPOE_MTU = 1492 # 1500 - 8 bytes of PPPoE header
|
||||
PPPOE_AC = "sim-isp"
|
||||
|
||||
# Failover probe targets. NOT 8.8.8.8/8.8.4.4: those are `system name-server`,
|
||||
# so a probe failure and a DNS failure would be the same event and the router
|
||||
# would flap the WAN every time DNS hiccuped.
|
||||
PROBE_TARGETS = ["9.9.9.9", "208.67.222.222"]
|
||||
# The bug this shape fixes (WI-8, found here, fixed in production): `ping -I
|
||||
# bond0.53` binds the SOURCE address but does not make the kernel use that
|
||||
# interface's gateway. On a cold boot where PPPoE won the default route, probes
|
||||
# for the 10 gig egressed via PPPoE, succeeded, and the 10 gig was still never
|
||||
# selected -- the house ran on the backup line silently. Pinning each target as
|
||||
# a /32 via `dhcp-interface` forces the probe onto the line being tested.
|
||||
WAN_DHCP_DISTANCE = 210 # NOT `no-default-route`, which blanks new_routers
|
||||
PPPOE_DISTANCE = 10 # in the lease file, leaving failover no gateway
|
||||
# to install and silently handing the default
|
||||
# route to the backup line.
|
||||
|
||||
# One MAC, cloned onto BOTH routers' bond0.53, mirroring production's use of the
|
||||
# retired USG's WAN2 MAC to keep its DHCP lease. The sim did not model a shared
|
||||
# MAC at all, which is exactly why bond0.53 has to stay on the config plane --
|
||||
# only VyOS config can move a MAC between boxes. Locally-administered, sim-only.
|
||||
WAN_DHCP_MAC = "02:53:10:61:00:53"
|
||||
|
||||
SIM_LAN = "172.31.0.0/16"
|
||||
|
||||
# The sim routers' own libvirt-NAT uplink, from before the ISP VMs existed. It
|
||||
# is a third default route that does not exist in production and quietly masks
|
||||
# WAN failures during a failover test. `--drop-scaffold` removes it.
|
||||
SCAFFOLD_IF = "eth2"
|
||||
SCAFFOLD_NAT_RULE = 100
|
||||
|
||||
|
||||
def bgp(role: str) -> list[str]:
|
||||
"""eBGP toward the k3s nodes. Identical on both routers except router-id."""
|
||||
octet = 252 if role == "primary" else 253
|
||||
out = [
|
||||
f"# --- BGP: AS{ROUTER_AS} <-> AS{CLUSTER_AS} (k3s/Cilium) ---",
|
||||
# FRR enforces RFC 8212: an eBGP session with no policy establishes but
|
||||
# exchanges ZERO prefixes, silently. Both directions need a policy or
|
||||
# the session looks perfectly healthy and carries nothing.
|
||||
f"set policy prefix-list {PFX_LIST} rule 10 action permit",
|
||||
f"set policy prefix-list {PFX_LIST} rule 10 prefix {SERVICE_CIDR}",
|
||||
# `le 32` because Cilium advertises individual /32 service addresses out
|
||||
# of the pool, not the aggregate.
|
||||
f"set policy prefix-list {PFX_LIST} rule 10 le 32",
|
||||
f"set policy route-map {RM_IN} rule 10 action permit",
|
||||
f"set policy route-map {RM_IN} rule 10 match ip address prefix-list {PFX_LIST}",
|
||||
# Deny everything outbound. The cluster must never learn a default route
|
||||
# from us -- Cilium would install it and blackhole pod egress.
|
||||
f"set policy route-map {RM_OUT} rule 10 action deny",
|
||||
f"set protocols bgp system-as {ROUTER_AS}",
|
||||
f"set protocols bgp parameters router-id 172.31.{K8S_VLAN}.{octet}",
|
||||
f"set protocols bgp address-family ipv4-unicast maximum-paths ebgp {MAX_PATHS}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP} remote-as {CLUSTER_AS}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP} address-family ipv4-unicast route-map import {RM_IN}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP} address-family ipv4-unicast route-map export {RM_OUT}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP} address-family ipv4-unicast maximum-prefix {MAX_PREFIX}",
|
||||
]
|
||||
out += [f"set protocols bgp neighbor {n} peer-group {PEER_GROUP}" for n in K8S_NODES]
|
||||
# --- IPv6 unicast: same policy shape, over a separate v6 peer-group ---
|
||||
# `prefix-list6` + `match ipv6 address` are the v6 spellings; `le 128` because
|
||||
# Cilium advertises each Gateway LoadBalancer address as a /128, not the
|
||||
# aggregate. Separate peer-group because the neighbors are v6 addresses; the
|
||||
# export deny is the same safety property (never hand the cluster a default).
|
||||
out += [
|
||||
"",
|
||||
f"set policy prefix-list6 {PFX_LIST_V6} rule 10 action permit",
|
||||
f"set policy prefix-list6 {PFX_LIST_V6} rule 10 prefix {SERVICE_CIDR_V6}",
|
||||
f"set policy prefix-list6 {PFX_LIST_V6} rule 10 le 128",
|
||||
f"set policy route-map {RM_IN_V6} rule 10 action permit",
|
||||
f"set policy route-map {RM_IN_V6} rule 10 match ipv6 address prefix-list {PFX_LIST_V6}",
|
||||
f"set policy route-map {RM_OUT_V6} rule 10 action deny",
|
||||
f"set protocols bgp address-family ipv6-unicast maximum-paths ebgp {MAX_PATHS}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP_V6} remote-as {CLUSTER_AS}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP_V6} address-family ipv6-unicast route-map import {RM_IN_V6}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP_V6} address-family ipv6-unicast route-map export {RM_OUT_V6}",
|
||||
f"set protocols bgp peer-group {PEER_GROUP_V6} address-family ipv6-unicast maximum-prefix {MAX_PREFIX}",
|
||||
]
|
||||
out += [f"set protocols bgp neighbor {n} peer-group {PEER_GROUP_V6}" for n in K8S_NODES_V6]
|
||||
out.append("")
|
||||
return out
|
||||
|
||||
|
||||
def wan(drop_scaffold: bool, role: str = "primary") -> list[str]:
|
||||
"""Dual WAN + health-checked failover. IDENTICAL on both routers.
|
||||
|
||||
It used to be primary-only, on the grounds that "two PPPoE clients sharing
|
||||
one credential against a single access concentrator is a different failure
|
||||
mode than anything production has". That was backwards: production has
|
||||
exactly that, and by omitting it the sim could not test the one thing most
|
||||
likely to go wrong. The secondary having no WAN is also why it ended up with
|
||||
zero NAT rules while the primary had ten -- the pair was not comparable.
|
||||
|
||||
Both routers therefore get the same WAN config. What differs is the RESTING
|
||||
STATE, and only for the DHCP line:
|
||||
|
||||
bond0.53 `disable` on BOTH. Its lease is bound to a cloned MAC, and two
|
||||
boxes claiming one MAC is the fault this whole design exists to
|
||||
prevent. vrrp-wan-reconcile removes `disable` on the master.
|
||||
pppoe0 enabled on BOTH, never `disable`d. `disable` unlinks
|
||||
/etc/ppp/peers/pppoe0, which is pppd's own options file, so the
|
||||
promotion path destroyed what it needed. Dialling is gated at
|
||||
the systemd unit instead -- see migration/ppp-vrrp-gate.conf.
|
||||
"""
|
||||
out = [
|
||||
"# --- WAN: DHCP (primary) + PPPoE (backup), health-checked ---",
|
||||
f"set interfaces bonding bond0 vif {WAN_PPPOE_VLAN} description "
|
||||
f"'WAN1 Vodafone-equivalent (sim ISP PPPoE)'",
|
||||
f"set interfaces bonding bond0 vif {WAN_DHCP_VLAN} address dhcp",
|
||||
f"set interfaces bonding bond0 vif {WAN_DHCP_VLAN} description "
|
||||
f"'WAN3 10gig-equivalent (sim ISP DHCP)'",
|
||||
f"set interfaces bonding bond0 vif {WAN_DHCP_VLAN} dhcp-options "
|
||||
f"default-route-distance {WAN_DHCP_DISTANCE}",
|
||||
# The cloned MAC. Production clones the old USG's WAN2 MAC so the ISP
|
||||
# keeps handing back the same lease; the sim did not model a shared MAC
|
||||
# at all, which is precisely why bond0.53 must stay on the config plane.
|
||||
# Modelling it lets the sim prove the lease returns to the new master.
|
||||
f"set interfaces bonding bond0 vif {WAN_DHCP_VLAN} mac {WAN_DHCP_MAC}",
|
||||
# Safe at rest on BOTH routers: only the VIP holder enables it.
|
||||
f"set interfaces bonding bond0 vif {WAN_DHCP_VLAN} disable",
|
||||
f"set interfaces pppoe pppoe0 source-interface bond0.{WAN_PPPOE_VLAN}",
|
||||
f"set interfaces pppoe pppoe0 authentication username {PPPOE_USER}",
|
||||
f"set interfaces pppoe pppoe0 authentication password {PPPOE_PASS}",
|
||||
f"set interfaces pppoe pppoe0 default-route-distance {PPPOE_DISTANCE}",
|
||||
f"set interfaces pppoe pppoe0 mtu {PPPOE_MTU}",
|
||||
# The ISP's resolvers would otherwise overwrite ours in resolv.conf every
|
||||
# time the session comes up.
|
||||
"set interfaces pppoe pppoe0 no-peer-dns",
|
||||
"",
|
||||
"# Failover: prefer the DHCP WAN, fall back to PPPoE when probes fail.",
|
||||
f"set protocols failover route 0.0.0.0/0 dhcp-interface bond0.{WAN_DHCP_VLAN} metric 1",
|
||||
f"set protocols failover route 0.0.0.0/0 dhcp-interface bond0.{WAN_DHCP_VLAN} check type icmp",
|
||||
f"set protocols failover route 0.0.0.0/0 dhcp-interface bond0.{WAN_DHCP_VLAN} check timeout 5",
|
||||
# any-available, not all: one unreachable public resolver is a normal
|
||||
# internet event, not a reason to abandon a working 10 gig line.
|
||||
f"set protocols failover route 0.0.0.0/0 dhcp-interface bond0.{WAN_DHCP_VLAN} check policy any-available",
|
||||
]
|
||||
for t in PROBE_TARGETS:
|
||||
out.append(f"set protocols failover route 0.0.0.0/0 dhcp-interface "
|
||||
f"bond0.{WAN_DHCP_VLAN} check target {t}")
|
||||
out.append("")
|
||||
out.append("# Pin the probe targets to the line under test (WI-8 -- see above).")
|
||||
for t in PROBE_TARGETS:
|
||||
out.append(f"set protocols static route {t}/32 dhcp-interface bond0.{WAN_DHCP_VLAN}")
|
||||
out += [
|
||||
"",
|
||||
"# Masquerade out of whichever WAN currently holds the default route.",
|
||||
f"set nat source rule 110 outbound-interface name bond0.{WAN_DHCP_VLAN}",
|
||||
f"set nat source rule 110 source address {SIM_LAN}",
|
||||
"set nat source rule 110 translation address masquerade",
|
||||
"set nat source rule 120 outbound-interface name pppoe0",
|
||||
f"set nat source rule 120 source address {SIM_LAN}",
|
||||
"set nat source rule 120 translation address masquerade",
|
||||
"",
|
||||
]
|
||||
if drop_scaffold:
|
||||
out += [
|
||||
"# Remove the pre-ISP-VM libvirt-NAT uplink: a third default route",
|
||||
"# that has no production equivalent and hides real WAN failures.",
|
||||
f"delete interfaces ethernet {SCAFFOLD_IF} address",
|
||||
f"delete nat source rule {SCAFFOLD_NAT_RULE}",
|
||||
"",
|
||||
]
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Firewall. The policy is: internal VLANs talk to each other and to the
|
||||
# internet; the internet initiates nothing inward.
|
||||
#
|
||||
# That was already the *effect* of the previous IPv4 ruleset, but it was built
|
||||
# as a blacklist -- `default-action accept` plus explicit drops on each WAN
|
||||
# interface. The result is identical right up until someone adds a WAN, at
|
||||
# which point it is wide open and nothing looks wrong. This is the same policy
|
||||
# expressed as a whitelist, so a new interface is closed until it is named.
|
||||
# ---------------------------------------------------------------------------
|
||||
# Management is `bond0.1`, NOT the bare `bond0`. Every VLAN is tagged and the
|
||||
# bond parent carries no subnet at all -- see NATIVE_VLAN in ovs.sh for why.
|
||||
#
|
||||
# This line is the trap in that change. The address move is the visible part and
|
||||
# the part you remember; leaving `bond0` here instead of `bond0.1` means the
|
||||
# whole Management VLAN falls outside the LAN group, and with a default-deny
|
||||
# ruleset that is every management session and all inter-VLAN routing for VLAN 1
|
||||
# dropped the instant the commit lands -- on a router you reach through itself.
|
||||
LAN_IFACES = ["bond0.1", "bond0.2", "bond0.3", "bond0.9", "bond0.10", "bond0.200"]
|
||||
LAN_GROUP = "LAN"
|
||||
|
||||
|
||||
def firewall(wan_dhcp_if: str | None = f"bond0.{WAN_DHCP_VLAN}") -> list[str]:
|
||||
"""wan_dhcp_if=None on a router with no DHCP WAN -- a firewall rule naming
|
||||
an interface that does not exist is rejected at commit."""
|
||||
out = [f"# --- firewall: LAN-to-anywhere, internet-to-nothing ---"]
|
||||
# Delete each filter before rebuilding it. `set` on a rule number is
|
||||
# ADDITIVE: if a rule 10 already exists carrying an inbound-interface
|
||||
# constraint, `set ... rule 10 state established` silently ANDs onto it,
|
||||
# and you get a stateful-accept rule that only applies to one interface
|
||||
# pair. Observed in labsim: return traffic from the internet matched
|
||||
# neither that rule nor the LAN rule and hit the default drop, so LAN
|
||||
# hosts could reach nothing outbound. Everything here is one commit, so
|
||||
# nftables is rebuilt atomically -- there is no window with no firewall.
|
||||
out += [f"delete firewall {fam} {hook} filter"
|
||||
for fam in ("ipv4", "ipv6") for hook in ("forward", "input")]
|
||||
out += [f"set firewall group interface-group {LAN_GROUP} interface {i}"
|
||||
for i in LAN_IFACES]
|
||||
out += [
|
||||
"",
|
||||
# INPUT -- traffic terminating ON the router.
|
||||
# Loopback first. Under a default-drop input policy, services talking to
|
||||
# 127.0.0.1 are filtered like anything else, and the failures are
|
||||
# bizarre and hard to attribute. Nothing off-box can forge iif lo.
|
||||
"set firewall ipv4 input filter rule 5 action accept",
|
||||
"set firewall ipv4 input filter rule 5 description 'loopback'",
|
||||
"set firewall ipv4 input filter rule 5 inbound-interface name lo",
|
||||
"set firewall ipv4 input filter rule 10 action accept",
|
||||
"set firewall ipv4 input filter rule 10 description 'established/related'",
|
||||
"set firewall ipv4 input filter rule 10 state established",
|
||||
"set firewall ipv4 input filter rule 10 state related",
|
||||
# One rule covers VRRP, conntrack-sync, kea HA, SSH, DNS and BGP,
|
||||
# because every one of them arrives on a LAN interface. Enumerating the
|
||||
# protocols instead would mean a new firewall rule every time the pair
|
||||
# gains a feature -- and a lockout the day someone forgets.
|
||||
f"set firewall ipv4 input filter rule 20 action accept",
|
||||
f"set firewall ipv4 input filter rule 20 description 'trusted LAN to the router'",
|
||||
f"set firewall ipv4 input filter rule 20 inbound-interface group {LAN_GROUP}",
|
||||
# DHCP client. Lease RENEWAL is unicast UDP to port 68 and conntrack
|
||||
# does not reliably cover it, so without this the WAN keeps working
|
||||
# until the lease expires and then dies -- a delayed failure that looks
|
||||
# nothing like a firewall change.
|
||||
"set firewall ipv4 input filter default-action drop",
|
||||
"",
|
||||
# FORWARD -- traffic passing THROUGH the router.
|
||||
"set firewall ipv4 forward filter rule 10 action accept",
|
||||
"set firewall ipv4 forward filter rule 10 description 'established/related'",
|
||||
"set firewall ipv4 forward filter rule 10 state established",
|
||||
"set firewall ipv4 forward filter rule 10 state related",
|
||||
# Inter-VLAN *and* LAN-to-internet in one rule: both are "came in on a
|
||||
# LAN interface". Deliberately no restriction between internal VLANs --
|
||||
# segmenting them is a separate decision, not a side effect of this one.
|
||||
f"set firewall ipv4 forward filter rule 20 action accept",
|
||||
f"set firewall ipv4 forward filter rule 20 description 'LAN to anywhere (inter-VLAN + internet)'",
|
||||
f"set firewall ipv4 forward filter rule 20 inbound-interface group {LAN_GROUP}",
|
||||
"set firewall ipv4 forward filter default-action drop",
|
||||
"",
|
||||
# IPv6 already runs default-deny. It only lacks the loopback rule.
|
||||
"set firewall ipv6 input filter rule 5 action accept",
|
||||
"set firewall ipv6 input filter rule 5 description 'loopback'",
|
||||
"set firewall ipv6 input filter rule 5 inbound-interface name lo",
|
||||
"set firewall ipv6 input filter rule 10 action accept",
|
||||
"set firewall ipv6 input filter rule 10 description 'replies to our own traffic'",
|
||||
"set firewall ipv6 input filter rule 10 state established",
|
||||
"set firewall ipv6 input filter rule 10 state related",
|
||||
# RFC 4890: filtering ICMPv6 wholesale breaks ND and PMTUD, which
|
||||
# presents as "IPv6 works until something large", not as a block.
|
||||
"set firewall ipv6 input filter rule 20 action accept",
|
||||
"set firewall ipv6 input filter rule 20 description 'ICMPv6 - ND/RA/PMTUD'",
|
||||
"set firewall ipv6 input filter rule 20 protocol icmpv6",
|
||||
f"set firewall ipv6 input filter rule 30 action accept",
|
||||
f"set firewall ipv6 input filter rule 30 description 'trusted LAN to the router'",
|
||||
f"set firewall ipv6 input filter rule 30 inbound-interface group {LAN_GROUP}",
|
||||
"set firewall ipv6 input filter default-action drop",
|
||||
"set firewall ipv6 forward filter rule 10 action accept",
|
||||
"set firewall ipv6 forward filter rule 10 description 'replies to our own traffic'",
|
||||
"set firewall ipv6 forward filter rule 10 state established",
|
||||
"set firewall ipv6 forward filter rule 10 state related",
|
||||
"set firewall ipv6 forward filter rule 20 action accept",
|
||||
"set firewall ipv6 forward filter rule 20 description 'ICMPv6 - ND/RA/PMTUD'",
|
||||
"set firewall ipv6 forward filter rule 20 protocol icmpv6",
|
||||
f"set firewall ipv6 forward filter rule 30 action accept",
|
||||
f"set firewall ipv6 forward filter rule 30 description 'trusted LAN interfaces only'",
|
||||
f"set firewall ipv6 forward filter rule 30 inbound-interface group {LAN_GROUP}",
|
||||
"set firewall ipv6 forward filter default-action drop",
|
||||
"",
|
||||
]
|
||||
if wan_dhcp_if:
|
||||
dhcp = [
|
||||
"set firewall ipv4 input filter rule 140 action accept",
|
||||
"set firewall ipv4 input filter rule 140 description 'DHCP client lease renewal'",
|
||||
"set firewall ipv4 input filter rule 140 protocol udp",
|
||||
"set firewall ipv4 input filter rule 140 destination port 68",
|
||||
f"set firewall ipv4 input filter rule 140 inbound-interface name {wan_dhcp_if}",
|
||||
]
|
||||
i = out.index("set firewall ipv4 input filter default-action drop")
|
||||
out[i:i] = dhcp
|
||||
return out
|
||||
|
||||
|
||||
def isp_dhcp(wan_if: str, uplink_if: str) -> list[str]:
|
||||
"""The 10gig-equivalent ISP: hands out a lease, NATs to the real internet."""
|
||||
return [
|
||||
f"# --- sim ISP: DHCP WAN on VLAN {WAN_DHCP_VLAN} ---",
|
||||
"set system host-name isp-dhcp",
|
||||
f"set interfaces ethernet {wan_if} address {ISP_DHCP_GW}/24",
|
||||
f"set interfaces ethernet {wan_if} description "
|
||||
f"'sim ISP - 10gig-equivalent WAN on VLAN{WAN_DHCP_VLAN}'",
|
||||
f"set interfaces ethernet {uplink_if} address dhcp",
|
||||
f"set interfaces ethernet {uplink_if} description 'uplink to the real internet'",
|
||||
f"set service dhcp-server shared-network-name WAN{WAN_DHCP_VLAN} "
|
||||
f"subnet {ISP_DHCP_NET} subnet-id 1",
|
||||
f"set service dhcp-server shared-network-name WAN{WAN_DHCP_VLAN} "
|
||||
f"subnet {ISP_DHCP_NET} option default-router {ISP_DHCP_GW}",
|
||||
f"set service dhcp-server shared-network-name WAN{WAN_DHCP_VLAN} "
|
||||
f"subnet {ISP_DHCP_NET} option name-server 8.8.8.8",
|
||||
f"set service dhcp-server shared-network-name WAN{WAN_DHCP_VLAN} "
|
||||
f"subnet {ISP_DHCP_NET} range CUST start {ISP_DHCP_POOL[0]}",
|
||||
f"set service dhcp-server shared-network-name WAN{WAN_DHCP_VLAN} "
|
||||
f"subnet {ISP_DHCP_NET} range CUST stop {ISP_DHCP_POOL[1]}",
|
||||
"",
|
||||
] + _isp_common(uplink_if, ISP_DHCP_NET, "sim ISP: NAT customers to the real internet")
|
||||
|
||||
|
||||
def isp_pppoe(wan_if: str, uplink_if: str) -> list[str]:
|
||||
"""The Vodafone-equivalent ISP: terminates PPPoE, NATs to the real internet."""
|
||||
return [
|
||||
f"# --- sim ISP: PPPoE WAN on VLAN {WAN_PPPOE_VLAN} ---",
|
||||
"set system host-name isp-pppoe",
|
||||
f"set interfaces ethernet {wan_if} description "
|
||||
f"'sim ISP - Vodafone-equivalent WAN on VLAN{WAN_PPPOE_VLAN} (PPPoE)'",
|
||||
f"set interfaces ethernet {uplink_if} address dhcp",
|
||||
f"set interfaces ethernet {uplink_if} description 'uplink to the real internet'",
|
||||
f"set service pppoe-server access-concentrator {PPPOE_AC}",
|
||||
f"set service pppoe-server interface {wan_if}",
|
||||
f"set service pppoe-server gateway-address {ISP_PPPOE_GW}",
|
||||
"set service pppoe-server authentication mode local",
|
||||
f"set service pppoe-server authentication local-users username {PPPOE_USER} "
|
||||
f"password {PPPOE_PASS}",
|
||||
f"set service pppoe-server client-ip-pool CUST range "
|
||||
f"{ISP_PPPOE_POOL[0]}-{ISP_PPPOE_POOL[1]}",
|
||||
"set service pppoe-server default-pool CUST",
|
||||
"set service pppoe-server name-server 8.8.8.8",
|
||||
"",
|
||||
] + _isp_common(uplink_if, ISP_PPPOE_NET, "sim ISP: NAT PPPoE customers to the real internet")
|
||||
|
||||
|
||||
def _isp_common(uplink_if: str, customer_net: str, desc: str) -> list[str]:
|
||||
return [
|
||||
"set nat source rule 100 description " + f"'{desc}'",
|
||||
f"set nat source rule 100 outbound-interface name {uplink_if}",
|
||||
f"set nat source rule 100 source address {customer_net}",
|
||||
"set nat source rule 100 translation address masquerade",
|
||||
"",
|
||||
# An ISP that drops return traffic is not simulating an ISP. The forward
|
||||
# chain defaults to accept here on purpose -- these VMs model the
|
||||
# internet, and the thing under test is the router's firewall, not this.
|
||||
"set firewall ipv4 forward filter default-action accept",
|
||||
"set firewall ipv4 forward filter rule 10 action accept",
|
||||
"set firewall ipv4 forward filter rule 10 state established",
|
||||
"set firewall ipv4 forward filter rule 10 state related",
|
||||
"set firewall ipv4 forward filter rule 10 description conntrack-engage",
|
||||
"",
|
||||
]
|
||||
|
||||
|
||||
def build(role: str, drop_scaffold: bool, wan_if: str, uplink_if: str) -> list[str]:
|
||||
if role in ("primary", "secondary"):
|
||||
# BOTH routers get the identical WAN. The sim used to give it to the
|
||||
# primary only, reasoning that two PPPoE clients sharing one credential
|
||||
# was "a different failure mode than anything production has" -- but
|
||||
# that IS production, and omitting it meant the failover path was the
|
||||
# one path the sim could not exercise. It also left the pair
|
||||
# incomparable: ten NAT rules on one box, none on the other.
|
||||
#
|
||||
# Safety comes from resting state, not from asymmetry: bond0.53 is
|
||||
# `disable`d on both (cloned MAC), pppoe0 is enabled on both but gated
|
||||
# at the systemd unit. See wan() and migration/ppp-vrrp-gate.conf.
|
||||
return ([f"# labsim routing -- {role}", ""]
|
||||
+ bgp(role) + wan(drop_scaffold, role)
|
||||
+ firewall(wan_dhcp_if=f"bond0.{WAN_DHCP_VLAN}"))
|
||||
if role == "isp-dhcp":
|
||||
return isp_dhcp(wan_if, uplink_if)
|
||||
return isp_pppoe(wan_if, uplink_if)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--role", required=True,
|
||||
choices=("primary", "secondary", "isp-dhcp", "isp-pppoe"))
|
||||
ap.add_argument("--drop-scaffold", action="store_true",
|
||||
help="also remove the pre-ISP-VM libvirt-NAT uplink (primary only)")
|
||||
# The ISP VMs' interface names depend on PCI enumeration order, which is not
|
||||
# stable across a rebuild: isp-dhcp came up as eth0/eth1 and isp-pppoe as
|
||||
# eth2/eth3 from identical XML. Check with `show interfaces` before applying
|
||||
# rather than trusting these defaults.
|
||||
ap.add_argument("--wan-if", default=None, help="ISP VM: customer-facing NIC")
|
||||
ap.add_argument("--uplink-if", default=None, help="ISP VM: internet-facing NIC")
|
||||
args = ap.parse_args()
|
||||
|
||||
defaults = {"isp-dhcp": ("eth0", "eth1"), "isp-pppoe": ("eth2", "eth3")}
|
||||
w, u = defaults.get(args.role, ("", ""))
|
||||
w, u = args.wan_if or w, args.uplink_if or u
|
||||
|
||||
if args.drop_scaffold and args.role != "primary":
|
||||
print("--drop-scaffold only applies to --role primary", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
lines = build(args.role, args.drop_scaffold, w, u)
|
||||
sys.stdout.write("\n".join(lines) + "\n")
|
||||
n = len([l for l in lines if l.startswith(("set ", "delete "))])
|
||||
print(f"{args.role}: {n} commands", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
6
labsim/vlan-leak-evidence/after-vlan1/capture-parent.txt
Normal file
6
labsim/vlan-leak-evidence/after-vlan1/capture-parent.txt
Normal file
@@ -0,0 +1,6 @@
|
||||
12:52:35.919490 52:54:00:6d:71:e7 > ff:ff:ff:ff:ff:ff, ethertype 802.1Q (0x8100), length 346: vlan 1, p 0, ethertype IPv4 (0x0800), 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:6d:71:e7, length 300
|
||||
12:52:35.920089 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype 802.1Q (0x8100), length 329: vlan 1, p 0, ethertype IPv4 (0x0800), 172.31.1.252.67 > 172.31.1.6.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:35.920307 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype 802.1Q (0x8100), length 329: vlan 1, p 0, ethertype IPv4 (0x0800), 172.31.1.252.67 > 172.31.1.7.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:35.922052 52:54:00:6d:71:e7 > ff:ff:ff:ff:ff:ff, ethertype 802.1Q (0x8100), length 346: vlan 1, p 0, ethertype IPv4 (0x0800), 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:6d:71:e7, length 300
|
||||
12:52:35.922509 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype 802.1Q (0x8100), length 329: vlan 1, p 0, ethertype IPv4 (0x0800), 172.31.1.252.67 > 172.31.1.6.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:35.923327 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype 802.1Q (0x8100), length 329: vlan 1, p 0, ethertype IPv4 (0x0800), 172.31.1.252.67 > 172.31.1.6.68: BOOTP/DHCP, Reply, length 283
|
||||
6
labsim/vlan-leak-evidence/after-vlan1/capture-vif.txt
Normal file
6
labsim/vlan-leak-evidence/after-vlan1/capture-vif.txt
Normal file
@@ -0,0 +1,6 @@
|
||||
12:52:35.919490 52:54:00:6d:71:e7 > ff:ff:ff:ff:ff:ff, ethertype IPv4 (0x0800), length 342: 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:6d:71:e7, length 300
|
||||
12:52:35.920081 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype IPv4 (0x0800), length 325: 172.31.1.252.67 > 172.31.1.6.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:35.920305 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype IPv4 (0x0800), length 325: 172.31.1.252.67 > 172.31.1.7.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:35.922052 52:54:00:6d:71:e7 > ff:ff:ff:ff:ff:ff, ethertype IPv4 (0x0800), length 342: 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:6d:71:e7, length 300
|
||||
12:52:35.922507 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype IPv4 (0x0800), length 325: 172.31.1.252.67 > 172.31.1.6.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:35.923326 52:54:00:e5:95:a2 > 52:54:00:6d:71:e7, ethertype IPv4 (0x0800), length 325: 172.31.1.252.67 > 172.31.1.6.68: BOOTP/DHCP, Reply, length 283
|
||||
4
labsim/vlan-leak-evidence/after-vlan1/client.txt
Normal file
4
labsim/vlan-leak-evidence/after-vlan1/client.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
udhcpc: started, v1.37.0
|
||||
udhcpc: broadcasting discover
|
||||
udhcpc: broadcasting select for 172.31.1.6, server 172.31.1.252
|
||||
udhcpc: lease of 172.31.1.6 obtained from 172.31.1.252, lease time 86400
|
||||
64
labsim/vlan-leak-evidence/after-vlan1/router-config.txt
Normal file
64
labsim/vlan-leak-evidence/after-vlan1/router-config.txt
Normal file
@@ -0,0 +1,64 @@
|
||||
set high-availability vrrp group native address 172.31.1.1/24
|
||||
set high-availability vrrp group native hello-source-address '172.31.1.252'
|
||||
set high-availability vrrp group native interface 'bond0.1'
|
||||
set high-availability vrrp group native no-preempt
|
||||
set high-availability vrrp group native peer-address '172.31.1.253'
|
||||
set high-availability vrrp group native priority '200'
|
||||
set high-availability vrrp group native vrid '1'
|
||||
set high-availability vrrp group vlan2 address 172.31.2.1/24
|
||||
set high-availability vrrp group vlan2 hello-source-address '172.31.2.252'
|
||||
set high-availability vrrp group vlan2 interface 'bond0.2'
|
||||
set high-availability vrrp group vlan2 no-preempt
|
||||
set high-availability vrrp group vlan2 peer-address '172.31.2.253'
|
||||
set high-availability vrrp group vlan2 priority '200'
|
||||
set high-availability vrrp group vlan2 vrid '2'
|
||||
set high-availability vrrp group vlan3 address 172.31.3.1/24
|
||||
set high-availability vrrp group vlan3 hello-source-address '172.31.3.252'
|
||||
set high-availability vrrp group vlan3 interface 'bond0.3'
|
||||
set high-availability vrrp group vlan3 no-preempt
|
||||
set high-availability vrrp group vlan3 peer-address '172.31.3.253'
|
||||
set high-availability vrrp group vlan3 priority '200'
|
||||
set high-availability vrrp group vlan3 vrid '3'
|
||||
set high-availability vrrp group vlan9 address 172.31.9.1/24
|
||||
set high-availability vrrp group vlan9 hello-source-address '172.31.9.252'
|
||||
set high-availability vrrp group vlan9 interface 'bond0.9'
|
||||
set high-availability vrrp group vlan9 no-preempt
|
||||
set high-availability vrrp group vlan9 peer-address '172.31.9.253'
|
||||
set high-availability vrrp group vlan9 priority '200'
|
||||
set high-availability vrrp group vlan9 vrid '9'
|
||||
set high-availability vrrp group vlan10 address 172.31.10.1/23
|
||||
set high-availability vrrp group vlan10 hello-source-address '172.31.10.252'
|
||||
set high-availability vrrp group vlan10 interface 'bond0.10'
|
||||
set high-availability vrrp group vlan10 no-preempt
|
||||
set high-availability vrrp group vlan10 peer-address '172.31.10.253'
|
||||
set high-availability vrrp group vlan10 priority '200'
|
||||
set high-availability vrrp group vlan10 vrid '10'
|
||||
set high-availability vrrp group vlan200 address 172.31.200.1/24
|
||||
set high-availability vrrp group vlan200 hello-source-address '172.31.200.252'
|
||||
set high-availability vrrp group vlan200 interface 'bond0.200'
|
||||
set high-availability vrrp group vlan200 no-preempt
|
||||
set high-availability vrrp group vlan200 peer-address '172.31.200.253'
|
||||
set high-availability vrrp group vlan200 priority '200'
|
||||
set high-availability vrrp group vlan200 vrid '200'
|
||||
set interfaces bonding bond0 description 'api-batch-test'
|
||||
set interfaces bonding bond0 hash-policy 'layer2+3'
|
||||
set interfaces bonding bond0 lacp-rate 'fast'
|
||||
set interfaces bonding bond0 member interface 'eth0'
|
||||
set interfaces bonding bond0 member interface 'eth1'
|
||||
set interfaces bonding bond0 mode '802.3ad'
|
||||
set interfaces bonding bond0 vif 1 address '172.31.1.252/24'
|
||||
set interfaces bonding bond0 vif 1 description 'management'
|
||||
set interfaces bonding bond0 vif 2 address '172.31.2.252/24'
|
||||
set interfaces bonding bond0 vif 2 description 'k8s'
|
||||
set interfaces bonding bond0 vif 3 address '172.31.3.252/24'
|
||||
set interfaces bonding bond0 vif 3 description 'kvm'
|
||||
set interfaces bonding bond0 vif 9 address '172.31.9.252/24'
|
||||
set interfaces bonding bond0 vif 9 description 'private'
|
||||
set interfaces bonding bond0 vif 10 address '172.31.10.252/23'
|
||||
set interfaces bonding bond0 vif 10 description 'lot'
|
||||
set interfaces bonding bond0 vif 51 description 'WAN1 Vodafone-equivalent (sim ISP PPPoE)'
|
||||
set interfaces bonding bond0 vif 53 address 'dhcp'
|
||||
set interfaces bonding bond0 vif 53 description 'WAN3 10gig-equivalent (sim ISP DHCP)'
|
||||
set interfaces bonding bond0 vif 53 dhcp-options default-route-distance '210'
|
||||
set interfaces bonding bond0 vif 200 address '172.31.200.252/24'
|
||||
set interfaces bonding bond0 vif 200 description 'roomates'
|
||||
6
labsim/vlan-leak-evidence/after/capture-parent.txt
Normal file
6
labsim/vlan-leak-evidence/after/capture-parent.txt
Normal file
@@ -0,0 +1,6 @@
|
||||
12:52:14.639170 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype 802.1Q (0x8100), length 346: vlan 3, p 0, ethertype IPv4 (0x0800), 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:52:14.640467 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype 802.1Q (0x8100), length 329: vlan 3, p 0, ethertype IPv4 (0x0800), 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:14.640846 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype 802.1Q (0x8100), length 329: vlan 3, p 0, ethertype IPv4 (0x0800), 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:14.642554 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype 802.1Q (0x8100), length 346: vlan 3, p 0, ethertype IPv4 (0x0800), 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:52:14.642766 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype 802.1Q (0x8100), length 329: vlan 3, p 0, ethertype IPv4 (0x0800), 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:14.643056 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype 802.1Q (0x8100), length 329: vlan 3, p 0, ethertype IPv4 (0x0800), 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
6
labsim/vlan-leak-evidence/after/capture-vif.txt
Normal file
6
labsim/vlan-leak-evidence/after/capture-vif.txt
Normal file
@@ -0,0 +1,6 @@
|
||||
12:52:14.639170 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype IPv4 (0x0800), length 342: 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:52:14.640465 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:14.640845 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:14.642554 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype IPv4 (0x0800), length 342: 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:52:14.642764 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:52:14.643055 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
4
labsim/vlan-leak-evidence/after/client.txt
Normal file
4
labsim/vlan-leak-evidence/after/client.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
udhcpc: started, v1.37.0
|
||||
udhcpc: broadcasting discover
|
||||
udhcpc: broadcasting select for 172.31.3.11, server 172.31.3.252
|
||||
udhcpc: lease of 172.31.3.11 obtained from 172.31.3.252, lease time 85374
|
||||
64
labsim/vlan-leak-evidence/after/router-config.txt
Normal file
64
labsim/vlan-leak-evidence/after/router-config.txt
Normal file
@@ -0,0 +1,64 @@
|
||||
set high-availability vrrp group native address 172.31.1.1/24
|
||||
set high-availability vrrp group native hello-source-address '172.31.1.252'
|
||||
set high-availability vrrp group native interface 'bond0.1'
|
||||
set high-availability vrrp group native no-preempt
|
||||
set high-availability vrrp group native peer-address '172.31.1.253'
|
||||
set high-availability vrrp group native priority '200'
|
||||
set high-availability vrrp group native vrid '1'
|
||||
set high-availability vrrp group vlan2 address 172.31.2.1/24
|
||||
set high-availability vrrp group vlan2 hello-source-address '172.31.2.252'
|
||||
set high-availability vrrp group vlan2 interface 'bond0.2'
|
||||
set high-availability vrrp group vlan2 no-preempt
|
||||
set high-availability vrrp group vlan2 peer-address '172.31.2.253'
|
||||
set high-availability vrrp group vlan2 priority '200'
|
||||
set high-availability vrrp group vlan2 vrid '2'
|
||||
set high-availability vrrp group vlan3 address 172.31.3.1/24
|
||||
set high-availability vrrp group vlan3 hello-source-address '172.31.3.252'
|
||||
set high-availability vrrp group vlan3 interface 'bond0.3'
|
||||
set high-availability vrrp group vlan3 no-preempt
|
||||
set high-availability vrrp group vlan3 peer-address '172.31.3.253'
|
||||
set high-availability vrrp group vlan3 priority '200'
|
||||
set high-availability vrrp group vlan3 vrid '3'
|
||||
set high-availability vrrp group vlan9 address 172.31.9.1/24
|
||||
set high-availability vrrp group vlan9 hello-source-address '172.31.9.252'
|
||||
set high-availability vrrp group vlan9 interface 'bond0.9'
|
||||
set high-availability vrrp group vlan9 no-preempt
|
||||
set high-availability vrrp group vlan9 peer-address '172.31.9.253'
|
||||
set high-availability vrrp group vlan9 priority '200'
|
||||
set high-availability vrrp group vlan9 vrid '9'
|
||||
set high-availability vrrp group vlan10 address 172.31.10.1/23
|
||||
set high-availability vrrp group vlan10 hello-source-address '172.31.10.252'
|
||||
set high-availability vrrp group vlan10 interface 'bond0.10'
|
||||
set high-availability vrrp group vlan10 no-preempt
|
||||
set high-availability vrrp group vlan10 peer-address '172.31.10.253'
|
||||
set high-availability vrrp group vlan10 priority '200'
|
||||
set high-availability vrrp group vlan10 vrid '10'
|
||||
set high-availability vrrp group vlan200 address 172.31.200.1/24
|
||||
set high-availability vrrp group vlan200 hello-source-address '172.31.200.252'
|
||||
set high-availability vrrp group vlan200 interface 'bond0.200'
|
||||
set high-availability vrrp group vlan200 no-preempt
|
||||
set high-availability vrrp group vlan200 peer-address '172.31.200.253'
|
||||
set high-availability vrrp group vlan200 priority '200'
|
||||
set high-availability vrrp group vlan200 vrid '200'
|
||||
set interfaces bonding bond0 description 'api-batch-test'
|
||||
set interfaces bonding bond0 hash-policy 'layer2+3'
|
||||
set interfaces bonding bond0 lacp-rate 'fast'
|
||||
set interfaces bonding bond0 member interface 'eth0'
|
||||
set interfaces bonding bond0 member interface 'eth1'
|
||||
set interfaces bonding bond0 mode '802.3ad'
|
||||
set interfaces bonding bond0 vif 1 address '172.31.1.252/24'
|
||||
set interfaces bonding bond0 vif 1 description 'management'
|
||||
set interfaces bonding bond0 vif 2 address '172.31.2.252/24'
|
||||
set interfaces bonding bond0 vif 2 description 'k8s'
|
||||
set interfaces bonding bond0 vif 3 address '172.31.3.252/24'
|
||||
set interfaces bonding bond0 vif 3 description 'kvm'
|
||||
set interfaces bonding bond0 vif 9 address '172.31.9.252/24'
|
||||
set interfaces bonding bond0 vif 9 description 'private'
|
||||
set interfaces bonding bond0 vif 10 address '172.31.10.252/23'
|
||||
set interfaces bonding bond0 vif 10 description 'lot'
|
||||
set interfaces bonding bond0 vif 51 description 'WAN1 Vodafone-equivalent (sim ISP PPPoE)'
|
||||
set interfaces bonding bond0 vif 53 address 'dhcp'
|
||||
set interfaces bonding bond0 vif 53 description 'WAN3 10gig-equivalent (sim ISP DHCP)'
|
||||
set interfaces bonding bond0 vif 53 dhcp-options default-route-distance '210'
|
||||
set interfaces bonding bond0 vif 200 address '172.31.200.252/24'
|
||||
set interfaces bonding bond0 vif 200 description 'roomates'
|
||||
5
labsim/vlan-leak-evidence/before/capture-parent.txt
Normal file
5
labsim/vlan-leak-evidence/before/capture-parent.txt
Normal file
@@ -0,0 +1,5 @@
|
||||
12:37:08.491910 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype 802.1Q (0x8100), length 346: vlan 3, p 0, ethertype IPv4 (0x0800), 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:37:08.492629 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.1.252.67 > 172.31.1.9.68: BOOTP/DHCP, Reply, length 283
|
||||
12:37:08.493628 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype 802.1Q (0x8100), length 329: vlan 3, p 0, ethertype IPv4 (0x0800), 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:37:08.495587 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype 802.1Q (0x8100), length 346: vlan 3, p 0, ethertype IPv4 (0x0800), 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:37:08.495946 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype 802.1Q (0x8100), length 329: vlan 3, p 0, ethertype IPv4 (0x0800), 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
4
labsim/vlan-leak-evidence/before/capture-vif.txt
Normal file
4
labsim/vlan-leak-evidence/before/capture-vif.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
12:37:08.491910 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype IPv4 (0x0800), length 342: 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:37:08.493625 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
12:37:08.495587 52:54:00:02:2e:b1 > ff:ff:ff:ff:ff:ff, ethertype IPv4 (0x0800), length 342: 0.0.0.0.68 > 255.255.255.255.67: BOOTP/DHCP, Request from 52:54:00:02:2e:b1, length 300
|
||||
12:37:08.495944 52:54:00:e5:95:a2 > 52:54:00:02:2e:b1, ethertype IPv4 (0x0800), length 325: 172.31.3.252.67 > 172.31.3.11.68: BOOTP/DHCP, Reply, length 283
|
||||
4
labsim/vlan-leak-evidence/before/client.txt
Normal file
4
labsim/vlan-leak-evidence/before/client.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
udhcpc: started, v1.37.0
|
||||
udhcpc: broadcasting discover
|
||||
udhcpc: broadcasting select for 172.31.3.11, server 172.31.3.252
|
||||
udhcpc: lease of 172.31.3.11 obtained from 172.31.3.252, lease time 86280
|
||||
63
labsim/vlan-leak-evidence/before/router-config.txt
Normal file
63
labsim/vlan-leak-evidence/before/router-config.txt
Normal file
@@ -0,0 +1,63 @@
|
||||
set high-availability vrrp group native address 172.31.1.1/24
|
||||
set high-availability vrrp group native hello-source-address '172.31.1.252'
|
||||
set high-availability vrrp group native interface 'bond0'
|
||||
set high-availability vrrp group native no-preempt
|
||||
set high-availability vrrp group native peer-address '172.31.1.253'
|
||||
set high-availability vrrp group native priority '200'
|
||||
set high-availability vrrp group native vrid '1'
|
||||
set high-availability vrrp group vlan2 address 172.31.2.1/24
|
||||
set high-availability vrrp group vlan2 hello-source-address '172.31.2.252'
|
||||
set high-availability vrrp group vlan2 interface 'bond0.2'
|
||||
set high-availability vrrp group vlan2 no-preempt
|
||||
set high-availability vrrp group vlan2 peer-address '172.31.2.253'
|
||||
set high-availability vrrp group vlan2 priority '200'
|
||||
set high-availability vrrp group vlan2 vrid '2'
|
||||
set high-availability vrrp group vlan3 address 172.31.3.1/24
|
||||
set high-availability vrrp group vlan3 hello-source-address '172.31.3.252'
|
||||
set high-availability vrrp group vlan3 interface 'bond0.3'
|
||||
set high-availability vrrp group vlan3 no-preempt
|
||||
set high-availability vrrp group vlan3 peer-address '172.31.3.253'
|
||||
set high-availability vrrp group vlan3 priority '200'
|
||||
set high-availability vrrp group vlan3 vrid '3'
|
||||
set high-availability vrrp group vlan9 address 172.31.9.1/24
|
||||
set high-availability vrrp group vlan9 hello-source-address '172.31.9.252'
|
||||
set high-availability vrrp group vlan9 interface 'bond0.9'
|
||||
set high-availability vrrp group vlan9 no-preempt
|
||||
set high-availability vrrp group vlan9 peer-address '172.31.9.253'
|
||||
set high-availability vrrp group vlan9 priority '200'
|
||||
set high-availability vrrp group vlan9 vrid '9'
|
||||
set high-availability vrrp group vlan10 address 172.31.10.1/23
|
||||
set high-availability vrrp group vlan10 hello-source-address '172.31.10.252'
|
||||
set high-availability vrrp group vlan10 interface 'bond0.10'
|
||||
set high-availability vrrp group vlan10 no-preempt
|
||||
set high-availability vrrp group vlan10 peer-address '172.31.10.253'
|
||||
set high-availability vrrp group vlan10 priority '200'
|
||||
set high-availability vrrp group vlan10 vrid '10'
|
||||
set high-availability vrrp group vlan200 address 172.31.200.1/24
|
||||
set high-availability vrrp group vlan200 hello-source-address '172.31.200.252'
|
||||
set high-availability vrrp group vlan200 interface 'bond0.200'
|
||||
set high-availability vrrp group vlan200 no-preempt
|
||||
set high-availability vrrp group vlan200 peer-address '172.31.200.253'
|
||||
set high-availability vrrp group vlan200 priority '200'
|
||||
set high-availability vrrp group vlan200 vrid '200'
|
||||
set interfaces bonding bond0 address '172.31.1.252/24'
|
||||
set interfaces bonding bond0 description 'api-batch-test'
|
||||
set interfaces bonding bond0 hash-policy 'layer2+3'
|
||||
set interfaces bonding bond0 lacp-rate 'fast'
|
||||
set interfaces bonding bond0 member interface 'eth0'
|
||||
set interfaces bonding bond0 member interface 'eth1'
|
||||
set interfaces bonding bond0 mode '802.3ad'
|
||||
set interfaces bonding bond0 vif 2 address '172.31.2.252/24'
|
||||
set interfaces bonding bond0 vif 2 description 'k8s'
|
||||
set interfaces bonding bond0 vif 3 address '172.31.3.252/24'
|
||||
set interfaces bonding bond0 vif 3 description 'kvm'
|
||||
set interfaces bonding bond0 vif 9 address '172.31.9.252/24'
|
||||
set interfaces bonding bond0 vif 9 description 'private'
|
||||
set interfaces bonding bond0 vif 10 address '172.31.10.252/23'
|
||||
set interfaces bonding bond0 vif 10 description 'lot'
|
||||
set interfaces bonding bond0 vif 51 description 'WAN1 Vodafone-equivalent (sim ISP PPPoE)'
|
||||
set interfaces bonding bond0 vif 53 address 'dhcp'
|
||||
set interfaces bonding bond0 vif 53 description 'WAN3 10gig-equivalent (sim ISP DHCP)'
|
||||
set interfaces bonding bond0 vif 53 dhcp-options default-route-distance '210'
|
||||
set interfaces bonding bond0 vif 200 address '172.31.200.252/24'
|
||||
set interfaces bonding bond0 vif 200 description 'roomates'
|
||||
17
labsim/vlan1-move-monitor.sh
Executable file
17
labsim/vlan1-move-monitor.sh
Executable file
@@ -0,0 +1,17 @@
|
||||
#!/bin/bash
|
||||
# Timestamped liveness log for the Management VLAN during the bond0 -> bond0.1 move.
|
||||
#
|
||||
# The question this answers is not "did it work" but "for how long was it not
|
||||
# working, and what held the VIP while it was not". Both are invisible after the
|
||||
# fact: VRRP reconverges and leaves no trace of who was master during the gap.
|
||||
#
|
||||
# ./vlan1-move-monitor.sh > /tmp/move.log &
|
||||
# Columns: time VIP-ping R1-ping R2-ping VIP-mac
|
||||
VIP="${VIP:-172.31.1.1}"; R1="${R1:-172.31.1.252}"; R2="${R2:-172.31.1.253}"
|
||||
p() { ping -c1 -W1 -n "$1" >/dev/null 2>&1 && echo up || echo DOWN; }
|
||||
while :; do
|
||||
mac="$(ip neigh show "$VIP" 2>/dev/null | awk '{for(i=1;i<=NF;i++) if($i=="lladdr") print $(i+1)}')"
|
||||
printf '%s vip=%-4s r1=%-4s r2=%-4s vipmac=%s\n' \
|
||||
"$(date +%H:%M:%S)" "$(p "$VIP")" "$(p "$R1")" "$(p "$R2")" "${mac:-none}"
|
||||
sleep 1
|
||||
done
|
||||
85
labsim/wan-failover-evidence/README.md
Normal file
85
labsim/wan-failover-evidence/README.md
Normal file
@@ -0,0 +1,85 @@
|
||||
# WAN failover evidence
|
||||
|
||||
Captured by `labsim/labsim-pppoe-ha-test.sh`. Each directory holds the state of
|
||||
both routers and the access concentrator at the end of one test.
|
||||
|
||||
## The number that sizes GRACE
|
||||
|
||||
`T4` destroys the master with `virsh destroy` — no LCP Terminate, no PADT, the
|
||||
router simply ceases — and times how long until the survivor holds a PPPoE
|
||||
session. Run across every policy VyOS can express, because Vodafone's is
|
||||
unknown:
|
||||
|
||||
Two independent runs, so these describe the AC's behaviour rather than one-offs:
|
||||
|
||||
| `session-control` | takeover | |
|
||||
|---|---|---|
|
||||
| `replace` | 26s / 26s | accel-ppp default; the new auth kills the old session |
|
||||
| **`deny`** | **148s / 141s** | the AC refuses the survivor until its own dead-peer timer frees the dead session |
|
||||
| `disable` | 21s / 20s | no single-session enforcement at all |
|
||||
|
||||
`deny` is the only one that matters for sizing, and `GRACE=300` in
|
||||
`migration/vrrp-wan.conf` comes from it. Session polling during that run caught
|
||||
the mechanism directly: the destroyed router's session stayed in the AC's table
|
||||
while the survivor's dial attempts appeared and were rejected — twice — before
|
||||
one finally took.
|
||||
|
||||
**Treat 148s as a floor, not a worst case.** These are idle 2-vCPU VMs, and the
|
||||
AC shares an OVS bridge with the routers, so `virsh destroy` removes the port
|
||||
and accel-ppp sees the peer physically vanish. A real BRAS reached over DSL
|
||||
never learns our router died; it waits out its own timers, which are longer and
|
||||
not ours to know.
|
||||
|
||||
## How these numbers were nearly wrong
|
||||
|
||||
Until 2026-09-06 the matrix set the policy with:
|
||||
|
||||
```sh
|
||||
isp "vbash -c 'source /opt/vyatta/etc/functions/script-template; configure; \
|
||||
set service pppoe-server session-control $mode; commit; save; exit'"
|
||||
```
|
||||
|
||||
That form never starts a config session. `commit` fails with
|
||||
`Invalid command: [commit]` on stderr, which `isp()` discards — so all three
|
||||
iterations ran against the accel-ppp default while printing the mode they were
|
||||
supposedly testing. `show configuration commands | grep session-control` on the
|
||||
ISP VM came back empty after a full run. The matrix reported `deny` at 25s; the
|
||||
real figure is 148s, and `GRACE` was sized against the fiction.
|
||||
|
||||
`isp_session_control()` now drives it from a real script file, reads the value
|
||||
back, and skips the iteration rather than measure the wrong policy. A harness
|
||||
that reports coverage it does not have is worse than one that reports a failure.
|
||||
|
||||
## The other tests
|
||||
|
||||
| | what it proves |
|
||||
|---|---|
|
||||
| `T0-baseline` | exactly one session, held by the VIP holder |
|
||||
| `T3-clean-failover` | `force-fault` moves `pppoe0` in 20–26s |
|
||||
| `T5-tengig-down` | 10 gig down → route falls to `pppoe0`, LAN back in 5s |
|
||||
| `T8-lease-expiry` | the guard hangs up a stale lease within ~5s |
|
||||
| `T11-no-peers-file` | a blessed box with no peers file does not restart-loop |
|
||||
| `T12-holdoff-keeps-session` | a 900s flap hold-off does **not** tear down a live session |
|
||||
|
||||
`T12` exists because it did. `ppp_dial()` checked the hold-off and returned
|
||||
before renewing `/run/vrrp-wan/may-dial`; that lease is what `vrrp-wan-guard`
|
||||
expires, so tripping the damper stopped the renewal and the guard hung up
|
||||
`pppoe0` on the master ~80s later:
|
||||
|
||||
```
|
||||
DIAL FLAP: >=6 attempts in 600s -- holding off 900s
|
||||
GUARD: lease stale (81s > 75s) -- hanging up pppoe0
|
||||
```
|
||||
|
||||
A damper meant to suppress repeated *dials* was destroying an established
|
||||
session instead.
|
||||
|
||||
## Reading `check_invariant`
|
||||
|
||||
The invariant that matters is **at most one of our routers holds `pppoe0`**.
|
||||
The AC's session count is only a proxy for it, and only while the AC enforces
|
||||
single-session — under `disable` it does not, so a destroyed router's session
|
||||
lingers and the count reads 2 with exactly one live router dialled. That is
|
||||
reported as a `WARN`, not a failure. It is not silenced, because an orphaned
|
||||
session still occupies the single slot at a real ISP: it is exactly what made
|
||||
`deny` take 148s.
|
||||
31
labsim/wan-failover-evidence/T0-baseline/state.txt
Normal file
31
labsim/wan-failover-evidence/T0-baseline/state.txt
Normal file
@@ -0,0 +1,31 @@
|
||||
=== 2026-09-06T00:27:16+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp1 | simdsl | 198.51.100.131 | | | 52:54:00:4e:0b:56 | | active | 00:04:24 | 1.1 KiB | 204 B
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup
|
||||
Sep 05 23:07:43 apitest vrrp-wan[22667]: MASTER: dialling pppoe0
|
||||
Sep 05 23:08:14 apitest vrrp-wan[24120]: MASTER: dialling pppoe0
|
||||
-- Boot c5f23399239b468c8c8b752a4305c515 --
|
||||
Sep 05 23:13:04 apitest vrrp-wan[6261]: MASTER: dialling pppoe0
|
||||
Sep 05 23:13:04 apitest vrrp-wan[6413]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:13:08 apitest vrrp-wan[7122]: bond0.53 enable commit took 4s
|
||||
-- Boot 46727d256db344d6ad4b37c142272071 --
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6258]: MASTER: dialling pppoe0
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6411]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:18:36 apitest vrrp-wan[7038]: bond0.53 enable commit took 5s
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=19 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.131 peer 198.51.100.1/32
|
||||
default nhid 57 dev pppoe0 proto static metric 20
|
||||
Sep 05 23:10:44 vyos vrrp-wan[221212]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:10:48 vyos vrrp-wan[221995]: bond0.53 enable commit took 4s
|
||||
-- Boot 1f635fc6900c4661b3ac6d016d63d293 --
|
||||
Sep 05 23:16:10 vyos vrrp-wan[7037]: MASTER: dialling pppoe0
|
||||
Sep 05 23:16:11 vyos vrrp-wan[7190]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:16:15 vyos vrrp-wan[7979]: bond0.53 enable commit took 4s
|
||||
-- Boot c0e38585d7354ad6ab018800c7f3f6be --
|
||||
Sep 05 23:22:54 vyos vrrp-wan[5970]: MASTER: dialling pppoe0
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
26
labsim/wan-failover-evidence/T11-no-peers-file/state.txt
Normal file
26
labsim/wan-failover-evidence/T11-no-peers-file/state.txt
Normal file
@@ -0,0 +1,26 @@
|
||||
=== 2026-09-06T00:29:55+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----+-----+--------+-------------+------------+-------+--------+----------+----------
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=no ppp_up=no ppp_active=yes may_dial=yes lease_age=1 dropin=yes role=master
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6411]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:18:36 apitest vrrp-wan[7038]: bond0.53 enable commit took 5s
|
||||
-- Boot 68bb1e8d56b34d55a511a301c24270fe --
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12502]: MASTER: dialling pppoe0
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12654]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:29:01 apitest vrrp-wan[16369]: GUARD: lease stale (200s > 75s; is vrrp-wan-reconcile.timer running?) -- hanging up pppoe0
|
||||
Sep 05 23:29:27 apitest vrrp-wan[17398]: MASTER: dialling pppoe0
|
||||
Sep 05 23:29:57 apitest vrrp-wan[18610]: MASTER: dialling pppoe0
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup
|
||||
Sep 05 23:16:11 vyos vrrp-wan[7190]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:16:15 vyos vrrp-wan[7979]: bond0.53 enable commit took 4s
|
||||
-- Boot c0e38585d7354ad6ab018800c7f3f6be --
|
||||
Sep 05 23:22:54 vyos vrrp-wan[5970]: MASTER: dialling pppoe0
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16037]: not MASTER: hanging up pppoe0
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
@@ -0,0 +1,29 @@
|
||||
=== 2026-09-06T00:31:55+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp0 | simdsl | 198.51.100.134 | | | 52:54:00:e5:95:a2 | | active | 00:01:58 | 670 B | 555 B
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=no ppp_up=yes ppp_active=yes may_dial=yes lease_age=21 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.134 peer 198.51.100.1/32
|
||||
default nhid 74 dev pppoe0 proto static metric 20
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6411]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:18:36 apitest vrrp-wan[7038]: bond0.53 enable commit took 5s
|
||||
-- Boot 68bb1e8d56b34d55a511a301c24270fe --
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12502]: MASTER: dialling pppoe0
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12654]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:29:01 apitest vrrp-wan[16369]: GUARD: lease stale (200s > 75s; is vrrp-wan-reconcile.timer running?) -- hanging up pppoe0
|
||||
Sep 05 23:29:27 apitest vrrp-wan[17398]: MASTER: dialling pppoe0
|
||||
Sep 05 23:29:57 apitest vrrp-wan[18610]: MASTER: dialling pppoe0
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup
|
||||
Sep 05 23:16:11 vyos vrrp-wan[7190]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:16:15 vyos vrrp-wan[7979]: bond0.53 enable commit took 4s
|
||||
-- Boot c0e38585d7354ad6ab018800c7f3f6be --
|
||||
Sep 05 23:22:54 vyos vrrp-wan[5970]: MASTER: dialling pppoe0
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16037]: not MASTER: hanging up pppoe0
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
30
labsim/wan-failover-evidence/T3-clean-failover/state.txt
Normal file
30
labsim/wan-failover-evidence/T3-clean-failover/state.txt
Normal file
@@ -0,0 +1,30 @@
|
||||
=== 2026-09-06T00:27:58+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp0 | simdsl | 198.51.100.132 | | | 52:54:00:e5:95:a2 | | active | 00:00:18 | 514 B | 204 B
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=7 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.132 peer 198.51.100.1/32
|
||||
default via 203.0.113.1 dev bond0.53 proto failover metric 1
|
||||
Sep 05 23:13:04 apitest vrrp-wan[6413]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:13:08 apitest vrrp-wan[7122]: bond0.53 enable commit took 4s
|
||||
-- Boot 46727d256db344d6ad4b37c142272071 --
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6258]: MASTER: dialling pppoe0
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6411]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:18:36 apitest vrrp-wan[7038]: bond0.53 enable commit took 5s
|
||||
-- Boot 68bb1e8d56b34d55a511a301c24270fe --
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12502]: MASTER: dialling pppoe0
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12654]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup
|
||||
Sep 05 23:16:11 vyos vrrp-wan[7190]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:16:15 vyos vrrp-wan[7979]: bond0.53 enable commit took 4s
|
||||
-- Boot c0e38585d7354ad6ab018800c7f3f6be --
|
||||
Sep 05 23:22:54 vyos vrrp-wan[5970]: MASTER: dialling pppoe0
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16037]: not MASTER: hanging up pppoe0
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
19
labsim/wan-failover-evidence/T4-hard-failover-deny/state.txt
Normal file
19
labsim/wan-failover-evidence/T4-hard-failover-deny/state.txt
Normal file
@@ -0,0 +1,19 @@
|
||||
=== 2026-09-06T00:37:11+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp0 | simdsl | 198.51.100.136 | | | 52:54:00:e5:95:a2 | | active | 00:00:13 | 438 B | 204 B
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=26 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.136 peer 198.51.100.1/32
|
||||
default via 203.0.113.1 dev bond0.53 proto failover metric 1
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12654]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:29:01 apitest vrrp-wan[16369]: GUARD: lease stale (200s > 75s; is vrrp-wan-reconcile.timer running?) -- hanging up pppoe0
|
||||
Sep 05 23:29:27 apitest vrrp-wan[17398]: MASTER: dialling pppoe0
|
||||
Sep 05 23:29:57 apitest vrrp-wan[18610]: MASTER: dialling pppoe0
|
||||
-- Boot 6efc3c9ba47f455e9668454ee6c2fc37 --
|
||||
Sep 05 23:34:48 apitest vrrp-wan[6266]: MASTER: dialling pppoe0
|
||||
Sep 05 23:34:49 apitest vrrp-wan[6422]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:34:54 apitest vrrp-wan[7130]: bond0.53 enable commit took 5s
|
||||
--- 172.31.1.253 ---
|
||||
@@ -0,0 +1,20 @@
|
||||
=== 2026-09-06T00:39:29+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp0 | simdsl | 198.51.100.136 | | | 52:54:00:e5:95:a2 | | active | 00:02:31 | 438 B | 204 B
|
||||
ppp1 | simdsl | 198.51.100.137 | | | 52:54:00:4e:0b:56 | | active | 00:00:26 | 1.0 KiB | 204 B
|
||||
--- 172.31.1.252 ---
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=5 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.137 peer 198.51.100.1/32
|
||||
default nhid 60 dev pppoe0 proto static metric 20
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
Sep 05 23:32:26 vyos vrrp-wan[23806]: MASTER: dialling pppoe0
|
||||
Sep 05 23:32:26 vyos vrrp-wan[23956]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:32:31 vyos vrrp-wan[24749]: bond0.53 enable commit took 5s
|
||||
-- Boot 4b0355878444477fb92738339f85843d --
|
||||
Sep 05 23:39:03 vyos vrrp-wan[5966]: MASTER: dialling pppoe0
|
||||
Sep 05 23:39:04 vyos vrrp-wan[6120]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:39:09 vyos vrrp-wan[6914]: bond0.53 enable commit took 5s
|
||||
@@ -0,0 +1,18 @@
|
||||
=== 2026-09-06T00:32:54+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp0 | simdsl | 198.51.100.135 | | | 52:54:00:4e:0b:56 | | active | 00:00:30 | 438 B | 204 B
|
||||
--- 172.31.1.252 ---
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=19 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.135 peer 198.51.100.1/32
|
||||
default nhid 77 dev pppoe0 proto static metric 20
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16037]: not MASTER: hanging up pppoe0
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
Sep 05 23:32:26 vyos vrrp-wan[23806]: MASTER: dialling pppoe0
|
||||
Sep 05 23:32:26 vyos vrrp-wan[23956]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:32:31 vyos vrrp-wan[24749]: bond0.53 enable commit took 5s
|
||||
30
labsim/wan-failover-evidence/T5-tengig-down/state.txt
Normal file
30
labsim/wan-failover-evidence/T5-tengig-down/state.txt
Normal file
@@ -0,0 +1,30 @@
|
||||
=== 2026-09-06T00:28:31+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----------------+-----+--------+-------------------+------------+--------+----------+----------+----------
|
||||
ppp0 | simdsl | 198.51.100.132 | | | 52:54:00:e5:95:a2 | | active | 00:00:52 | 682 B | 372 B
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=yes ppp_up=yes ppp_active=yes may_dial=yes lease_age=7 dropin=yes role=master
|
||||
pppoe0 UNKNOWN 198.51.100.132 peer 198.51.100.1/32
|
||||
default nhid 58 dev pppoe0 proto static metric 20
|
||||
Sep 05 23:13:04 apitest vrrp-wan[6413]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:13:08 apitest vrrp-wan[7122]: bond0.53 enable commit took 4s
|
||||
-- Boot 46727d256db344d6ad4b37c142272071 --
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6258]: MASTER: dialling pppoe0
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6411]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:18:36 apitest vrrp-wan[7038]: bond0.53 enable commit took 5s
|
||||
-- Boot 68bb1e8d56b34d55a511a301c24270fe --
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12502]: MASTER: dialling pppoe0
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12654]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup
|
||||
Sep 05 23:16:11 vyos vrrp-wan[7190]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:16:15 vyos vrrp-wan[7979]: bond0.53 enable commit took 4s
|
||||
-- Boot c0e38585d7354ad6ab018800c7f3f6be --
|
||||
Sep 05 23:22:54 vyos vrrp-wan[5970]: MASTER: dialling pppoe0
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16037]: not MASTER: hanging up pppoe0
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
27
labsim/wan-failover-evidence/T8-lease-expiry/state.txt
Normal file
27
labsim/wan-failover-evidence/T8-lease-expiry/state.txt
Normal file
@@ -0,0 +1,27 @@
|
||||
=== 2026-09-06T00:29:16+01:00 ===
|
||||
--- AC sessions ---
|
||||
ifname | username | ip | ip6 | ip6-dp | calling-sid | rate-limit | state | uptime | rx-bytes | tx-bytes
|
||||
--------+----------+----+-----+--------+-------------+------------+-------+--------+----------+----------
|
||||
--- 172.31.1.252 ---
|
||||
vip=172.31.1.1 holds_vip=yes wan_disabled=no wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=master
|
||||
Sep 05 23:13:08 apitest vrrp-wan[7122]: bond0.53 enable commit took 4s
|
||||
-- Boot 46727d256db344d6ad4b37c142272071 --
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6258]: MASTER: dialling pppoe0
|
||||
Sep 05 23:18:31 apitest vrrp-wan[6411]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:18:36 apitest vrrp-wan[7038]: bond0.53 enable commit took 5s
|
||||
-- Boot 68bb1e8d56b34d55a511a301c24270fe --
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12502]: MASTER: dialling pppoe0
|
||||
Sep 05 23:27:40 apitest vrrp-wan[12654]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:27:45 apitest vrrp-wan[13361]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:29:01 apitest vrrp-wan[16369]: GUARD: lease stale (200s > 75s; is vrrp-wan-reconcile.timer running?) -- hanging up pppoe0
|
||||
--- 172.31.1.253 ---
|
||||
vip=172.31.1.1 holds_vip=no wan_disabled=yes wan_up=no ppp_up=no ppp_active=no may_dial=no lease_age=- dropin=yes role=backup
|
||||
Sep 05 23:16:11 vyos vrrp-wan[7190]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:16:15 vyos vrrp-wan[7979]: bond0.53 enable commit took 4s
|
||||
-- Boot c0e38585d7354ad6ab018800c7f3f6be --
|
||||
Sep 05 23:22:54 vyos vrrp-wan[5970]: MASTER: dialling pppoe0
|
||||
Sep 05 23:22:54 vyos vrrp-wan[6126]: MASTER with bond0.53 disabled -> enabling
|
||||
Sep 05 23:22:59 vyos vrrp-wan[6915]: bond0.53 enable commit took 5s
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16037]: not MASTER: hanging up pppoe0
|
||||
Sep 05 23:27:40 vyos vrrp-wan[16198]: not MASTER but bond0.53 enabled -> releasing
|
||||
Sep 05 23:27:47 vyos vrrp-wan[16840]: bond0.53 disable commit took 7s
|
||||
137
migration/MANAGEMENT-VLAN-TAGGED.md
Normal file
137
migration/MANAGEMENT-VLAN-TAGGED.md
Normal file
@@ -0,0 +1,137 @@
|
||||
# Moving Management onto a tagged VLAN
|
||||
|
||||
Rehearsed end to end in labsim on 2026-09-02. This is the fix for kea serving
|
||||
addresses from the wrong VLAN's pool.
|
||||
|
||||
## Why
|
||||
|
||||
ISC Kea [#1117](https://gitlab.isc.org/isc-projects/kea/-/issues/1117): with
|
||||
`dhcp-socket-type: raw`, a frame tagged for a sub-interface is **also** delivered
|
||||
to the parent's `AF_PACKET` socket. If the parent serves a subnet, kea answers
|
||||
from it too. Ours does — Management is the native/untagged VLAN on `bond0` while
|
||||
VLANs 2/3/9/10/200 are sub-interfaces of that same bond — so one DISCOVER on
|
||||
VLAN 3 produces two OFFERs and the *client* decides which to keep:
|
||||
|
||||
```
|
||||
bond0.3 : 192.168.3.14 correct
|
||||
bond0 : 192.168.1.28 UNTAGGED, Management pool, wrong
|
||||
```
|
||||
|
||||
The fix is to leave **no subnet on the parent**: every VLAN tagged, Management
|
||||
included, moved from `bond0` to `bond0.1`.
|
||||
|
||||
Confirmed in labsim across all six LAN VLANs: fails before, passes after.
|
||||
`labsim/labsim-vlan-leak-test.sh` is the test; evidence in
|
||||
`labsim/vlan-leak-evidence/`.
|
||||
|
||||
## What must change together
|
||||
|
||||
Per router:
|
||||
|
||||
| | from | to |
|
||||
|---|---|---|
|
||||
| address | `interfaces bonding bond0 address` | `interfaces bonding bond0 vif 1 address` |
|
||||
| firewall | `interface-group LAN interface bond0` | `... interface bond0.1` |
|
||||
| VRRP | `vrrp group native interface bond0` | `... interface bond0.1` |
|
||||
| kea | — | **restart it** (see traps) |
|
||||
|
||||
On the switch: Native VLAN = **None** on the trunk to that firewall, with VLAN 1
|
||||
added to the tagged set.
|
||||
|
||||
## The ordering constraint
|
||||
|
||||
**There is no overlap state.** An 802.1Q port always egresses its native VLAN
|
||||
untagged, so while VLAN 1 is native the router can *send* tagged VLAN 1 but can
|
||||
never *receive* it. Verified: a tagged VLAN 1 ARP sent from the switch arrived on
|
||||
`bond0` untagged and never on `bond0.1`. Configuring "native VLAN 1 **and** VLAN 1
|
||||
tagged" as a make-before-break does not work; the switch and router changes for a
|
||||
given firewall are strictly simultaneous, and that router loses Management in
|
||||
between.
|
||||
|
||||
What makes this safe anyway: **tagged and untagged Management coexist on the
|
||||
same VLAN.** One VLAN is one broadcast domain no matter how each port tags it, so
|
||||
the firewalls can be converted one at a time — verified with the primary untagged
|
||||
and the secondary already tagged, both reachable, VIP up, VLAN 1 clients fine.
|
||||
|
||||
Access ports are untouched throughout. The UniFi controller at 192.168.1.5 and
|
||||
your workstation are on access ports and never traverse the firewall trunks, so
|
||||
you keep the controller you are making the change from. Only the router being
|
||||
converted goes dark, and only until its own config lands.
|
||||
|
||||
## Procedure
|
||||
|
||||
Do the **backup** router first, then fail the VIPs over and do the other. You
|
||||
need console (JetKVM) on the router being converted — its Management SSH dies the
|
||||
moment the switch port changes.
|
||||
|
||||
For each router in turn:
|
||||
|
||||
1. Confirm the *other* router is MASTER and healthy:
|
||||
`show vrrp` and `sudo /config/vrrp-wan-health; echo $?` (must be 0).
|
||||
2. Start the monitor from a workstation on an access port:
|
||||
`labsim/vlan1-move-monitor.sh` (edit the three addresses for production).
|
||||
3. UniFi: on this firewall's trunk ports, Native VLAN → None, VLAN 1 → tagged.
|
||||
This router's Management drops now.
|
||||
4. Over the console, in one commit:
|
||||
```
|
||||
set interfaces bonding bond0 vif 1 address '192.168.1.252/24' # .253 on vyos002
|
||||
set interfaces bonding bond0 vif 1 description 'management'
|
||||
delete interfaces bonding bond0 address
|
||||
set firewall group interface-group LAN interface 'bond0.1'
|
||||
delete firewall group interface-group LAN interface 'bond0'
|
||||
set high-availability vrrp group native interface 'bond0.1'
|
||||
commit
|
||||
save
|
||||
```
|
||||
5. `sudo systemctl restart isc-kea-dhcp4-server` — see traps.
|
||||
6. Verify: Management SSH back, `show vrrp` shows `native` on `bond0.1`, and the
|
||||
leak test passes.
|
||||
|
||||
Then fail back if the VIPs moved (below), and repeat for the other router.
|
||||
|
||||
### Measured windows (labsim)
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| this router's own Management unreachable | ~27 s (the console apply) |
|
||||
| VIP `.1` unreachable, peer already converted | **0 s** |
|
||||
| VIP `.1` unreachable, converting the current MASTER | ~6 s (VRRP failover) |
|
||||
| VIP unreachable if you convert both routers before the switch | **5 min 30 s** |
|
||||
|
||||
That last row is the failure mode to avoid: with both routers untagged and the
|
||||
trunks already changed, the VIP is a black hole and **the healthy BACKUP does not
|
||||
take over**. Its `native` group stays BACKUP because the *other* VLANs still hear
|
||||
the master, and the sync group holds them together. Redundancy does not help you
|
||||
here; only ordering does.
|
||||
|
||||
## Traps
|
||||
|
||||
- **Restart kea.** VyOS does not restart it for an interface address change, so
|
||||
it keeps a raw socket bound with the old address and keeps emitting the wrong
|
||||
offers. The first post-fix test in the sim failed for this reason alone and
|
||||
looked exactly like the fix not working.
|
||||
- **`interface-group LAN`.** Moving the address without moving the group means
|
||||
Management falls outside the group, and with default-deny that is every
|
||||
management session and all VLAN 1 inter-VLAN routing, gone on commit — on a
|
||||
router you reach through itself. Use `commit-confirm` if you are not on console.
|
||||
- **The VIPs may move, and `no-preempt` keeps them moved.** Converting a router
|
||||
restarts keepalived and re-initialises *every* group, not just `native`. In one
|
||||
rehearsal the priority-100 secondary took all six VIPs and held them while the
|
||||
priority-200 primary sat at BACKUP; in another the restart was quick enough that
|
||||
nothing moved. It is non-deterministic — check afterwards, every time.
|
||||
Fail back with `restart vrrp` **on the router currently holding them**.
|
||||
- **Duplicate delivery does not stop**, and should not be read as failure. #1117
|
||||
only promises there is no longer a subnet on the parent to match. Expect two
|
||||
identical replies per DISCOVER, both from the correct pool.
|
||||
- **Both firewalls' trunks must end up the same.** If UniFi shares one port
|
||||
profile between them, changing it converts both at once and you get the 5m30s
|
||||
row above. Check before you start; use per-port overrides if it does.
|
||||
|
||||
## Not covered by the rehearsal
|
||||
|
||||
- Whether UniFi's port profile can express "no native VLAN" the way OVS can, and
|
||||
whether the two firewalls share a profile. Unverified — check on the controller.
|
||||
- Why the JetKVM consoles specifically accepted the wrong OFFER when a VLAN 3
|
||||
access port should not receive an untagged VLAN 1 frame at all. Their port
|
||||
profile likely passes VLAN 1 untagged. Worth confirming, though it does not
|
||||
change the fix.
|
||||
360
migration/PPPOE-HA.md
Normal file
360
migration/PPPOE-HA.md
Normal file
@@ -0,0 +1,360 @@
|
||||
# PPPoE high availability
|
||||
|
||||
Proven in labsim. **Deployed to production 2026-09-06** — mechanism on both
|
||||
routers, `pppoe0 disable` removed from vyos002, vyos002 out of FAULT and in
|
||||
BACKUP, Pulumi model merged, and a controlled failover drill passed
|
||||
(takeover 52s, failback 36s). `vyos:verify` reports both routers in sync.
|
||||
|
||||
## What it does
|
||||
|
||||
One consumer ISP account, two routers. The 10 gig lease is bound to a cloned MAC
|
||||
(`f0:9f:c2:12:9b:4f`, the retired USG's) and the Vodafone line to a single
|
||||
credential, so neither may be live on both boxes. The WAN follows VRRP
|
||||
mastership — but the two halves use different control planes, and that is the
|
||||
whole design:
|
||||
|
||||
| | plane | why |
|
||||
|---|---|---|
|
||||
| `bond0.53` (10 gig) | VyOS **config** (`disable`) | only config can move a MAC |
|
||||
| `pppoe0` (Vodafone) | **systemd** unit gate | see below |
|
||||
|
||||
## Why PPPoE cannot live on the config plane
|
||||
|
||||
`interfaces_pppoe.py` treats `disable` and `delete` identically: both **unlink
|
||||
`/etc/ppp/peers/pppoe0`**, call `PPPoEIf.remove()` (withdrawing the FRR default
|
||||
route) and stop the unit. That path is pppd's own options file
|
||||
(`ExecStart=/usr/sbin/pppd call %I`), so the resting state destroyed exactly what
|
||||
the promotion path needed. `ppp@pppoe0` then restart-looped against the missing
|
||||
file — 47 restarts observed, zero sessions at the access concentrator — and
|
||||
never tripped systemd's limiter, because `RestartSec=5s` against the default
|
||||
10s/5-burst window is only two restarts per interval.
|
||||
|
||||
It also made op-mode `connect interface pppoe0` unusable (it refuses without the
|
||||
peers file), and put every failover behind a priority-322 commit where one
|
||||
unrelated invalid node fails the whole thing.
|
||||
|
||||
## The gate
|
||||
|
||||
`pppoe0` is configured identically and **enabled on both** routers, so the peers
|
||||
file always exists. Dialling is gated by
|
||||
`/etc/systemd/system/ppp@pppoe0.service.d/10-vrrp-wan-gate.conf`:
|
||||
|
||||
```ini
|
||||
ConditionPathExists=/run/vrrp-wan/may-dial
|
||||
ConditionPathExists=/etc/ppp/peers/pppoe0
|
||||
StartLimitIntervalSec=600
|
||||
StartLimitBurst=6
|
||||
```
|
||||
|
||||
`/run` is tmpfs, so the gate is shut at boot and neither box can dial before VRRP
|
||||
has decided. **This is load-bearing, not a nicety:** with the node enabled,
|
||||
`interfaces_pppoe.py` restarts ppp on *every* commit touching the pppoe subtree
|
||||
when the daemon is not running — so the backup actively tries to dial whenever
|
||||
anything commits (`pulumi up`, a hand commit, the boot-time config load). The
|
||||
gate is the only thing making that a no-op, which is why `vrrp-wan-reconcile`
|
||||
**refuses to bless a box whose drop-in is missing**: `/etc` is per-image, so a
|
||||
VyOS upgrade silently removes the protection, and failing closed turns that into
|
||||
"PPPoE never dials" rather than "both routers dial".
|
||||
|
||||
`may-dial` is a **lease, not a flag**. `ConditionPathExists` is evaluated at
|
||||
start only — it can prevent a dial, never revoke one. `vrrp-wan-reconcile`
|
||||
renews it every 30s; `vrrp-wan-guard` runs every 5s and only ever revokes, on
|
||||
either "I do not hold the VIP" or "the lease is stale".
|
||||
|
||||
## Measured in labsim
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Clean failover (`force-fault`) | `pppoe0` moves in **20-26s**, reproducible |
|
||||
| 10 gig down → PPPoE | route falls to `pppoe0`; LAN back online in **5s** |
|
||||
| Stale lease | guard hangs up within ~5s |
|
||||
| Missing peers file | `NRestarts=0` — no loop |
|
||||
| Flap holdoff vs live session | session survives a forged 900s holdoff |
|
||||
| Invariant | never more than one router dialled, in any run |
|
||||
|
||||
On the last row, precisely: the AC *did* report two `simdsl` sessions during the
|
||||
`session-control=disable` run — one live, one orphaned from the router that had
|
||||
just been destroyed, which that policy does not clean up. Only one live router
|
||||
was ever dialled. The count is a proxy for the real invariant and only a valid
|
||||
one while the AC enforces single-session, so it is reported as a `WARN` rather
|
||||
than silenced: an orphaned session still occupies the single slot at a real ISP,
|
||||
and that is exactly what made `deny` take 141–148s.
|
||||
|
||||
## Two failure modes found by running it, not by reading it
|
||||
|
||||
**The flap damper tore down a healthy WAN.** `ppp_dial()` checked the hold-off
|
||||
and returned *before* renewing `may-dial`. That file is a lease the guard
|
||||
expires after `LEASE_TTL`, so tripping the damper stopped the renewal and the
|
||||
guard hung up `pppoe0` **on the master** ~80s later:
|
||||
|
||||
```
|
||||
DIAL FLAP: >=6 attempts in 600s -- holding off 900s
|
||||
GUARD: lease stale (81s > 75s) -- hanging up pppoe0
|
||||
```
|
||||
|
||||
A damper meant to suppress repeated *dials* was destroying an established
|
||||
session instead. An active session now renews the lease and returns before
|
||||
every other check; everything below only decides whether to start a **new**
|
||||
session. T12 is the regression test.
|
||||
|
||||
**A missing peers file is silent.** `/etc/ppp/peers/pppoe0` is both pppd's
|
||||
options file and the gate's second condition, and it is only written by a commit
|
||||
that touches the pppoe subtree. Without it systemd logs
|
||||
`skipped because of an unmet condition check` exactly once and then nothing —
|
||||
a router that cannot dial at all looks identical to a healthy backup.
|
||||
`ppp_dial()` now says so on every tick, and distinguishes the two causes:
|
||||
configured-but-not-rendered (re-commit the subtree) versus no `pppoe0` in the
|
||||
config at all.
|
||||
|
||||
The second cause is the one to watch in production: **a commit that was never
|
||||
`save`d reverts on reboot and takes `pppoe0` with it.** That is exactly how the
|
||||
sim secondary lost its WAN and spent hours looking like an ISP problem. After
|
||||
any hand commit to the pppoe subtree, `save` — or the next reboot produces a
|
||||
standby that can never take over.
|
||||
|
||||
## Deploying — steps 1–9 done 2026-09-06
|
||||
|
||||
1. `sudo /config/vyos-known-good save` on both.
|
||||
2. `migration/vrrp-wan-install --vip 192.168.1.1 --host vyos@10.0.1.253` then the
|
||||
same for `.252`. Then `--check` on both. **No config change yet** — verify
|
||||
nothing dials.
|
||||
3. Confirm `/config/wan-secrets` is present and identical on both.
|
||||
4. **vyos002 first** (the non-master), on its own commit — `interfaces pppoe` is
|
||||
priority 322 and one bad node fails everything:
|
||||
`delete interfaces pppoe pppoe0 disable`, `commit-confirm 10`.
|
||||
5. Verify vyos002 did **not** dial — check on the wire, not from state:
|
||||
`sudo tcpdump -i bond0.51 -nn pppoed` should show no PADI. Then `confirm`; `save`.
|
||||
6. vyos001: nothing to change; it already has `pppoe0` enabled.
|
||||
7. Confirm `vif 53 disable` is in **both** `config.boot`s. vyos001's lacked it;
|
||||
fixed with `migration/vif53-pin-boot-disable` — see below.
|
||||
8. **Done 2026-09-06** (`kubernetes-deployment@45033dd`, on `main`). Both
|
||||
overrides merged, transition-scripts applied to both boxes by hand rather
|
||||
than left as drift, and `vyos:verify` is clean: 533 / 512 nodes, zero drift.
|
||||
The staging file `migration/pulumi-override-pppoe-gated.json` is kept as the
|
||||
record of why the ordering mattered. It was staged, unapplied, on purpose: another agent runs `pulumi up` on that repo, so
|
||||
merging it *is* a production change made by someone else at a time you do not
|
||||
choose. Removing `pppoe0 disable` from vyos002 before the gate exists there
|
||||
lets it dial on the next commit and take the single Vodafone session off
|
||||
vyos001. Run `npm run vyos:export && npm run vyos:render` first so the model
|
||||
follows whichever router actually holds the WAN.
|
||||
9. Add the drop-in re-install to the VyOS image-upgrade runbook.
|
||||
**Done** — `migration/VYOS-IMAGE-UPGRADE.md`. An upgrade replaces `/etc` and
|
||||
so removes the gate; the reconciler fails closed, giving "PPPoE never
|
||||
dials" rather than "both routers dial".
|
||||
|
||||
Step 4 is the one that matters most and is worth stopping on. vyos002 has been
|
||||
in **FAULT on all six groups for over three days** — verified again while
|
||||
writing this, alongside vyos001 holding `192.168.1.1` on `bond0.1` and
|
||||
`pppoe0` up on `83.106.5.72`. Until vyos002 reaches BACKUP there is no standby
|
||||
at all: if vyos001 died today nothing would pick up the gateway VIPs. The
|
||||
existing `vrrp-health-check-wan-present` override predicted exactly this in its
|
||||
own reason text — *"with the primary genuinely dead the secondary stays FAULT
|
||||
and nothing holds the gateway. The fix for that is WAN-follows-master, which is
|
||||
a separate change."* This is that change.
|
||||
|
||||
## Proven in production — controlled drill, 2026-09-06
|
||||
|
||||
`migration/wan-drill` force-faulted vyos001 and timed a real failover:
|
||||
|
||||
```
|
||||
TAKEOVER OK: vyos002 held the VIP and reached the internet in 52s
|
||||
FAILBACK OK: 36s
|
||||
```
|
||||
|
||||
vyos002 took the VIPs at t+18s and had **both** WANs by t+52s. Failback put
|
||||
vyos001 back with a WAN in 36s. Total interruption ≈ 88s across two deliberate
|
||||
transitions.
|
||||
|
||||
**The cloned-MAC lease transfers.** This was the largest untested item in the
|
||||
whole design — whether the 10 gig ISP would re-issue `87.192.101.48` to
|
||||
`f0:9f:c2:12:9b:4f` arriving on a different switch port. It did, same address,
|
||||
within the takeover window. That risk is now closed.
|
||||
|
||||
**Vodafone did not refuse the re-dial**, so its `session-control` behaves like
|
||||
`replace` rather than the hostile `deny`. `GRACE=300` was sized against the
|
||||
sim's 148s `deny` case and is therefore comfortable — but it should stay where
|
||||
it is, because one drill on one evening does not establish the ISP's policy
|
||||
under all conditions.
|
||||
|
||||
**Vodafone hands out a different IPv4 on every dial**: `83.106.5.72` →
|
||||
`90.251.153.180` (vyos002) → `90.251.142.103` (vyos001, after failback).
|
||||
Nothing may be pinned to the PPPoE address. The HE IPv6 tunnel is pinned to
|
||||
`87.192.101.48`, which is the **10 gig** (`bond0.53`) and stable across
|
||||
failover. Anything added later that hardcodes a WAN IP must use the 10 gig one,
|
||||
not `pppoe0`'s.
|
||||
|
||||
> **CORRECTED 2026-09-06.** This paragraph originally continued "so `tun0`
|
||||
> survived untouched and IPv6 stayed up at 15.5ms." **That conclusion was
|
||||
> wrong, and it should never have been recorded as a result.** The premise is
|
||||
> right — the endpoint address is stable — but it does not follow. During
|
||||
> takeover the reconciler disables `bond0.53` on the demoted box, so
|
||||
> `87.192.101.48` *leaves vyos001 and appears on vyos002*, and vyos002 has no
|
||||
> `tun0` at all: no tunnel, no `he-tunnel-follow`, no `/config/he-secrets`, no
|
||||
> VLAN 9 prefix, no `route6 ::/0`. Inbound protocol 41 from HE lands on a router
|
||||
> with nothing to decapsulate it. vyos001's own journal for the drill window
|
||||
> reads `08:36:01 he-tunnel-follow: no default route; refusing to guess`.
|
||||
>
|
||||
> The reading was taken either side of the window, not through it — `wan-drill`
|
||||
> contained **no IPv6 check of any kind**, and neither did any other part of the
|
||||
> mechanism. That is now fixed: the drill probes IPv6 in both timing loops and
|
||||
> asserts that a router-level failover makes **zero** HE API calls. Until a
|
||||
> drill produces that figure, the IPv6 behaviour of a failover is *unmeasured*,
|
||||
> not "fine".
|
||||
>
|
||||
> The gap itself was real: **IPv6 was single-homed on vyos001 while the WAN
|
||||
> beneath it was HA.** **Closed and measured 2026-09-06** — see the drill below.
|
||||
|
||||
Production takeover (52s) is about twice the sim's `replace` figure (26s), which
|
||||
is the expected direction: the VP2440s commit under kea, BGP and conntrack while
|
||||
the sim routers are idle.
|
||||
|
||||
## IPv6 follows the WAN — measured, 2026-09-06
|
||||
|
||||
The second drill of the day, run after IPv6-follows-master was deployed. This is
|
||||
the first time the IPv6 behaviour of a failover has been a **measurement** rather
|
||||
than an assertion:
|
||||
|
||||
```
|
||||
TAKEOVER OK: vyos002 held the VIP and reached the internet in 37s
|
||||
IPv6 followed in 37s (v4 37s, gap 0s)
|
||||
FAILBACK OK in 32s
|
||||
IPv6 back in 44s
|
||||
tun0 src : 87.192.101.48 -> 87.192.101.48 OK: unchanged across the drill
|
||||
HE updates from vyos001: 0
|
||||
HE updates from vyos002: 0
|
||||
```
|
||||
|
||||
Full log: `migration/drill-evidence/wan-drill-2026-09-06-ipv6.txt`.
|
||||
|
||||
**Zero HE API calls across a full takeover and failback.** This is the invariant
|
||||
the design rests on and it now has evidence: the 10 gig lease follows the cloned
|
||||
MAC, so the tunnel endpoint is the *same address* on whichever router holds the
|
||||
WAN, and there is nothing to tell Hurricane Electric. Only the within-box fall
|
||||
back to PPPoE needs an HE update.
|
||||
|
||||
**IPv6 is no longer the laggard, but the two directions are not symmetric.** On
|
||||
takeover it arrived in the same 5s sample as IPv4; on failback it trailed by 12s.
|
||||
That asymmetry is the reconciler's tick, not a fault: on promotion it enables
|
||||
`bond0.53` first, and `v6_take` only raises the tunnel once the source address
|
||||
actually exists, so it can land on the following 30s tick. Bound is one tick.
|
||||
Note the sampling granularity — the drill polls every 5s, so "gap 0s" means
|
||||
"within the same sample", not "simultaneous".
|
||||
|
||||
Takeover was 37s here against 52s in the morning drill. Do not read that as an
|
||||
IPv6 improvement; it is the same IPv4 mechanism on a different run, and the
|
||||
morning figure included the first-ever cloned-MAC lease transfer.
|
||||
|
||||
**Vodafone confirmed the every-dial-a-new-address behaviour again**: `pppoe0`
|
||||
came back as `90.251.152.236`, having been `90.251.142.103` before the drill.
|
||||
|
||||
## config.boot pins `vif 53 disable` on both — fixed 2026-09-06
|
||||
|
||||
The convention is that **both** `config.boot`s hold `vif 53 disable`, so a reboot
|
||||
in any order comes up unable to claim the cloned MAC and the reconciler enables
|
||||
it on whichever box holds the VIP. vyos001's did not; its `config.boot` predated
|
||||
this work.
|
||||
|
||||
There is no clean way to express "boot disabled, run enabled" in VyOS: **`save`
|
||||
writes the RUNNING config, not the candidate.** Setting the node, saving and
|
||||
discarding was tested in labsim and `config.boot` came back *without* `disable`,
|
||||
the WAN untouched. So the node must genuinely be disabled, saved, and
|
||||
re-enabled. `migration/vif53-pin-boot-disable` does exactly that, is idempotent,
|
||||
and no-ops on a box that already has it.
|
||||
|
||||
Cost, measured on vyos001: a **32s** window (10s in the sim — production commits
|
||||
under kea/BGP/conntrack are slower), `bond0.53` re-leased `87.192.101.48` 5s
|
||||
after re-enable, and `vyos-failover` restored the primary route about a minute
|
||||
later:
|
||||
|
||||
```
|
||||
09:32:23 ip route del 0.0.0.0/0 ... dev bond0.53
|
||||
09:32:32 Check fail for route 0.0.0.0/0 interface "bond0.53"
|
||||
09:33:23 ip route add 0.0.0.0/0 via 87.192.96.1 dev bond0.53 metric 1 proto failover
|
||||
```
|
||||
|
||||
The house rode `pppoe0` for that minute rather than losing the internet, which
|
||||
is the T5 path working. Note the shape of that recovery before reading a fresh
|
||||
`ip route show` as a regression: for ~60s after the bounce the default really is
|
||||
on `pppoe0`, because `vyos-failover` only re-adds the `bond0.53` route once its
|
||||
probes pass again.
|
||||
|
||||
**A race worth knowing about.** The first sim run collided with
|
||||
`vrrp-wan-reconcile`'s own commit — *"Configuration system temporarily locked due
|
||||
to another commit in progress"* — and the `save` landed while the **re-enable did
|
||||
not**, leaving the master with its 10 gig down. The script now takes the
|
||||
reconciler's `/run/vrrp-wan.lock` (which the reconciler skips a tick rather than
|
||||
block on), with `9>&-` so the config session's unionfs child cannot inherit it.
|
||||
Even the bad run ended correctly — the reconciler logged *"MASTER with bond0.53
|
||||
disabled -> enabling"* and repaired it in 4s — so a half-completed run is
|
||||
survivable by design. The script no longer leans on that, and verifies the
|
||||
re-enable rather than reporting a success it did not achieve.
|
||||
|
||||
**Rollback**, from either box: `set interfaces pppoe pppoe0 disable` on both and
|
||||
`rm /run/vrrp-wan/may-dial`. That restores today's behaviour exactly.
|
||||
|
||||
## What the sim cannot prove
|
||||
|
||||
- **Vodafone's `session-control`.** The matrix now brackets it properly.
|
||||
Destroying the master and timing the survivor's session:
|
||||
|
||||
| policy | takeover |
|
||||
|---|---|
|
||||
| `replace` (accel-ppp default) | 26s / 26s |
|
||||
| **`deny`** (hostile) | **148s / 141s** |
|
||||
| `disable` | 21s / 20s |
|
||||
|
||||
`deny` is the sizing case: the AC refuses the survivor until its own
|
||||
dead-peer timer frees the dead session, and the poller caught two dial
|
||||
attempts being rejected before one succeeded. `GRACE` is set from that — see
|
||||
`migration/vrrp-wan.conf`. This still cannot tell you which policy Vodafone
|
||||
runs, and account rate-limiting or lockout on repeated dials has no sim
|
||||
analogue at all; the flap damper (6 dials / 600s → 15 min hold-off) exists
|
||||
for that.
|
||||
|
||||
Treat the 148s as a floor rather than a worst case. These are idle 2-vCPU
|
||||
VMs, and the AC shares an OVS bridge with the routers, so `virsh destroy`
|
||||
removes the port and accel-ppp sees the peer physically vanish. A real BRAS
|
||||
reached over DSL does not learn that our router died — it waits out its own
|
||||
timers, which are longer and not ours to know.
|
||||
|
||||
Worth knowing how close this came to being missed: until 2026-09-06 the
|
||||
matrix set the policy with `vbash -c 'source script-template; configure;
|
||||
...; commit'`, which never starts a config session. `commit` failed to
|
||||
stderr, the helper discarded it, and all three iterations ran against the
|
||||
default while printing the mode they were supposedly testing. It reported
|
||||
`deny` at 25s. The real figure is 148s.
|
||||
- **Whether Vodafone honours our LCP Terminate / PADT** on a graceful stop.
|
||||
- ~~**The cloned-MAC lease.**~~ **Answered 2026-09-06**: the drill moved it and
|
||||
the ISP re-issued `87.192.101.48` to `f0:9f:c2:12:9b:4f` on vyos002's port
|
||||
within the takeover window. See "Proven in production" above.
|
||||
- ~~**Whether VyOS can dial Vodafone at all.**~~ **Answered 2026-09-06**: both
|
||||
routers dialled successfully during the drill. MTU/MSS under sustained load
|
||||
is still unmeasured, and Vodafone hands out a different IPv4 every dial.
|
||||
- **Timing under load.** The sim routers are idle 2-vCPU VMs; commit latency on
|
||||
the VP2440s under kea + BGP + conntrack will be worse, and commit latency is
|
||||
the dominant term in the `bond0.53` half of a failover.
|
||||
|
||||
## How the model handles the asymmetry — resolved
|
||||
|
||||
An earlier draft of this file said an override "must assert `vif 53 disable` on
|
||||
**both** routers". **That advice was wrong and has been removed**; do not
|
||||
reintroduce it. `vif 53 disable` is *runtime* state owned by
|
||||
`vrrp-wan-reconcile`, keyed on who holds the management VIP, so pinning it in
|
||||
the model would fight the reconciler on every apply and would briefly disable
|
||||
the live master's 10 gig each time.
|
||||
|
||||
What is actually done, and why it is safe:
|
||||
|
||||
- **Runtime (Pulumi): follow reality.** Run `npm run vyos:export && npm run
|
||||
vyos:render` immediately before any `pulumi up` touching vyos. Whichever
|
||||
router currently holds the WAN keeps it; the apply is a no-op on that node.
|
||||
There is deliberately **no** override for `vif 53 disable`.
|
||||
- **Boot (`config.boot`): hardcode safe.** Both routers pin `vif 53 disable`,
|
||||
so a reboot in any order comes up unable to claim the cloned MAC and the
|
||||
reconciler enables it on whoever holds the VIP. See the section above.
|
||||
- **Install time (PXE, nothing to follow).** `migration/vyos-mode-delta.py`
|
||||
emits `vif 53 disable` for a box with no WAN, and deliberately does *not*
|
||||
emit `pppoe pppoe0 disable`.
|
||||
|
||||
Verified 2026-09-06: `vyos:verify` reports both routers in sync, 533 and 512
|
||||
nodes, zero drift.
|
||||
194
migration/RECOVERY-CARD-vlan1-move.md
Normal file
194
migration/RECOVERY-CARD-vlan1-move.md
Normal file
@@ -0,0 +1,194 @@
|
||||
# Recovery card — moving Management to tagged VLAN 1
|
||||
|
||||
Print or keep open. **During this change there is no internet, so no Claude.**
|
||||
Everything you need is on this page.
|
||||
|
||||
---
|
||||
|
||||
## The one thing that matters
|
||||
|
||||
```
|
||||
ssh vyos@10.0.1.252
|
||||
```
|
||||
|
||||
Your workstation is `10.0.0.210/23`; vyos001's LoT leg is `10.0.1.252/23`. Same
|
||||
subnet, same VLAN, **direct L2** — verified: `ip route get` returns
|
||||
`dev lanbr0 src 10.0.0.210` with no `via`, MAC `64:62:66:25:96:45`.
|
||||
|
||||
It therefore does **not** depend on: the Management VLAN, VRRP, the VIPs,
|
||||
inter-VLAN routing, DNS, or the switch trunk config. If the router is up and its
|
||||
bond has link, this works. `bond0.10` is untouched by the change and stays in the
|
||||
firewall `LAN` group throughout.
|
||||
|
||||
vyos002, once it is up, is `10.0.1.253` the same way.
|
||||
|
||||
Other legs that also survive: `192.168.3.4` (kvm), `192.168.2.252` (Roomates).
|
||||
|
||||
---
|
||||
|
||||
## Before you touch anything
|
||||
|
||||
```
|
||||
ssh vyos@10.0.1.252
|
||||
sudo /config/vyos-known-good save
|
||||
```
|
||||
|
||||
The existing snapshot is from **2026-08-24** and predates today's fixes
|
||||
(eth2 removal, VRRP health-check) — restoring that one would undo them. Take a
|
||||
fresh one first. Check with `sudo /config/vyos-known-good status`.
|
||||
|
||||
---
|
||||
|
||||
## Order: switch FIRST, router SECOND
|
||||
|
||||
This matters and is easy to get backwards.
|
||||
|
||||
The UniFi controller is `192.168.1.5`, on the **Management** VLAN. Your
|
||||
workstation is on LoT and reaches it *through vyos001*. The moment the router
|
||||
has Management on `bond0.1` while the switch is still sending it untagged, that
|
||||
routing is dead — **and you lose the controller**, which is the thing you still
|
||||
need in order to change the switch.
|
||||
|
||||
So:
|
||||
|
||||
1. **UniFi first**, while everything still works:
|
||||
USW Aggregation → port 1 `firewall001` (LAG, members 1+2) →
|
||||
Native VLAN: Management → **None**, and make sure VLAN 1 is tagged/allowed.
|
||||
*vyos001 loses Management the instant this lands. That is expected.*
|
||||
Do **not** touch port 3 `firewall002` — that is vyos002, and it is down.
|
||||
2. **Router second**, over `ssh vyos@10.0.1.252` (still works — L2 direct).
|
||||
|
||||
If UniFi will not offer "no native VLAN", stop and read *"If UniFi cannot do it"*
|
||||
below rather than improvising.
|
||||
|
||||
---
|
||||
|
||||
## The router change
|
||||
|
||||
```
|
||||
ssh vyos@10.0.1.252
|
||||
configure
|
||||
set interfaces bonding bond0 vif 1 address '192.168.1.252/24'
|
||||
set interfaces bonding bond0 vif 1 description 'management'
|
||||
delete interfaces bonding bond0 address
|
||||
set firewall group interface-group LAN interface 'bond0.1'
|
||||
delete firewall group interface-group LAN interface 'bond0'
|
||||
set high-availability vrrp group native interface 'bond0.1'
|
||||
commit-confirm 10
|
||||
save
|
||||
exit
|
||||
```
|
||||
|
||||
**Use `commit-confirm 10`, not `commit`.** If it goes wrong and you cannot get
|
||||
back in, the router reverts itself after 10 minutes and comes back on its own.
|
||||
That is your safety net with no internet and no help.
|
||||
|
||||
Once you have confirmed it works (below), run:
|
||||
|
||||
```
|
||||
configure
|
||||
confirm
|
||||
save
|
||||
exit
|
||||
```
|
||||
|
||||
`save` after `confirm`, or a reboot loses it.
|
||||
|
||||
### Then, and this is the step that gets forgotten
|
||||
|
||||
```
|
||||
sudo systemctl restart isc-kea-dhcp4-server
|
||||
```
|
||||
|
||||
VyOS does **not** restart kea for an interface address change. Without this it
|
||||
keeps a raw socket bound to the old address and keeps handing out wrong-VLAN
|
||||
addresses — the fix looks like it did nothing. Give it ~60s before judging;
|
||||
kea reopens sockets on a retry loop and answers nothing for a while after a
|
||||
restart (measured: still silent at 55s in the sim, then fine).
|
||||
|
||||
Also check DNS came back, since the forwarder binds the VIP `192.168.1.1`:
|
||||
|
||||
```
|
||||
sudo systemctl status pdns-recursor --no-pager | head -3
|
||||
dig @192.168.1.1 google.com +short
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Verify
|
||||
|
||||
```
|
||||
ssh vyos@192.168.1.252 # Management back, now tagged
|
||||
show vrrp # native should be on bond0.1
|
||||
show dhcp server leases | head
|
||||
```
|
||||
|
||||
Then from a machine on VLAN 3, force a DHCP renew and confirm it gets a
|
||||
`192.168.3.x` address and not a `192.168.1.x` one.
|
||||
|
||||
---
|
||||
|
||||
## If you are locked out
|
||||
|
||||
In order:
|
||||
|
||||
1. **Wait 10 minutes.** `commit-confirm` reverts by itself. This is the answer
|
||||
most of the time. Do not power-cycle during this — you will lose the revert.
|
||||
2. `ssh vyos@10.0.1.252` — the LoT leg. Then `configure` / `rollback 1` / `commit`.
|
||||
3. Other legs: `ssh vyos@192.168.3.4`, `ssh vyos@192.168.2.252`.
|
||||
4. `sudo /config/vyos-known-good restore` — back to the snapshot you took at the
|
||||
start. It is itself commit-confirmed, so even this cannot strand you.
|
||||
5. Put the UniFi port back: Native VLAN → Management on USW Aggregation port 1.
|
||||
That alone restores the old shape and Management comes back untagged.
|
||||
|
||||
**Do not** power-cycle vyos001 as a first move. Everything above is faster and
|
||||
non-destructive, and a reboot loses an unsaved `commit-confirm` revert.
|
||||
|
||||
---
|
||||
|
||||
## Do NOT power on vyos002 yet
|
||||
|
||||
It still has `interfaces ethernet eth2 address 192.168.8.144/23` on the box — the
|
||||
same subnet as `bond0.2`. That is what ARP-poisoned `192.168.8.1` and took the
|
||||
cluster down. It also has no `/config/vrrp-wan-health`, so it can take the
|
||||
floating IPs with no WAN.
|
||||
|
||||
Its console (`kvm - vyos002`, US24 port 9) is currently **unreachable** — it sits
|
||||
on a VLAN 3 port holding a Management lease `192.168.1.28`, which is the very bug
|
||||
being fixed here. Fixing DHCP first is what gets that console back.
|
||||
|
||||
---
|
||||
|
||||
## If UniFi cannot do it
|
||||
|
||||
Classic UniFi (this is a classic controller, 10.4.57) may not offer
|
||||
"Native VLAN = None" — every switch port has a PVID. Two things make this awkward
|
||||
here: Management is UniFi's *default* network with **no VLAN ID at all**
|
||||
(`vlan: null`), so there may be nothing to "tag VLAN 1" with.
|
||||
|
||||
If so, **stop and change nothing.** The workaround is to point the trunk's native
|
||||
VLAN at a VLAN the router does not serve (so `bond0` still ends up with no
|
||||
subnet), which needs a throwaway VLAN-only network created first. That is a
|
||||
design decision, not something to improvise at 1am with no internet. Put the port
|
||||
back to Native = Management and everything returns to today's working state.
|
||||
|
||||
---
|
||||
|
||||
## Facts worth having on paper
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| vyos001 Management | `192.168.1.252` → becomes `bond0.1` |
|
||||
| vyos001 LoT (recovery) | `10.0.1.252`, L2-direct from your workstation |
|
||||
| vyos002 Management | `192.168.1.253` (down) |
|
||||
| VIP Management | `192.168.1.1` |
|
||||
| UniFi controller | `192.168.1.5` (on Management — you lose it mid-change) |
|
||||
| firewall001 trunk | USW Aggregation port 1, LAG members 1+2 |
|
||||
| firewall002 trunk | USW Aggregation port 3, LAG members 3+4 |
|
||||
| SSH user / pass | `vyos` / `vyos` |
|
||||
| vyos001 bond MAC | `64:62:66:25:96:45` |
|
||||
|
||||
Measured in labsim: converting the router while the peer is already converted
|
||||
costs **0s** of VIP downtime; converting it while it holds the VIPs costs about
|
||||
**6s**. vyos002 is down, so vyos001 holds everything — expect the ~6s, and expect
|
||||
Management to be gone from the UniFi change until the router change lands.
|
||||
99
migration/RECOVERY-CARD-wan-panic.md
Normal file
99
migration/RECOVERY-CARD-wan-panic.md
Normal file
@@ -0,0 +1,99 @@
|
||||
# RECOVERY CARD — internet is down after a WAN failover
|
||||
|
||||
No internet means no Claude. Everything here runs from the routers themselves.
|
||||
**Print this or keep it on a phone.**
|
||||
|
||||
## 1. Get to a router
|
||||
|
||||
The LoT leg is L2-direct on `bond0.10`. It survives Management, VRRP and
|
||||
routing being broken:
|
||||
|
||||
```
|
||||
ssh vyos@10.0.1.252 # vyos001 (normally MASTER, has the WAN)
|
||||
ssh vyos@10.0.1.253 # vyos002 (normally BACKUP, has nothing)
|
||||
```
|
||||
|
||||
Password is the usual one. If SSH is dead, use the JetKVM consoles.
|
||||
|
||||
## 2. See what is going on
|
||||
|
||||
```
|
||||
sudo /config/wan-panic status
|
||||
```
|
||||
|
||||
Run it on both. You want exactly ONE box saying `holds VIP : YES`, and that
|
||||
same box showing a `WAN` line and a `route`.
|
||||
|
||||
| what you see | what it means |
|
||||
|---|---|
|
||||
| one box `YES` with WAN + route | healthy, look elsewhere for the fault |
|
||||
| one box `YES`, **no WAN**, no route | the failover half-worked — go to §3 |
|
||||
| **both** `YES` | VRRP split — go to §3, run it on vyos002 |
|
||||
| **neither** `YES` | both faulted — go to §4 |
|
||||
|
||||
## 3. Give the WAN back to vyos001
|
||||
|
||||
**Run this on vyos002 (`10.0.1.253`).** This is the one that matters — vyos002
|
||||
standing down is what lets vyos001 take over.
|
||||
|
||||
```
|
||||
sudo /config/wan-panic
|
||||
```
|
||||
|
||||
Wait ~30s. Then on vyos001 (`10.0.1.252`):
|
||||
|
||||
```
|
||||
sudo /config/wan-panic status
|
||||
```
|
||||
|
||||
Expect `holds VIP : YES` and a `WAN` line with `bond0.53=…` and/or `pppoe0=…`.
|
||||
|
||||
Once the house is back online and you want vyos002 to be a standby again:
|
||||
|
||||
```
|
||||
sudo rm /run/vrrp-wan/force-fault # on vyos002
|
||||
```
|
||||
|
||||
Leave it set if you would rather have no standby than any more surprises —
|
||||
that is the pre-2026-09-06 arrangement and the house runs fine on it.
|
||||
|
||||
## 4. Stop the mechanism touching anything
|
||||
|
||||
If the WAN keeps moving, or you do not trust the automation:
|
||||
|
||||
```
|
||||
sudo /config/wan-panic undo # on BOTH routers
|
||||
```
|
||||
|
||||
Stops the reconcile and guard timers. Whatever WAN is up **stays** up. Nothing
|
||||
will move it again until you re-enable the timers:
|
||||
|
||||
```
|
||||
sudo systemctl enable --now vrrp-wan-reconcile.timer vrrp-wan-guard.timer
|
||||
```
|
||||
|
||||
## 5. Nuclear — put the config back
|
||||
|
||||
Only if the config itself is wrong. **This reboots the router.**
|
||||
|
||||
```
|
||||
sudo /config/vyos-known-good restore # on BOTH routers
|
||||
```
|
||||
|
||||
The pinned config is from 2026-09-06, immediately before the PPPoE HA rollout:
|
||||
1112 lines on vyos001, 1069 on vyos002.
|
||||
|
||||
## Why the WAN can only be on one box
|
||||
|
||||
One ISP account each way. The 10 gig lease is bound to a cloned MAC
|
||||
(`f0:9f:c2:12:9b:4f`, the old USG's) and Vodafone is a single-session PPPoE
|
||||
credential. Two routers holding either at once is worse than one holding
|
||||
neither — that is why the safe state is "vyos001 has it, vyos002 is inert".
|
||||
|
||||
## Known gap
|
||||
|
||||
vyos001's `config.boot` does not carry `vif 53 disable`. If vyos001 reboots
|
||||
**while vyos002 is master**, the cloned MAC is briefly live on both. It clears
|
||||
itself within 30s (the reconciler disables it on the non-master). If you see
|
||||
MAC flapping on the WAN switch port right after a vyos001 reboot, that is this,
|
||||
and it will stop on its own.
|
||||
92
migration/VYOS-IMAGE-UPGRADE.md
Normal file
92
migration/VYOS-IMAGE-UPGRADE.md
Normal file
@@ -0,0 +1,92 @@
|
||||
# Upgrading a VyOS image on the router pair
|
||||
|
||||
VyOS keeps `/config` across an image upgrade and **replaces `/etc`**. Everything
|
||||
in `/config` survives; anything the WAN mechanism put in `/etc` does not.
|
||||
|
||||
## What an upgrade silently removes
|
||||
|
||||
`/etc/systemd/system/ppp@pppoe0.service.d/10-vrrp-wan-gate.conf` — the gate.
|
||||
|
||||
That drop-in is the only thing stopping the **backup** from dialling. With
|
||||
`pppoe0` enabled on both routers (which it is, by design), `interfaces_pppoe.py`
|
||||
restarts ppp on every commit touching the pppoe subtree when the daemon is not
|
||||
running. Without the gate, the backup dials on the next `pulumi up`, hand
|
||||
commit, or boot-time config load — and Vodafone is a single-session account, so
|
||||
it takes the session off the live master.
|
||||
|
||||
`vrrp-wan-reconcile` fails **closed** here: it refuses to dial at all when the
|
||||
drop-in is missing, and logs
|
||||
|
||||
```
|
||||
REFUSING to dial: gate drop-in ... is missing (VyOS upgrade?)
|
||||
```
|
||||
|
||||
so the symptom is "PPPoE never comes up", not "both routers dialled". That is
|
||||
the safe direction, but it does mean an upgraded router has no PPPoE until the
|
||||
gate is reinstalled.
|
||||
|
||||
The systemd **units and timers** also live in `/etc` and go the same way.
|
||||
|
||||
## Do this, one router at a time
|
||||
|
||||
Never both at once — the surviving router must be able to hold the VIPs.
|
||||
|
||||
1. **Upgrade the BACKUP first.** Confirm which it is:
|
||||
```
|
||||
sudo /config/wan-panic status # "holds VIP : no"
|
||||
```
|
||||
2. Install the image and reboot as normal.
|
||||
3. **Reinstall the mechanism** from a checkout of `lab`:
|
||||
```
|
||||
migration/vrrp-wan-install --vip 192.168.1.1 --host vyos@<router>
|
||||
migration/vrrp-wan-install --check --host vyos@<router> # must be clean
|
||||
```
|
||||
4. **Verify the gate is shut and nothing dialled:**
|
||||
```
|
||||
systemctl show ppp@pppoe0 -p ConditionResult -p ActiveState -p NRestarts
|
||||
```
|
||||
Want `ConditionResult=no`, `ActiveState=inactive`, `NRestarts=0`. Check the
|
||||
wire too, not just state: `sudo tcpdump -i bond0.51 -nn pppoed` — no PADI.
|
||||
5. Confirm it reaches **BACKUP**, not FAULT:
|
||||
```
|
||||
show vrrp
|
||||
```
|
||||
FAULT on every group means the health check is failing — most likely
|
||||
`/config/vrrp-wan-health` did not get reinstalled, or `vrrp-wan.conf` is
|
||||
missing so `GRACE` and the VIP fall back to defaults.
|
||||
5b. **Check IPv6 came back with it.** `vrrp-wan-install` now carries
|
||||
`he-tunnel-follow`, so `--check` covers it, but `/config/he-secrets` is a
|
||||
secret placed by Pulumi and is only checked for *presence*:
|
||||
```
|
||||
sudo /config/he-tunnel-follow status # role, tunnel src, MTU
|
||||
```
|
||||
Want the box's own role, and — on the master — a tunnel source equal to the
|
||||
**10 gig** address with MTU 1480. `/config` survives an upgrade, so the
|
||||
`system task-scheduler` entry that runs this every minute survives too; it is
|
||||
the units and the ppp gate in `/etc` that do not.
|
||||
6. Confirm `config.boot` still pins the safe resting state:
|
||||
```
|
||||
sudo /config/vif53-pin-boot-disable --check # config.boot disable : 1
|
||||
```
|
||||
7. Only once the upgraded box is a healthy BACKUP, fail over and repeat for the
|
||||
other router. `migration/wan-drill` does that unattended, or by hand:
|
||||
`sudo /config/wan-panic` on the box you want to give up mastership.
|
||||
|
||||
## After both are done
|
||||
|
||||
```
|
||||
migration/vrrp-wan-install --check --host vyos@10.0.1.252
|
||||
migration/vrrp-wan-install --check --host vyos@10.0.1.253
|
||||
cd kubernetes-deployment && npm run vyos:export -- --router vyos001=10.0.1.252 --router vyos002=10.0.1.253
|
||||
npm run vyos:render && npm run vyos:verify -- --router vyos001=10.0.1.252 --router vyos002=10.0.1.253
|
||||
```
|
||||
|
||||
Both should read "in sync". If the export shows `interfaces pppoe pppoe0
|
||||
disable` coming back, something reinstated it — the
|
||||
`pppoe-gated-not-config-disabled` override exists to prevent exactly that, so
|
||||
check it is still in `infra/vyos/subtrees/overrides.json`.
|
||||
|
||||
## If it goes wrong
|
||||
|
||||
`migration/RECOVERY-CARD-wan-panic.md`, or `/config/RECOVERY-CARD.md` on either
|
||||
router. Short version, on vyos002: `sudo /config/wan-panic`.
|
||||
38
migration/drill-evidence/wan-drill-2026-09-06-ipv6.txt
Normal file
38
migration/drill-evidence/wan-drill-2026-09-06-ipv6.txt
Normal file
@@ -0,0 +1,38 @@
|
||||
LOG=/tmp/claude-1000/-home-michal-developer-michalzxc-claude-lab/d66e2cbb-c178-46d7-a628-0ecb17b48a09/scratchpad/wan-drill-150954.log
|
||||
15:09:54 === pre-flight ===
|
||||
15:09:54 holder now : 10.0.1.252
|
||||
15:09:54 10.0.1.252 WAN : bond0.53=87.192.101.48 pppoe0=90.251.142.103
|
||||
15:09:55 10.0.1.253 WAN :
|
||||
15:09:55 internet : UP via 10.0.1.252
|
||||
15:09:56 10.0.1.252 tun0 src : 87.192.101.48
|
||||
15:09:56 10.0.1.253 tun0 src : 87.192.101.48
|
||||
15:09:56 IPv6 : UP via 10.0.1.252
|
||||
15:09:56 === arming auto-abort on 10.0.1.253 ===
|
||||
watchdog armed (pid 1014868): stand down after 150s holding the VIP with no WAN
|
||||
15:09:56 === DRILL: force-faulting 10.0.1.252 ===
|
||||
15:10:03 t+7s holder=10.0.1.252 vyos002_wan=[] v6=down
|
||||
15:10:09 t+13s holder=10.0.1.252 vyos002_wan=[] v6=down
|
||||
15:10:15 t+19s holder=10.0.1.253 vyos002_wan=[] v6=down
|
||||
15:10:21 t+25s holder=10.0.1.253 vyos002_wan=[] v6=down
|
||||
15:10:33 t+37s holder=10.0.1.253 vyos002_wan=[bond0.53=87.192.101.48 ] v6=up
|
||||
15:10:33 *** TAKEOVER OK: 10.0.1.253 held the VIP and reached the internet in 37s ***
|
||||
15:10:33 *** IPv6 followed in 37s (v4 37s, gap 0s) ***
|
||||
15:10:33 === failing back to 10.0.1.252 ===
|
||||
15:10:46 t+13s holder=10.0.1.253 vyos001_wan=[] v6=down
|
||||
15:10:53 t+20s holder=10.0.1.253 vyos001_wan=[] v6=down
|
||||
15:10:59 t+26s holder=10.0.1.252 vyos001_wan=[] v6=down
|
||||
15:11:04 t+31s holder=10.0.1.252 vyos001_wan=[bond0.53=87.192.101.48 ] v6=down
|
||||
15:11:11 t+38s holder=10.0.1.252 vyos001_wan=[bond0.53=87.192.101.48 ] v6=down
|
||||
15:11:17 t+44s holder=10.0.1.252 vyos001_wan=[bond0.53=87.192.101.48 ] v6=up
|
||||
15:11:17 *** FAILBACK OK in 32s ***
|
||||
15:11:17 *** IPv6 back in 44s ***
|
||||
15:11:17 === IPv6 invariants ===
|
||||
15:11:17 tun0 src : 87.192.101.48 -> 87.192.101.48
|
||||
15:11:17 OK: tunnel source unchanged across the drill
|
||||
15:11:18 HE updates from 10.0.1.252 since 2026-09-06 15:09:54: 0
|
||||
15:11:18 HE updates from 10.0.1.253 since 2026-09-06 15:09:54: 0
|
||||
15:11:18 --- cleanup (always runs) ---
|
||||
15:11:39 final holder : 10.0.1.252
|
||||
15:11:40 final WAN : 10.0.1.252 [bond0.53=87.192.101.48 pppoe0=90.251.152.236 ] 10.0.1.253 []
|
||||
15:11:40 final internet: UP via 10.0.1.252
|
||||
15:11:40 log: /tmp/claude-1000/-home-michal-developer-michalzxc-claude-lab/d66e2cbb-c178-46d7-a628-0ecb17b48a09/scratchpad/wan-drill-150954.log
|
||||
181
migration/he-tunnel-follow
Executable file
181
migration/he-tunnel-follow
Executable file
@@ -0,0 +1,181 @@
|
||||
#!/bin/bash
|
||||
# Keep the Hurricane Electric 6in4 tunnel pointed at whichever WAN is live.
|
||||
#
|
||||
# The tunnel is anchored to a source IPv4. When failover moves the default route
|
||||
# from the 10 gig to PPPoE, 6in4 packets keep leaving with the old source, HE
|
||||
# drops them, and IPv6 goes dark while IPv4 keeps working -- a partial outage
|
||||
# that presents as "some sites are broken", which is far worse to diagnose than
|
||||
# a clean one.
|
||||
#
|
||||
# Installed on BOTH routers and gated on VRRP mastership: the backup exits
|
||||
# immediately, and vrrp-wan-reconcile brings tun0 up and calls this script the
|
||||
# moment it takes the VIP. The HE endpoint itself needs no update when the WAN
|
||||
# moves between routers -- 87.192.101.48 is the 10 gig lease bound to the cloned
|
||||
# MAC, so it follows the VIP to the other box unchanged (proven by the 2026-09-06
|
||||
# drill). HE only has to be told about the WITHIN-box fall back to PPPoE.
|
||||
#
|
||||
# Changes are made at KERNEL level (`ip tunnel change`), not in VyOS config, on
|
||||
# purpose:
|
||||
# - no commit per WAN flip, so a flapping line cannot churn the config;
|
||||
# - no drift against the Pulumi model, so `vyos-verify` stays meaningful;
|
||||
# - a reboot restores config.boot, which pins the 10 gig -- the correct
|
||||
# default -- so the wrong state cannot survive a restart.
|
||||
#
|
||||
# he-tunnel-follow status what is live vs what should be (read-only)
|
||||
# he-tunnel-follow run reconcile, updating HE if the source changed
|
||||
# he-tunnel-follow run --dry say what it would do, change nothing
|
||||
#
|
||||
# Credentials in /config/he-secrets (0600), NOT in git:
|
||||
# HE_USER=<tunnelbroker username>
|
||||
# HE_UPDATE_KEY=<from the tunnel's Advanced tab -- replaces the account password>
|
||||
# HE_TUNNEL_ID=<numeric tunnel id>
|
||||
set -uo pipefail
|
||||
|
||||
TUNNEL="${TUNNEL:-tun0}"
|
||||
SECRETS="${SECRETS:-/config/he-secrets}"
|
||||
STATE="${STATE:-/run/he-tunnel-follow.state}"
|
||||
ROLE_STATE="${ROLE_STATE:-/run/he-tunnel-follow.role}"
|
||||
|
||||
# HE's update endpoint, as a variable so labsim can point it at a stub. The sim
|
||||
# has no public IPv4 and no HE account, which is the whole reason the tunnel was
|
||||
# never rehearsed; with this the sim can exercise the HE-side half too.
|
||||
HE_UPDATE_URL="${HE_UPDATE_URL:-https://ipv4.tunnelbroker.net/nic/update}"
|
||||
|
||||
# This script is installed on BOTH routers -- the same principle as the PPPoE
|
||||
# gate: configured identically everywhere, gated at runtime. So it must know
|
||||
# when it is the backup. Left ungated, the backup copy either dies on "no
|
||||
# default route" every minute, or, far worse, sees its own idle pppoe0 address
|
||||
# and points the HE endpoint at it. Vodafone hands out a different IPv4 on every
|
||||
# dial, so that is an IPv6 blackhole plus a wasted write against a rate-limited
|
||||
# API -- and it would fire on the backup, where nobody is looking.
|
||||
#
|
||||
# The VIP comes from the same /config/vrrp-wan.conf the reconciler and the
|
||||
# health check read, so there is exactly one definition of "master" on the box.
|
||||
WAN_CONF="${WAN_CONF:-/config/vrrp-wan.conf}"
|
||||
# shellcheck disable=SC1090
|
||||
[ -r "$WAN_CONF" ] && . "$WAN_CONF"
|
||||
VIP="${VRRP_WAN_VIP:-192.168.1.1}"
|
||||
# 6in4 costs 20 bytes. The 10 gig path is 1500 -> 1480; PPPoE is 1492 -> 1472.
|
||||
# Getting this wrong is the classic "IPv6 works until something large" failure.
|
||||
declare -A WAN_MTU=( ["bond0.53"]=1480 ["pppoe0"]=1472 )
|
||||
# Require the same answer twice before acting. HE rate-limits updates, and a
|
||||
# flapping WAN would otherwise hammer the API exactly when it is needed most.
|
||||
HYSTERESIS="${HYSTERESIS:-2}"
|
||||
|
||||
log() { logger -t he-tunnel-follow -- "$*"; printf ' %s\n' "$*"; }
|
||||
die() { logger -t he-tunnel-follow -p user.err -- "$*"; printf ' ERROR: %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
holds_vip() { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; }
|
||||
active_wan() { ip -4 route show default 2>/dev/null | awk '/^default/{for(i=1;i<=NF;i++) if($i=="dev") print $(i+1); exit}'; }
|
||||
addr_of() { ip -4 -br addr show "$1" 2>/dev/null | awk '{print $3}' | cut -d/ -f1; }
|
||||
tunnel_src() { ip tunnel show "$TUNNEL" 2>/dev/null | sed -nE 's/.* local ([0-9.]+).*/\1/p'; }
|
||||
tunnel_mtu() { cat "/sys/class/net/$TUNNEL/mtu" 2>/dev/null; }
|
||||
|
||||
# HE's dyndns-style endpoint. `myip` is passed EXPLICITLY rather than letting HE
|
||||
# infer it from the request source: mid-failover the request itself may egress
|
||||
# either line, and inferring would happily point the tunnel at the WAN we just
|
||||
# left.
|
||||
he_update() {
|
||||
local ip="$1"
|
||||
[ -r "$SECRETS" ] || die "no $SECRETS -- create it with HE_USER / HE_UPDATE_KEY / HE_TUNNEL_ID (0600)"
|
||||
# shellcheck disable=SC1090
|
||||
. "$SECRETS"
|
||||
[ -n "${HE_USER:-}" ] && [ -n "${HE_UPDATE_KEY:-}" ] && [ -n "${HE_TUNNEL_ID:-}" ] \
|
||||
|| die "$SECRETS is missing HE_USER, HE_UPDATE_KEY or HE_TUNNEL_ID"
|
||||
|
||||
local out
|
||||
out="$(curl -sS --max-time 25 \
|
||||
--data-urlencode "username=$HE_USER" \
|
||||
--data-urlencode "password=$HE_UPDATE_KEY" \
|
||||
--data-urlencode "hostname=$HE_TUNNEL_ID" \
|
||||
--data-urlencode "myip=$ip" \
|
||||
"$HE_UPDATE_URL" 2>&1)"
|
||||
# dyndns protocol: "good <ip>" or "nochg <ip>" are both success.
|
||||
case "$out" in
|
||||
good*|nochg*) log "HE endpoint set to $ip ($out)"; return 0 ;;
|
||||
*) die "HE update refused: $out" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Log only when the role CHANGES. On a 1-minute timer an unconditional line
|
||||
# would be 1440 entries a day on the backup, which is how a real message gets
|
||||
# lost. The marker lives in /run, so a reboot re-announces the role once.
|
||||
note_role() {
|
||||
local role="$1" last=""
|
||||
[ -r "$ROLE_STATE" ] && read -r last < "$ROLE_STATE"
|
||||
[ "$last" = "$role" ] && return 0
|
||||
echo "$role" > "$ROLE_STATE"
|
||||
log "role is now $role"
|
||||
}
|
||||
|
||||
reconcile() {
|
||||
local dry="${1:-}"
|
||||
local wan src want_mtu cur_src cur_mtu
|
||||
|
||||
# The backup owns nothing here. vrrp-wan-reconcile holds tun0 down on this box
|
||||
# and will run this script itself the moment it takes the VIP, so there is
|
||||
# nothing to do and nothing to say.
|
||||
if ! holds_vip; then
|
||||
note_role backup
|
||||
rm -f "$STATE" # start a promoted box with a clean hysteresis count
|
||||
return 0
|
||||
fi
|
||||
note_role master
|
||||
|
||||
wan="$(active_wan)"; [ -n "$wan" ] || die "no default route; refusing to guess"
|
||||
src="$(addr_of "$wan")"; [ -n "$src" ] || die "no IPv4 address on $wan"
|
||||
want_mtu="${WAN_MTU[$wan]:-}"
|
||||
[ -n "$want_mtu" ] || die "unknown WAN '$wan' -- add it to WAN_MTU rather than guessing an MTU"
|
||||
cur_src="$(tunnel_src)"; cur_mtu="$(tunnel_mtu)"
|
||||
|
||||
if [ "$cur_src" = "$src" ] && [ "$cur_mtu" = "$want_mtu" ]; then
|
||||
rm -f "$STATE"
|
||||
log "in sync: $TUNNEL via $wan src $src mtu $cur_mtu"
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Hysteresis: count consecutive runs agreeing on the same target.
|
||||
local seen=0 last=""
|
||||
[ -r "$STATE" ] && { read -r last seen < "$STATE"; }
|
||||
if [ "$last" = "$src" ]; then seen=$((seen + 1)); else seen=1; fi
|
||||
echo "$src $seen" > "$STATE"
|
||||
if [ "$seen" -lt "$HYSTERESIS" ]; then
|
||||
log "change seen ($cur_src -> $src) but waiting for stability ($seen/$HYSTERESIS)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ "$dry" = "--dry" ]; then
|
||||
log "DRY RUN: would set HE endpoint to $src, then $TUNNEL local $src mtu $want_mtu"
|
||||
return 0
|
||||
fi
|
||||
|
||||
# HE first, then local. Either order costs a brief drop, but changing locally
|
||||
# first guarantees HE discards our packets for the whole window.
|
||||
he_update "$src" || return 1
|
||||
sudo ip tunnel change "$TUNNEL" mode sit local "$src" || die "failed to set tunnel local address"
|
||||
sudo ip link set "$TUNNEL" mtu "$want_mtu" || die "failed to set tunnel MTU"
|
||||
rm -f "$STATE"
|
||||
log "moved $TUNNEL to $wan: src $cur_src -> $src, mtu $cur_mtu -> $want_mtu"
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
status)
|
||||
wan="$(active_wan)"
|
||||
printf ' role : %s (vip %s)\n' "$(holds_vip && echo master || echo backup)" "$VIP"
|
||||
printf ' active WAN : %s\n' "${wan:-<none>}"
|
||||
printf ' wan addr : %s\n' "$(addr_of "${wan:-lo}")"
|
||||
printf ' tunnel : %s\n' "$(ip -br link show "$TUNNEL" 2>/dev/null | awk '{print $2}' || echo '<absent>')"
|
||||
printf ' tunnel src : %s\n' "$(tunnel_src)"
|
||||
# `${WAN_MTU[$wan]}` with an EMPTY subscript is a hard bash error --
|
||||
# "bad array subscript" -- not an empty expansion, and the :- default never
|
||||
# gets a chance to apply. A backup router has no default route, so `wan` is
|
||||
# empty there and `status` printed an error line on exactly the box whose
|
||||
# state you most need to read. Only index the array once there is a key.
|
||||
printf ' tunnel mtu : %s (want %s)\n' "$(tunnel_mtu)" \
|
||||
"$([ -n "${wan:-}" ] && echo "${WAN_MTU[$wan]:-?}" || echo '- (no WAN; this box is not master)')"
|
||||
printf ' he endpoint: %s\n' "$HE_UPDATE_URL"
|
||||
[ -r "$SECRETS" ] && printf ' credentials: present\n' || printf ' credentials: MISSING (%s)\n' "$SECRETS"
|
||||
;;
|
||||
run) reconcile "${2:-}" ;;
|
||||
*) die "usage: he-tunnel-follow {status|run [--dry]}" ;;
|
||||
esac
|
||||
50
migration/ppp-vrrp-gate.conf
Normal file
50
migration/ppp-vrrp-gate.conf
Normal file
@@ -0,0 +1,50 @@
|
||||
# Installed to /etc/systemd/system/ppp@pppoe0.service.d/10-vrrp-wan-gate.conf
|
||||
#
|
||||
# This drop-in is the ONLY thing preventing both routers from dialling the one
|
||||
# ISP credential at the same time. Do not remove it without reading this.
|
||||
#
|
||||
# pppoe0 is configured identically and ENABLED on both routers, because the
|
||||
# alternative -- `set interfaces pppoe pppoe0 disable` -- unlinks
|
||||
# /etc/ppp/peers/pppoe0 (interfaces_pppoe.py treats `disable` and `delete`
|
||||
# identically), and pppd's options file IS that path. A promotion then had to
|
||||
# re-render it via a full config commit at priority 322, where one unrelated
|
||||
# invalid node fails the whole commit and takes the 10 gig down with it. It also
|
||||
# made op-mode `connect interface pppoe0` unusable, since that refuses when the
|
||||
# peers file is absent.
|
||||
#
|
||||
# With the node enabled, interfaces_pppoe.py's apply() does this on EVERY commit
|
||||
# that touches the pppoe subtree:
|
||||
#
|
||||
# if not is_systemd_service_running('ppp@pppoe0.service') or shutdown_required:
|
||||
# call('systemctl restart ppp@pppoe0.service')
|
||||
#
|
||||
# -- i.e. the backup actively tries to dial whenever anything commits. A
|
||||
# `pulumi up`, a `sim-net-apply.sh apply`, or the boot-time config load are all
|
||||
# that commit. This gate is what makes that a no-op.
|
||||
#
|
||||
# /run is tmpfs, so the gate is shut at boot on both boxes and neither can dial
|
||||
# before VRRP has decided. ppp@.service is already After=vyos-router.service, so
|
||||
# no extra ordering is needed.
|
||||
[Unit]
|
||||
# Both must hold; multiple ConditionPathExists are ANDed.
|
||||
# may-dial -- vrrp-wan-reconcile has blessed this box (a renewed lease)
|
||||
# /etc/ppp/peers -- refuse to start pppd against a missing options file, which
|
||||
# is what produced a restart loop of 47 and counting on
|
||||
# 2026-09-05. A failed Condition is NOT a failure: the job
|
||||
# succeeds, the unit stays inactive, and `systemctl start`
|
||||
# exits 0 -- so callers must check is-active, never rc.
|
||||
ConditionPathExists=/run/vrrp-wan/may-dial
|
||||
ConditionPathExists=/etc/ppp/peers/pppoe0
|
||||
|
||||
# Belt to that brace. The stock unit is Restart=on-failure/RestartSec=5s against
|
||||
# systemd's default StartLimitIntervalSec=10s/Burst=5 -- two restarts per window,
|
||||
# so the limiter can never trip and a doomed pppd retries for ever.
|
||||
StartLimitIntervalSec=600
|
||||
StartLimitBurst=6
|
||||
|
||||
[Service]
|
||||
RestartSec=15
|
||||
# A hung pppd must be resolved inside the failover budget. The stock 90s means a
|
||||
# demoted router could still hold the session while the new master is dialling.
|
||||
# 20s still allows a clean LCP Terminate + PADT in the normal case.
|
||||
TimeoutStopSec=20
|
||||
188
migration/pulumi-override-he-tunnel-both.json
Normal file
188
migration/pulumi-override-he-tunnel-both.json
Normal file
@@ -0,0 +1,188 @@
|
||||
{
|
||||
"id": "he-ipv6-tunnel",
|
||||
"reason": "6in4 tunnel to Hurricane Electric, bringing 2001:470:187e::/48 in, on BOTH routers. SUPERSEDES he-ipv6-tunnel-vyos001, whose reason said 'vyos002 carries bond0.53 disabled, so the source address does not exist there and the tunnel would simply stay down -- adding it there is for a later takeover story, not now.' Both halves of that expired on 2026-09-06. `vif 53 disable` is no longer a property of vyos002: it is RUNTIME state owned by vrrp-wan-reconcile, keyed on who holds the management VIP, and deliberately absent from this model (see pppoe-gated-not-config-disabled and PPPOE-HA.md). And the takeover story shipped -- a drill moved the WAN and back, 52s and 36s. IPv6 did not follow it, so every failover took the whole v6 estate down for as long as vyos002 held the VIP. WHY THIS IS CHEAP: 87.192.101.48 is the 10 gig lease bound to the cloned MAC, and the drill proved the ISP re-issues THE SAME address to that MAC on the other router's port. The tunnel endpoint is therefore stable across the pair, so a router-level failover needs no HE API call at all -- only the tunnel present on both boxes and live on exactly one. HE updates remain solely for the within-box fall back to PPPoE, which /config/he-tunnel-follow already handles. WHY THE RUNTIME GATE IS LOAD-BEARING, not a nicety: measured in labsim 2026-09-06, VyOS ACCEPTS a tunnel whose source-address does not exist on the box (commit rc=0) and brings the link UP anyway. It is a blackhole that will happily attract the v6 default route -- not the inert node the old reason assumed. vrrp-wan-reconcile holds tun0 down on the backup and brings it up on the master, at kernel level, with no commit in the failover path. Verified in the sim: backup tun=DOWN radvd=inactive, master tun=UP radvd=active. MTU 1480, not 1500: 6in4 adds a 20-byte outer IPv4 header. Leave it at 1500 and IPv6 appears to work while large transfers hang. The RA link-mtu is 1472 -- the PPPoE figure -- deliberately, on BOTH routers: it cannot be reconciled at runtime because it needs a commit, so advertise the lower of the two paths and be correct on either WAN. Production previously pinned 1480 and was silently wrong whenever the WAN fell back. bond0.9 takes ::1 on vyos001 and ::2 on vyos002 -- NOT the same address: SLAAC hosts take their gateway from the advertising router's link-local, so the global address need not move, and duplicating it would only produce a DAD conflict. default-preference is high on vyos001 and low on vyos002 so that if both ever advertise at once -- radvd's config is rendered into /run and a booting backup starts it before VRRP has decided -- hosts prefer the normal master, while a genuinely dead vyos001 still leaves vyos002 as the only router on the link. No firewall change is needed: the IPv6 firewall accepts only from interface-group LAN, so tun0 is untrusted by default. It is already default-deny on BOTH routers -- that ordering held. REHEARSAL: labsim/labsim-he-endpoint.sh now builds a fake HE endpoint and stub tunnelbroker API, closing the 'the sim has no public IPv4 and no HE endpoint' gap the old reason cited as why this was never tested. labsim/labsim-ipv6-ha-test.sh is the matrix. Mechanism proven there; the end-to-end v6 datapath is not yet, because the sim's inter-island transit crosses libvirt NAT -- see that file's KNOWN SIM GAP header.",
|
||||
"set": [
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"tunnel",
|
||||
"tun0",
|
||||
"encapsulation"
|
||||
],
|
||||
"value": "sit"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"tunnel",
|
||||
"tun0",
|
||||
"source-address"
|
||||
],
|
||||
"value": "87.192.101.48"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"tunnel",
|
||||
"tun0",
|
||||
"remote"
|
||||
],
|
||||
"value": "216.66.88.98"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"tunnel",
|
||||
"tun0",
|
||||
"address"
|
||||
],
|
||||
"value": "2001:470:1f1c:f6::2/64"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"tunnel",
|
||||
"tun0",
|
||||
"mtu"
|
||||
],
|
||||
"value": "1480"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"tunnel",
|
||||
"tun0",
|
||||
"description"
|
||||
],
|
||||
"value": "HE 6in4 tunnel - 2001:470:187e::/48"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"protocols",
|
||||
"static",
|
||||
"route6",
|
||||
"::/0",
|
||||
"next-hop",
|
||||
"2001:470:1f1c:f6::1"
|
||||
],
|
||||
"value": {}
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"system",
|
||||
"task-scheduler",
|
||||
"task",
|
||||
"he-tunnel-follow",
|
||||
"executable",
|
||||
"path"
|
||||
],
|
||||
"value": "/config/he-tunnel-follow"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"system",
|
||||
"task-scheduler",
|
||||
"task",
|
||||
"he-tunnel-follow",
|
||||
"executable",
|
||||
"arguments"
|
||||
],
|
||||
"value": "run"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"system",
|
||||
"task-scheduler",
|
||||
"task",
|
||||
"he-tunnel-follow",
|
||||
"interval"
|
||||
],
|
||||
"value": "1m"
|
||||
}
|
||||
],
|
||||
"perRouter": {
|
||||
"vyos001": [
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"bonding",
|
||||
"bond0",
|
||||
"vif",
|
||||
"9",
|
||||
"address"
|
||||
],
|
||||
"value": "2001:470:187e:9::1/64"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"service",
|
||||
"router-advert",
|
||||
"interface",
|
||||
"bond0.9",
|
||||
"default-preference"
|
||||
],
|
||||
"value": "high"
|
||||
}
|
||||
],
|
||||
"vyos002": [
|
||||
{
|
||||
"path": [
|
||||
"interfaces",
|
||||
"bonding",
|
||||
"bond0",
|
||||
"vif",
|
||||
"9",
|
||||
"address"
|
||||
],
|
||||
"value": "2001:470:187e:9::2/64"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"service",
|
||||
"router-advert",
|
||||
"interface",
|
||||
"bond0.9",
|
||||
"default-preference"
|
||||
],
|
||||
"value": "low"
|
||||
}
|
||||
]
|
||||
},
|
||||
"sharedRouterAdvert": [
|
||||
{
|
||||
"path": [
|
||||
"service",
|
||||
"router-advert",
|
||||
"interface",
|
||||
"bond0.9",
|
||||
"link-mtu"
|
||||
],
|
||||
"value": "1472"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"service",
|
||||
"router-advert",
|
||||
"interface",
|
||||
"bond0.9",
|
||||
"prefix",
|
||||
"2001:470:187e:9::/64",
|
||||
"preferred-lifetime"
|
||||
],
|
||||
"value": "604800"
|
||||
},
|
||||
{
|
||||
"path": [
|
||||
"service",
|
||||
"router-advert",
|
||||
"interface",
|
||||
"bond0.9",
|
||||
"prefix",
|
||||
"2001:470:187e:9::/64",
|
||||
"valid-lifetime"
|
||||
],
|
||||
"value": "2592000"
|
||||
}
|
||||
],
|
||||
"_staging_note": "MERGED 2026-09-06 as kubernetes-deployment@6d4e080 on main. This file is now only a historical record; the live model is infra/vyos/subtrees/overrides.json, where the four IPv6 overrides (he-ipv6-tunnel, he-tunnel-follow-scheduler, firewall-accept-he-6in4, ipv6-vlan9-private + ipv6-vlan9-addr-vyos00[12]) are scoped to both routers. vyos:verify reports 533/534 nodes, zero drift. TWO THINGS THE MERGE ITSELF CAUGHT, worth carrying: (1) firewall-accept-he-6in4 was vyos001-only, so vyos002 had no proto-41 accept under its IPv4 input default-deny -- a failover would have left tun0 up and radvd running on the new master while HE's encapsulated traffic was dropped by its own firewall. The tunnel deployment alone was NOT sufficient. (2) The per-router bond0.9 override must list the IPv4 address next to the IPv6 one, because an override set declares the COMPLETE set for its path. Also: do not stage vyos model changes on a feature branch. The first attempt at this merge was made on fix/openbao-preview-blockers, which is 15 commits behind main and predates the PPPoE HA overrides -- committing it would have reverted pppoe-gated-not-config-disabled and vrrp-transition-scripts-wan-follows-master. main lives in the .worktrees/grafana-token worktree."
|
||||
}
|
||||
55
migration/pulumi-override-pppoe-gated.json
Normal file
55
migration/pulumi-override-pppoe-gated.json
Normal file
@@ -0,0 +1,55 @@
|
||||
{
|
||||
"_comment": [
|
||||
"STAGED, NOT APPLIED. Paste these two objects into",
|
||||
"kubernetes-deployment infra/vyos/subtrees/overrides.json -- but ONLY after",
|
||||
"migration/vrrp-wan-install has run on BOTH production routers.",
|
||||
"",
|
||||
"Ordering is not a nicety. Another agent runs `pulumi up` on that repo, so",
|
||||
"merging this file IS a production change, made by someone else, at a time",
|
||||
"you do not choose. Removing `interfaces pppoe pppoe0 disable` from vyos002",
|
||||
"before the gate exists there lets vyos002 dial the moment anything commits,",
|
||||
"and Vodafone is a single-session account: it would take the live session off",
|
||||
"vyos001 and drop the household's internet.",
|
||||
"",
|
||||
"Install the gate first. Verify `systemctl show ppp@pppoe0 -p ConditionResult`",
|
||||
"reads `no` on vyos002. Only then merge.",
|
||||
"",
|
||||
"Note there is deliberately NO override asserting `vif 53 disable`.",
|
||||
"applyTree is delete-then-set per subtree, so the model is authoritative and",
|
||||
"omission means deletion -- but the 10 gig resting state is RUNTIME state",
|
||||
"owned by vrrp-wan-reconcile, keyed on who holds the management VIP. Pinning",
|
||||
"it in the model would fight the reconciler on every apply. The rule is",
|
||||
"`npm run vyos:export && npm run vyos:render` immediately before any",
|
||||
"`pulumi up`, so the model follows whichever router actually holds the WAN."
|
||||
],
|
||||
"overrides": [
|
||||
{
|
||||
"id": "pppoe-gated-not-config-disabled",
|
||||
"routers": ["vyos002"],
|
||||
"reason": "PPPoE cannot live on the VyOS config plane. interfaces_pppoe.py treats `disable` and `delete` identically: both unlink /etc/ppp/peers/pppoe0, which is pppd's own options file (ExecStart=/usr/sbin/pppd call %I), call PPPoEIf.remove() to withdraw the FRR default route, and stop the unit. So the resting state destroyed exactly what the promotion path needed, and ppp@pppoe0 restart-looped against the missing file -- 47 restarts observed, zero sessions at the access concentrator -- without ever tripping systemd's limiter, because RestartSec=5s against the default 10s/5-burst window is only two restarts per interval. It also made op-mode `connect interface pppoe0` unusable and put every failover behind a priority-322 commit where one unrelated invalid node fails the whole thing. pppoe0 is now ENABLED on both routers so the peers file always exists, and dialling is gated by ppp@pppoe0.service.d/10-vrrp-wan-gate.conf on /run/vrrp-wan/may-dial, a lease renewed by vrrp-wan-reconcile and revoked by vrrp-wan-guard. /run is tmpfs, so the gate is shut at boot and neither box dials before VRRP has decided. This override encodes must-never-come-back: a re-run of migration/vyos-mode-delta.py or a stale import must not re-pin `disable`. PREREQUISITE: the gate drop-in must already be installed on vyos002, or removing `disable` lets it dial and steal the single ISP session. Proven in labsim across session-control replace/deny/disable; see lab migration/PPPOE-HA.md.",
|
||||
"remove": [["interfaces", "pppoe", "pppoe0", "disable"]]
|
||||
},
|
||||
{
|
||||
"id": "vrrp-transition-scripts-wan-follows-master",
|
||||
"reason": "Gives the WAN a fast path on top of the 30s reconcile timer. Both hooks exec the same reconciler -- one code path, asked at different moments -- so a lost or duplicated transition cannot desynchronise anything; the timer remains the correctness guarantee and the scripts are only latency. `stop` and `fault` matter as much as `backup`: a stopped keepalived is a demotion too, and without those a box would keep the WAN while holding no VIPs, which is the 2026-09-02 outage shape. They go on the SYNC GROUP because VyOS refuses a per-group script while the group is in a sync group. Do not rely on these alone: on 2026-09-02 keepalived-fifo.py logged NOTHING for a promotion while Keepalived_vrrp logged all six instances entering MASTER, which is precisely why the reconcile timer exists.",
|
||||
"set": [
|
||||
{
|
||||
"path": ["high-availability", "vrrp", "sync-group", "MAIN", "transition-script", "master"],
|
||||
"value": "/config/vrrp-wan-take"
|
||||
},
|
||||
{
|
||||
"path": ["high-availability", "vrrp", "sync-group", "MAIN", "transition-script", "backup"],
|
||||
"value": "/config/vrrp-wan-release"
|
||||
},
|
||||
{
|
||||
"path": ["high-availability", "vrrp", "sync-group", "MAIN", "transition-script", "fault"],
|
||||
"value": "/config/vrrp-wan-release"
|
||||
},
|
||||
{
|
||||
"path": ["high-availability", "vrrp", "sync-group", "MAIN", "transition-script", "stop"],
|
||||
"value": "/config/vrrp-wan-release"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
111
migration/vif53-pin-boot-disable
Normal file
111
migration/vif53-pin-boot-disable
Normal file
@@ -0,0 +1,111 @@
|
||||
#!/bin/sh
|
||||
# Make config.boot carry `vif 53 disable` while the RUNNING config keeps the
|
||||
# 10 gig up. Run on the master. Idempotent.
|
||||
#
|
||||
# WHY THIS IS AWKWARD
|
||||
#
|
||||
# The convention is that BOTH routers' config.boot hold `vif 53 disable`, so a
|
||||
# reboot in any order comes up unable to claim the cloned MAC
|
||||
# (f0:9f:c2:12:9b:4f) and vrrp-wan-reconcile then enables it on whichever box
|
||||
# holds the VIP. The master's RUNNING config must have it enabled -- it is
|
||||
# carrying the WAN -- so config.boot and the running config must deliberately
|
||||
# disagree.
|
||||
#
|
||||
# VyOS gives no clean way to express that. `save` writes the RUNNING config,
|
||||
# not the candidate: setting the node, saving, then discarding was tested in
|
||||
# labsim and config.boot came back without `disable`, the WAN untouched. So the
|
||||
# only route is to genuinely disable it, save that, and re-enable -- a real,
|
||||
# brief interruption of the 10 gig.
|
||||
#
|
||||
# WHAT THE INTERRUPTION ACTUALLY COSTS
|
||||
#
|
||||
# Not the internet, on a healthy pair: pppoe0 is up on the master and the
|
||||
# failover route falls to it while bond0.53 is down, which is exactly the
|
||||
# behaviour T5 proves in labsim (LAN back in 5s). Traffic moves to the slower
|
||||
# line and back. Expect a few seconds, plus however long the ISP takes to
|
||||
# re-issue the DHCP lease afterwards.
|
||||
#
|
||||
# IF THIS SCRIPT DIES HALFWAY it leaves the master with bond0.53 disabled --
|
||||
# which vrrp-wan-reconcile repairs within 30s ("MASTER with bond0.53 disabled
|
||||
# -> enabling"). The failure mode is bounded by design, not by luck.
|
||||
#
|
||||
# sudo /config/vif53-pin-boot-disable do it
|
||||
# sudo /config/vif53-pin-boot-disable --check report only, change nothing
|
||||
|
||||
VIF=53
|
||||
BOOT=/config/config.boot
|
||||
|
||||
cfg() { /opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands 2>/dev/null; }
|
||||
boot_has() { awk "/vif ${VIF} {/,/^ }/" "$BOOT" 2>/dev/null | grep -qc disable 2>/dev/null; }
|
||||
boot_disable_count() { awk "/vif ${VIF} {/,/^ }/" "$BOOT" 2>/dev/null | grep -c disable; }
|
||||
run_disable_count() { cfg | grep -c "vif ${VIF} disable"; }
|
||||
|
||||
report() {
|
||||
printf ' running config disable : %s\n' "$(run_disable_count)"
|
||||
printf ' config.boot disable : %s\n' "$(boot_disable_count)"
|
||||
printf ' bond0.%s address : %s\n' "$VIF" \
|
||||
"$(ip -4 addr show "bond0.${VIF}" 2>/dev/null | sed -n 's/.*inet \([0-9.]*\).*/\1/p')"
|
||||
}
|
||||
|
||||
if [ "${1:-}" = "--check" ]; then report; exit 0; fi
|
||||
|
||||
if [ "$(boot_disable_count)" -ge 1 ]; then
|
||||
echo " config.boot already pins 'vif ${VIF} disable' -- nothing to do"
|
||||
report
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Refuse on a box that is not carrying the WAN: there the running config should
|
||||
# already have `disable`, and a plain `save` is all that is needed. Doing the
|
||||
# dance here would be pointless downtime.
|
||||
if [ "$(run_disable_count)" -ge 1 ]; then
|
||||
echo " this box already has 'vif ${VIF} disable' in the running config;"
|
||||
echo " a plain 'save' is enough and costs nothing. Not touching the WAN."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Serialise against vrrp-wan-reconcile by taking ITS lock. Without this the two
|
||||
# commit at the same time and VyOS refuses one of them with "Configuration
|
||||
# system temporarily locked due to another commit in progress" -- observed in
|
||||
# labsim, where the `save` landed but the RE-ENABLE did not, leaving the master
|
||||
# with its 10 gig down. The reconciler uses `flock -n` and simply skips a tick
|
||||
# it cannot get, so holding this is cheap and safe.
|
||||
exec 9>/run/vrrp-wan.lock
|
||||
if ! flock -w 60 9; then
|
||||
echo " could not take /run/vrrp-wan.lock within 60s -- is a commit stuck?"
|
||||
echo " refusing to race vrrp-wan-reconcile for the config lock."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
logger -t vif53-pin "pinning 'vif ${VIF} disable' into config.boot -- bond0.${VIF} will bounce"
|
||||
echo " bouncing bond0.${VIF} to get 'disable' into config.boot..."
|
||||
t0=$(date +%s)
|
||||
|
||||
# 9>&- so the config session's long-lived unionfs-fuse child does not inherit
|
||||
# the lock fd and hold it for ever -- the same trap vrrp-wan-reconcile documents.
|
||||
/bin/vbash 9>&- <<VBASH
|
||||
source /opt/vyatta/etc/functions/script-template
|
||||
configure
|
||||
set interfaces bonding bond0 vif ${VIF} disable
|
||||
commit
|
||||
save
|
||||
delete interfaces bonding bond0 vif ${VIF} disable
|
||||
commit
|
||||
exit
|
||||
VBASH
|
||||
rc=$?
|
||||
|
||||
echo " window: $(( $(date +%s) - t0 ))s (exit $rc)"
|
||||
logger -t vif53-pin "done in $(( $(date +%s) - t0 ))s"
|
||||
report
|
||||
|
||||
# Belt and braces: if the re-enable did not take, say so loudly. The reconciler
|
||||
# will fix it within 30s on a master, but silence here would look like success.
|
||||
if [ "$(run_disable_count)" -ge 1 ]; then
|
||||
echo " *** WARNING: bond0.${VIF} is STILL disabled in the running config."
|
||||
echo " *** vrrp-wan-reconcile should re-enable it within 30s on the master."
|
||||
echo " *** Force it now with: sudo /config/vrrp-wan-reconcile"
|
||||
exit 1
|
||||
fi
|
||||
[ "$(boot_disable_count)" -ge 1 ] || { echo " *** config.boot still not pinned"; exit 1; }
|
||||
echo " OK: config.boot pinned, running config still has the WAN"
|
||||
189
migration/vlan2-v6-apply
Executable file
189
migration/vlan2-v6-apply
Executable file
@@ -0,0 +1,189 @@
|
||||
#!/bin/bash
|
||||
# Give VLAN 2 (the k8s VLAN) IPv6: addresses, router advertisements and DHCPv6
|
||||
# reservations. Production half of dual-stack phase 2b.
|
||||
#
|
||||
# Rehearsed first as labsim/labsim-dualstack-net.sh, which is where the four
|
||||
# VyOS facts below were paid for rather than guessed.
|
||||
#
|
||||
# ADDRESSING ONLY -- NOT EGRESS. The prefix is advertised with
|
||||
# `default-lifetime 0`, so nodes get their reserved addresses but neither router
|
||||
# becomes an IPv6 default router. Turning on real v6 egress moves cluster image
|
||||
# pulls onto the HE tunnel (1480, or 1472 on PPPoE) whose throughput has never
|
||||
# been measured, and that is not a thing to switch on unattended. Flipping it is
|
||||
# one line: `set service router-advert interface bond0.2 default-lifetime '1800'`
|
||||
# plus a default-preference, once somebody is watching.
|
||||
#
|
||||
# LISTEN-INTERFACE IS NOT OPTIONAL. Without it kea6 renders
|
||||
# `interfaces: [ "*" ]` and serves DHCPv6 on EVERY VLAN, not just this one.
|
||||
# Observed in production the moment this was first applied: kea started
|
||||
# answering SOLICIT/REQUEST from an unrelated device on bond0.10 (LoT). That is
|
||||
# a DHCPv6 server switched on estate-wide as a side effect of configuring one
|
||||
# VLAN -- the same family of mistake as the kea IPv4 cross-VLAN bug (ISC #1117)
|
||||
# this estate already fought. Pin the interface.
|
||||
#
|
||||
# WHY DHCPv6 RATHER THAN SLAAC: k3s resolves node-ip once at start-up, so a
|
||||
# node's address must be knowable in advance and stable. The estate already
|
||||
# answers that for IPv4 with kea reservations keyed on MAC; IPv6 answers it the
|
||||
# same way, from the same MACs, so there is one source of truth. VyOS's
|
||||
# static-mapping accepts `mac` as well as `duid`, which is what makes that
|
||||
# possible -- DHCPv6 normally keys on a client-generated DUID.
|
||||
#
|
||||
# vlan2-v6-apply plan print what would be applied, change nothing
|
||||
# vlan2-v6-apply apply apply to both routers
|
||||
# vlan2-v6-apply verify what the routers and nodes now hold
|
||||
# vlan2-v6-apply revert remove it again
|
||||
set -uo pipefail
|
||||
|
||||
R1="${R1:-10.0.1.252}" # vyos001 -> ::1
|
||||
R2="${R2:-10.0.1.253}" # vyos002 -> ::2
|
||||
PW="${VYOS_PW:-vyos}"
|
||||
V6_PREFIX="${V6_PREFIX:-2001:470:187e:2}"
|
||||
LINK_MTU="${LINK_MTU:-1472}"
|
||||
SUBNET_ID="${SUBNET_ID:-2}" # VyOS requires a unique id per DHCPv6 subnet
|
||||
SHARED_NET="${SHARED_NET:-TheLab-k8s}"
|
||||
|
||||
# name:v4-host-octet -- the IPv6 host part mirrors the IPv4 one so a reservation
|
||||
# is readable next to its twin. MACs are read from the LIVE IPv4 reservations at
|
||||
# run time, never duplicated here: one source of truth, and a node that is
|
||||
# re-homed cannot end up with a stale v6 mapping.
|
||||
NODES=(worker0-k8s0:23 worker1-k8s0:13 worker2-k8s0:25 spark-2935:12 aitopatom-3a1c:27)
|
||||
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=8)
|
||||
|
||||
# stderr, NOT stdout. build_mappings() is captured with $(...) and log() lines
|
||||
# went straight into the config stream, where VyOS rejected each one as
|
||||
# "Invalid command: [[0" -- ANSI escapes and all. The valid sets still applied,
|
||||
# so the routers ended up correct but NOT identical: different lines were lost
|
||||
# on each. A progress message is not data; keep it off the data channel.
|
||||
log() { printf '\033[0;36m[vlan2-v6]\033[0m %s\n' "$*" >&2; }
|
||||
die() { printf '\033[0;31m[vlan2-v6]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
r() { timeout 45 ssh "${SSH[@]}" "vyos@$1" "${@:2}" 2>/dev/null; }
|
||||
|
||||
# Drive VyOS from a script FILE with plain commit + save.
|
||||
# - `vbash -c` never starts a config session; the commit fails to stderr and a
|
||||
# helper discards it, so the run reports success having changed nothing.
|
||||
# - `commit-confirm` hangs non-interactively and strands an orphaned
|
||||
# config-mgmt commit_confirm holding the config lock.
|
||||
# - vbash exits 0 even when the commit fails, so the OUTPUT is the only honest
|
||||
# signal. Read it.
|
||||
vyos_apply() {
|
||||
local h="$1" out
|
||||
out="$({ printf '#!/bin/vbash\nsource /opt/vyatta/etc/functions/script-template\nconfigure\n'
|
||||
cat
|
||||
printf 'commit\nsave\nexit\n'
|
||||
} | timeout 150 ssh "${SSH[@]}" "vyos@$h" \
|
||||
'cat > /tmp/vlan2-v6.sh && chmod +x /tmp/vlan2-v6.sh && sudo /tmp/vlan2-v6.sh' 2>&1)"
|
||||
printf '%s\n' "$out" | grep -vE '^\s*$' | sed 's/^/ /' | tail -8
|
||||
# "Invalid command" was NOT in this list the first time, so a run that fed
|
||||
# rubbish to VyOS reported success. vbash exits 0 regardless, so every
|
||||
# rejection shape has to be named explicitly.
|
||||
printf '%s' "$out" | grep -qiE 'Commit failed|\[\[.*\]\] failed|Set failed|Invalid command' && return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
# Read each node's MAC out of the live IPv4 reservation.
|
||||
#
|
||||
# Scoped to the IPv4 subnet on purpose. Once this script has run once, the node
|
||||
# has TWO `static-mapping <name> mac` lines -- the v4 one and the v6 one it just
|
||||
# created -- and an unscoped match returned both concatenated
|
||||
# ("9c:76:0e:49:e9:179c:76:0e:49:e9:17"), which VyOS then rejected as an invalid
|
||||
# value. Self-inflicted on the second run: the lookup has to name the subnet it
|
||||
# means, or the script poisons its own input as soon as it succeeds.
|
||||
V4_SUBNET="${V4_SUBNET:-192.168.8.0/23}"
|
||||
mac_of() {
|
||||
r "$R1" "/opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands 2>/dev/null \
|
||||
| grep -F 'subnet $V4_SUBNET' \
|
||||
| sed -n \"s/.*static-mapping $1 mac '\\(.*\\)'/\\1/p\"" | tr -d ' \n'
|
||||
}
|
||||
|
||||
build_mappings() {
|
||||
local out="" name hextet mac
|
||||
for entry in "${NODES[@]}"; do
|
||||
name="${entry%%:*}"; hextet="${entry##*:}"
|
||||
mac="$(mac_of "$name")"
|
||||
[ -n "$mac" ] || die "no IPv4 reservation found for $name -- refusing to invent a MAC"
|
||||
log " $name $mac -> ${V6_PREFIX}::${hextet}"
|
||||
[ -n "$out" ] && out+=$'\n'
|
||||
out+="set service dhcpv6-server shared-network-name ${SHARED_NET} subnet ${V6_PREFIX}::/64 static-mapping ${name} mac '${mac}'
|
||||
set service dhcpv6-server shared-network-name ${SHARED_NET} subnet ${V6_PREFIX}::/64 static-mapping ${name} ipv6-address '${V6_PREFIX}::${hextet}'"
|
||||
done
|
||||
printf '%s' "$out"
|
||||
}
|
||||
|
||||
# One commit per router, and `no-autonomous-flag` is in it. That is not a
|
||||
# style choice: turning autonomous off LATER does not retract addresses already
|
||||
# formed, so a prefix advertised even briefly without it leaves every node
|
||||
# holding an unreserved EUI-64 address with a 30-day lifetime. Observed in the
|
||||
# sim. VLAN 2 has no IPv6 today, so this is the one chance to get it right.
|
||||
config_for() {
|
||||
local self="$1" mappings="$2"
|
||||
cat <<EOF
|
||||
set interfaces bonding bond0 vif 2 address '${V6_PREFIX}::${self}/64'
|
||||
set service router-advert interface bond0.2 prefix ${V6_PREFIX}::/64 no-autonomous-flag
|
||||
set service router-advert interface bond0.2 prefix ${V6_PREFIX}::/64 valid-lifetime '2592000'
|
||||
set service router-advert interface bond0.2 prefix ${V6_PREFIX}::/64 preferred-lifetime '604800'
|
||||
set service router-advert interface bond0.2 managed-flag
|
||||
set service router-advert interface bond0.2 link-mtu '${LINK_MTU}'
|
||||
set service router-advert interface bond0.2 default-lifetime '0'
|
||||
set service dhcpv6-server listen-interface bond0.2
|
||||
set service dhcpv6-server shared-network-name ${SHARED_NET} subnet ${V6_PREFIX}::/64 subnet-id '${SUBNET_ID}'
|
||||
set service dhcpv6-server shared-network-name ${SHARED_NET} subnet ${V6_PREFIX}::/64 interface 'bond0.2'
|
||||
${mappings}
|
||||
EOF
|
||||
}
|
||||
|
||||
cmd_plan() {
|
||||
log "reading MACs from the live IPv4 reservations"
|
||||
local m; m="$(build_mappings)" || exit 1
|
||||
echo; log "--- vyos001 (${V6_PREFIX}::1) ---"; config_for 1 "$m" | sed 's/^/ /'
|
||||
echo; log "--- vyos002 (${V6_PREFIX}::2) ---"; config_for 2 "$m" | sed 's/^/ /'
|
||||
}
|
||||
|
||||
cmd_apply() {
|
||||
log "reading MACs from the live IPv4 reservations"
|
||||
local m; m="$(build_mappings)" || exit 1
|
||||
local h self
|
||||
for h in "$R1" "$R2"; do
|
||||
[ "$h" = "$R1" ] && self=1 || self=2
|
||||
log "applying to $h (${V6_PREFIX}::${self})"
|
||||
config_for "$self" "$m" | vyos_apply "$h" \
|
||||
|| die "commit failed on $h -- NOTE: VyOS commits node groups independently, so re-read the config rather than assuming rollback"
|
||||
done
|
||||
log "applied. Nodes should take their reservations within a few minutes."
|
||||
}
|
||||
|
||||
cmd_verify() {
|
||||
local h
|
||||
for h in "$R1" "$R2"; do
|
||||
printf ' --- %s ---\n' "$h"
|
||||
printf ' bond0.2 v6 : %s\n' "$(r "$h" 'ip -6 -br addr show bond0.2 | tr -s " "')"
|
||||
printf ' radvd : %s (inactive on the backup is CORRECT)\n' "$(r "$h" 'systemctl is-active radvd')"
|
||||
printf ' kea-dhcp6 : %s\n' "$(r "$h" 'c=$(ps -ef | grep -c "[k]ea-dhcp6"); [ "$c" -gt 0 ] && echo running || echo "NOT running"')"
|
||||
printf ' role : %s\n' "$(r "$h" 'sudo /config/vrrp-wan-reconcile --status 2>/dev/null | grep -o "role=[a-z]*"')"
|
||||
printf ' IPv4 sane : %s\n' "$(r "$h" 'ip -4 route show default | head -1')"
|
||||
done
|
||||
}
|
||||
|
||||
cmd_revert() {
|
||||
local h self
|
||||
for h in "$R1" "$R2"; do
|
||||
[ "$h" = "$R1" ] && self=1 || self=2
|
||||
log "reverting $h"
|
||||
vyos_apply "$h" <<EOF
|
||||
delete service dhcpv6-server
|
||||
delete service router-advert interface bond0.2
|
||||
delete interfaces bonding bond0 vif 2 address '${V6_PREFIX}::${self}/64'
|
||||
EOF
|
||||
done
|
||||
log "reverted"
|
||||
}
|
||||
|
||||
case "${1:-plan}" in
|
||||
plan) cmd_plan ;;
|
||||
apply) cmd_apply ;;
|
||||
verify) cmd_verify ;;
|
||||
revert) cmd_revert ;;
|
||||
*) die "usage: $0 {plan|apply|verify|revert}" ;;
|
||||
esac
|
||||
133
migration/vlan2-v6-watchdog
Executable file
133
migration/vlan2-v6-watchdog
Executable file
@@ -0,0 +1,133 @@
|
||||
#!/bin/bash
|
||||
# Auto-revert the VLAN 2 IPv6 change if IPv4 stops working.
|
||||
#
|
||||
# The change it guards is additive and IPv6-only, so in theory it cannot affect
|
||||
# IPv4 at all. This exists because "in theory" is exactly what has been wrong
|
||||
# repeatedly, and because it is applied in an unattended window: the person who
|
||||
# would notice is away, and a router that has lost IPv4 takes the house and the
|
||||
# cluster with it.
|
||||
#
|
||||
# Modelled on wan-drill-watchdog: arm before the risky thing, disarm after, and
|
||||
# in between let the ROUTER decide for itself rather than depending on anything
|
||||
# off-box. A watchdog that needs the network it is protecting is not a watchdog.
|
||||
#
|
||||
# vlan2-v6-watchdog arm [seconds] start watching (default 2400 = 40 min)
|
||||
# vlan2-v6-watchdog disarm stop
|
||||
# vlan2-v6-watchdog status is it armed, and what has it seen
|
||||
#
|
||||
# Runs ON the router, out of /config, detached via setsid so it outlives the ssh
|
||||
# session that started it.
|
||||
set -uo pipefail
|
||||
|
||||
STATE=/run/vlan2-v6-watchdog
|
||||
PIDF=$STATE/pid
|
||||
LOGF=$STATE/log
|
||||
V6_PREFIX="${V6_PREFIX:-2001:470:187e:2}"
|
||||
|
||||
# How long IPv4 must be continuously bad before reverting. Long enough that a
|
||||
# commit's own brief disruption, or one lost probe, does not trigger it; short
|
||||
# enough to matter. The reconciler ticks at 30s, so 90s is three chances.
|
||||
GRACE="${GRACE:-90}"
|
||||
PROBE="${PROBE:-9.9.9.9}"
|
||||
|
||||
log() { mkdir -p "$STATE"; printf '%s %s\n' "$(date -Is)" "$*" >> "$LOGF"; }
|
||||
|
||||
# IPv4 health, asked of the box itself -- and it MUST be role-aware.
|
||||
#
|
||||
# The first version required a default route plus internet on both routers, and
|
||||
# would have reverted on vyos002 within 90s of being armed. That box is the
|
||||
# gated BACKUP: by design it holds no VIP, has bond0.53 disabled and pppoe0
|
||||
# down, so it has no default route and no internet, and that is the correct
|
||||
# resting state rather than a fault. Caught at 20s of the 90s grace.
|
||||
#
|
||||
# So: the master must be able to reach the internet. The backup only has to
|
||||
# still be on the network and able to see its peer -- which is what would
|
||||
# actually be at risk if a VLAN 2 change went wrong.
|
||||
VIP="${VIP:-192.168.1.1}"
|
||||
holds_vip() { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; }
|
||||
|
||||
v4_ok() {
|
||||
if holds_vip; then
|
||||
ip -4 route show default 2>/dev/null | grep -q . || return 1
|
||||
ping -c1 -W2 "$PROBE" >/dev/null 2>&1
|
||||
else
|
||||
# Backup: management address present and the VIP answers. If the VIP has
|
||||
# gone too then the pair has a bigger problem than this change, and
|
||||
# reverting an IPv6 addition would not help -- so this deliberately does
|
||||
# not fire on peer loss alone.
|
||||
ip -4 -o addr show bond0.1 2>/dev/null | grep -q 'inet ' || return 1
|
||||
ping -c1 -W2 "$VIP" >/dev/null 2>&1
|
||||
fi
|
||||
}
|
||||
|
||||
revert() {
|
||||
log "REVERTING: IPv4 has been down for ${GRACE}s"
|
||||
# A script file, not vbash -c: the latter never starts a config session and
|
||||
# the commit fails to stderr where nobody sees it.
|
||||
cat > /tmp/vlan2-v6-revert.sh <<'REOF'
|
||||
#!/bin/vbash
|
||||
source /opt/vyatta/etc/functions/script-template
|
||||
configure
|
||||
delete service dhcpv6-server
|
||||
delete service router-advert interface bond0.2
|
||||
commit
|
||||
save
|
||||
exit
|
||||
REOF
|
||||
chmod +x /tmp/vlan2-v6-revert.sh
|
||||
/tmp/vlan2-v6-revert.sh >> "$LOGF" 2>&1
|
||||
# The interface address is deleted separately: it is the one node whose
|
||||
# removal could plausibly disturb something else, so it goes last and its
|
||||
# failure does not block the rest.
|
||||
for n in 1 2; do
|
||||
ip -6 addr del "${V6_PREFIX}::${n}/64" dev bond0.2 2>/dev/null
|
||||
done
|
||||
log "revert done; IPv4 now $(v4_ok && echo OK || echo STILL BAD)"
|
||||
}
|
||||
|
||||
watch_loop() {
|
||||
local deadline=$(( $(date +%s) + $1 )) bad=0
|
||||
log "armed for $1s (grace ${GRACE}s, probe ${PROBE})"
|
||||
while [ "$(date +%s)" -lt "$deadline" ]; do
|
||||
if v4_ok; then
|
||||
[ "$bad" -ne 0 ] && log "IPv4 recovered after ${bad}s"
|
||||
bad=0
|
||||
else
|
||||
bad=$((bad + 10))
|
||||
log "IPv4 bad (${bad}s/${GRACE}s)"
|
||||
if [ "$bad" -ge "$GRACE" ]; then
|
||||
revert
|
||||
log "disarming after revert"
|
||||
rm -f "$PIDF"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
sleep 10
|
||||
done
|
||||
log "expired without incident"
|
||||
rm -f "$PIDF"
|
||||
}
|
||||
|
||||
case "${1:-status}" in
|
||||
arm)
|
||||
mkdir -p "$STATE"
|
||||
[ -f "$PIDF" ] && kill -0 "$(cat "$PIDF")" 2>/dev/null && { echo "already armed"; exit 0; }
|
||||
setsid "$0" _run "${2:-2400}" >/dev/null 2>&1 < /dev/null &
|
||||
sleep 1
|
||||
[ -f "$PIDF" ] && echo " armed (pid $(cat "$PIDF"), ${2:-2400}s)" || echo " FAILED to arm"
|
||||
;;
|
||||
_run)
|
||||
echo $$ > "$PIDF"
|
||||
watch_loop "${2:-2400}"
|
||||
;;
|
||||
disarm)
|
||||
if [ -f "$PIDF" ]; then kill "$(cat "$PIDF")" 2>/dev/null; rm -f "$PIDF"; echo " disarmed"
|
||||
else echo " not armed"; fi
|
||||
;;
|
||||
status)
|
||||
if [ -f "$PIDF" ] && kill -0 "$(cat "$PIDF")" 2>/dev/null; then echo " armed (pid $(cat "$PIDF"))"
|
||||
else echo " not armed"; fi
|
||||
echo " --- log ---"; tail -12 "$LOGF" 2>/dev/null | sed 's/^/ /' || echo " (none)"
|
||||
;;
|
||||
*) echo "usage: $0 {arm [seconds]|disarm|status}" >&2; exit 2 ;;
|
||||
esac
|
||||
52
migration/vrrp-wan-apply
Normal file
52
migration/vrrp-wan-apply
Normal file
@@ -0,0 +1,52 @@
|
||||
#!/bin/vbash
|
||||
# Enable or disable the DHCP WAN (bond0.53). The ONLY part of the failover that
|
||||
# touches VyOS configuration.
|
||||
#
|
||||
# PPPoE deliberately does NOT appear here any more, and should not be added back
|
||||
# for symmetry. `set interfaces pppoe pppoe0 disable` unlinks
|
||||
# /etc/ppp/peers/pppoe0 -- interfaces_pppoe.py treats `disable` and `delete`
|
||||
# identically -- and that path is pppd's options file, so the resting state
|
||||
# destroyed what the promotion path needed and the unit restart-looped (47 times,
|
||||
# zero sessions at the AC). It also made a pppoe node able to fail this commit
|
||||
# and take the 10 gig down with it: `interfaces pppoe` is priority 322 and one
|
||||
# invalid node fails the whole commit. PPPoE is now gated at the systemd unit
|
||||
# instead; see migration/ppp-vrrp-gate.conf and vrrp-wan-reconcile.
|
||||
#
|
||||
# bond0.53 stays here because its lease is bound to a cloned MAC and only VyOS
|
||||
# config can move a MAC between boxes.
|
||||
#
|
||||
# `source /opt/vyatta/etc/functions/script-template` must be the FIRST thing the
|
||||
# script does. Sourced after an if, an exec and a mkdir it terminated the script
|
||||
# inside the source, rc=0, no output -- the caller reported success having done
|
||||
# nothing. Only a single assignment may precede it (the template resets the
|
||||
# positional parameters), which is the shape /config/vyos-known-good uses.
|
||||
#
|
||||
# vrrp-wan-apply enable take the DHCP WAN
|
||||
# vrrp-wan-apply disable release it
|
||||
MODE="${1:-}"
|
||||
source /opt/vyatta/etc/functions/script-template
|
||||
|
||||
CONF=/config/vrrp-wan.conf
|
||||
[ -r "$CONF" ] && . "$CONF"
|
||||
WAN_VIF="${WAN_VIF:-53}"
|
||||
|
||||
cfg() { /opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands 2>/dev/null; }
|
||||
wan_disabled(){ cfg | grep -q "vif ${WAN_VIF} disable"; }
|
||||
|
||||
configure
|
||||
if [ "$MODE" = enable ]; then
|
||||
# Guarded: `delete` of an absent node aborts the whole batch with
|
||||
# "Nothing to delete", which once left the box detected-but-unfixed.
|
||||
wan_disabled && delete interfaces bonding bond0 vif ${WAN_VIF} disable
|
||||
else
|
||||
wan_disabled || set interfaces bonding bond0 vif ${WAN_VIF} disable
|
||||
fi
|
||||
# Report the commit's verdict. This script previously ended on `exit` (a
|
||||
# script-template function) and returned 0 even after "Commit failed", so the
|
||||
# reconciler logged a release that had not happened -- the worst kind of failure
|
||||
# for something whose job is to keep two routers from holding one WAN.
|
||||
if commit 2>&1 | tee /tmp/vrrp-wan-commit.log | grep -qi "commit failed"; then
|
||||
logger -t vrrp-wan "COMMIT FAILED applying '$MODE' -- see /tmp/vrrp-wan-commit.log"
|
||||
exit 1
|
||||
fi
|
||||
exit
|
||||
45
migration/vrrp-wan-guard
Executable file
45
migration/vrrp-wan-guard
Executable file
@@ -0,0 +1,45 @@
|
||||
#!/bin/sh
|
||||
# Revoke the PPPoE dial lease. Runs every 5s. Only ever takes the WAN AWAY.
|
||||
#
|
||||
# vrrp-wan-reconcile is the single writer and runs every 30s; this is the
|
||||
# watcher, and it exists because `may-dial` must be a LEASE, not a flag.
|
||||
# ConditionPathExists is evaluated at START only -- it can prevent a dial, it can
|
||||
# never revoke one. So if the reconciler stops running (timer masked, box wedged,
|
||||
# someone stops it during maintenance) and the box is then demoted, nothing would
|
||||
# ever hang up: it would keep the one ISP session while the new master tries to
|
||||
# take it.
|
||||
#
|
||||
# Two conditions revoke, both biased the safe way:
|
||||
# - this box does not hold the management VIP
|
||||
# - the lease has not been renewed within LEASE_TTL (the reconciler is dead)
|
||||
#
|
||||
# Deliberately tiny: no config mode, no flock, no commit. It cannot wedge the
|
||||
# router's configuration system, which is what earns it a 5s timer. Running at
|
||||
# 5s rather than the reconciler's 30s is also what shrinks the double-dial
|
||||
# window on a demotion from up to 30s down to about 5.
|
||||
|
||||
CONF=/config/vrrp-wan.conf
|
||||
[ -r "$CONF" ] && . "$CONF"
|
||||
VIP="${VRRP_WAN_VIP:-192.168.1.1}"
|
||||
LEASE_TTL="${LEASE_TTL:-75}"
|
||||
STATE=/run/vrrp-wan
|
||||
|
||||
revoke() {
|
||||
rm -f "$STATE/may-dial"
|
||||
systemctl is-active --quiet ppp@pppoe0 2>/dev/null || return 0
|
||||
logger -t vrrp-wan "GUARD: $1 -- hanging up pppoe0"
|
||||
systemctl stop ppp@pppoe0 2>/dev/null
|
||||
}
|
||||
|
||||
if ! ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; then
|
||||
revoke "does not hold ${VIP}"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Holds the VIP, so it is entitled to dial -- but only while something is
|
||||
# actively renewing the lease on its behalf.
|
||||
if [ -f "$STATE/may-dial" ]; then
|
||||
age=$(( $(date +%s) - $(stat -c %Y "$STATE/may-dial" 2>/dev/null || echo 0) ))
|
||||
[ "$age" -gt "$LEASE_TTL" ] && revoke "lease stale (${age}s > ${LEASE_TTL}s; is vrrp-wan-reconcile.timer running?)"
|
||||
fi
|
||||
exit 0
|
||||
10
migration/vrrp-wan-guard.service
Normal file
10
migration/vrrp-wan-guard.service
Normal file
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
# Revokes the PPPoE dial lease. Only ever takes the WAN away, never grants it,
|
||||
# which is what makes a 5s cadence safe: it touches no VyOS configuration and
|
||||
# cannot wedge the commit lock.
|
||||
Description=Revoke the PPPoE dial lease when this router is not master
|
||||
After=vyos-router.service
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/config/vrrp-wan-guard
|
||||
13
migration/vrrp-wan-guard.timer
Normal file
13
migration/vrrp-wan-guard.timer
Normal file
@@ -0,0 +1,13 @@
|
||||
[Unit]
|
||||
Description=Revoke the PPPoE dial lease every 5s
|
||||
|
||||
[Timer]
|
||||
# 5s, against the reconciler's 30s. A demotion must hang up fast -- the window
|
||||
# between "no longer master" and "stopped dialling" is the window in which two
|
||||
# routers can hold one ISP session.
|
||||
OnBootSec=10
|
||||
OnUnitActiveSec=5
|
||||
AccuracySec=1
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
148
migration/vrrp-wan-health
Executable file
148
migration/vrrp-wan-health
Executable file
@@ -0,0 +1,148 @@
|
||||
#!/bin/sh
|
||||
# VRRP health check: may THIS router hold the floating IPs?
|
||||
#
|
||||
# It may only if it can actually carry the WAN. Without this, VRRP decides
|
||||
# mastership purely on whether the peer is still advertising -- so a router with
|
||||
# no WAN at all happily takes the VIPs and blackholes the entire LAN's internet
|
||||
# while looking perfectly healthy. That is not hypothetical: it is the outage of
|
||||
# 2026-09-02, reproduced in labsim.
|
||||
#
|
||||
# ---------------------------------------------------------------------------
|
||||
# The first version of this script asked one question: "do I have an address on
|
||||
# a WAN interface". That is correct for a pair where both routers hold WAN all
|
||||
# the time. Ours cannot: the 10 gig lease is bound to a cloned MAC and the
|
||||
# PPPoE line to a single credential, so the WAN follows mastership (see
|
||||
# vrrp-wan-take). Against that design the old check DEADLOCKS --
|
||||
#
|
||||
# may I be master? -> only if I already have WAN
|
||||
# do I have WAN? -> only if I am master
|
||||
#
|
||||
# -- and the backup sits in FAULT for ever. vyos002 sat exactly there, which
|
||||
# meant the pair could not fail over at all: the safety check had quietly
|
||||
# removed the redundancy it was protecting.
|
||||
#
|
||||
# So the question is now asked in the right order: enforce "must have WAN" only
|
||||
# on the router that is actually HOLDING the VIPs, and give a new master time to
|
||||
# bring the WAN up before judging it.
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# exit 0 = eligible for MASTER, non-zero = release and let the peer have it.
|
||||
|
||||
CONF=/config/vrrp-wan.conf
|
||||
[ -r "$CONF" ] && . "$CONF"
|
||||
STATE=/run/vrrp-wan
|
||||
VIP="${VRRP_WAN_VIP:-192.168.1.1}"
|
||||
|
||||
# Seconds a new master may go without any WAN. Sourced from vrrp-wan.conf; the
|
||||
# fallback is deliberately NOT the old 90. accel-ppp's dead-peer budget is
|
||||
# lcp-echo-interval(30) x lcp-echo-failure(3) = 90s, so a hard failover into an
|
||||
# access concentrator that does not replace the stale session lands exactly on
|
||||
# the boundary: the new master fails its own check, sheds the VIPs, and the peer
|
||||
# -- in the same position -- does likewise. Both end in FAULT, which is worse
|
||||
# than the outage this check exists to prevent.
|
||||
#
|
||||
# The fallback tracks vrrp-wan.conf, where the reasoning and the measurements
|
||||
# live. Short version: labsim T4 timed a destroyed master's takeover at 26s
|
||||
# (replace), 148s (deny) and 21s (disable); `deny` is the sizing case and 180
|
||||
# left only 32s over it. Keep the two in step -- keepalived runs this script
|
||||
# with no environment, so if vrrp-wan.conf is ever missing THIS number is the
|
||||
# one that decides mastership.
|
||||
GRACE="${GRACE:-300}"
|
||||
|
||||
# A deliberate hand-over lever.
|
||||
#
|
||||
# There is no reliable way to MAKE this pair fail over on demand. VyOS offers
|
||||
# only `restart vrrp`, and neither that nor `systemctl restart keepalived` is
|
||||
# dependable: with advert_int 1 the peer declares the master dead after ~3.6s,
|
||||
# and a restart usually finishes inside that window. Measured in labsim -- the
|
||||
# same command moved mastership on one run and not on the next three. A
|
||||
# fail-back procedure you cannot trigger on purpose is not a procedure.
|
||||
#
|
||||
# Failing the health check IS the supported way to shed mastership: the sync
|
||||
# group goes FAULT, releases every VIP, and the peer takes over -- the same path
|
||||
# a genuine WAN loss takes, so the planned drill exercises the real mechanism
|
||||
# rather than a special case.
|
||||
#
|
||||
# touch /run/vrrp-wan/force-fault hand over within failure-count*interval
|
||||
# rm /run/vrrp-wan/force-fault become eligible again (no-preempt keeps
|
||||
# it BACKUP until the peer hands back)
|
||||
#
|
||||
# It lives in /run deliberately: a reboot clears it, so a forgotten drill cannot
|
||||
# leave a router permanently ineligible.
|
||||
[ -f /run/vrrp-wan/force-fault ] && exit 1
|
||||
|
||||
# Am I holding the VIPs? Asked of REALITY -- is the management VIP actually on
|
||||
# this box -- and not of a /run marker.
|
||||
#
|
||||
# The marker was the first design and it is unsafe: it is written by the VRRP
|
||||
# transition script, and in labsim that script silently failed to run on a
|
||||
# promotion (VyOS's keepalived-fifo.py helper stopped delivering while
|
||||
# keepalived's own notifies kept working). The router then believed it was
|
||||
# backup, passed this check, and sat holding every VIP with no WAN -- the exact
|
||||
# outage this script exists to prevent, re-created by trusting the reporter
|
||||
# instead of the fact.
|
||||
if [ -z "$(ip -4 -o addr show 2>/dev/null | grep " ${VIP}/")" ]; then
|
||||
# Clear the grace stamp on the way down, HERE, not only in the reconciler.
|
||||
# The reconciler runs every 30s; this runs every 5s. A promotion that
|
||||
# inherited a stamp from an earlier mastership scored grace = hours, failed
|
||||
# immediately, and took the sync group to FAULT ~5s after passing -- with the
|
||||
# peer already faulted, that left BOTH routers in FAULT and the LAN with no
|
||||
# gateway. The stamp must belong to the CURRENT mastership or it is worse
|
||||
# than useless.
|
||||
rm -f "$STATE/since" 2>/dev/null
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Start the grace clock HERE, the moment mastership is first observed.
|
||||
#
|
||||
# It used to be stamped only by vrrp-wan-reconcile, which runs on a 30s timer --
|
||||
# so a freshly promoted master reached this check with no stamp, scored grace=0,
|
||||
# failed, and went FAULT before it had any chance to bring the WAN up. The peer
|
||||
# then found itself alone with no WAN either and did the same. Observed in
|
||||
# labsim: BOTH routers in FAULT, nobody holding the VIPs, the LAN with no
|
||||
# gateway at all. That is worse than the outage this script exists to prevent,
|
||||
# and it would have hit a REAL failover, not just a drill -- the health check
|
||||
# runs every 5s and the reconciler had not yet ticked.
|
||||
mkdir -p "$STATE" 2>/dev/null
|
||||
[ -f "$STATE/since" ] || date +%s > "$STATE/since"
|
||||
|
||||
# Master with an address on a WAN interface: healthy.
|
||||
#
|
||||
# Deliberately NOT "can I reach the internet" and NOT "do I have a default
|
||||
# route". During a real ISP outage the route disappears on BOTH routers; a check
|
||||
# keyed on that would put both into FAULT, nobody would hold the VIPs, and the
|
||||
# LAN would lose inter-VLAN routing too -- turning an internet outage into a
|
||||
# total one. A DHCP lease survives an ISP outage, so an address still
|
||||
# distinguishes "this box structurally cannot route" from "the internet is down
|
||||
# right now", which is the distinction that matters.
|
||||
# Any WAN counts. Requiring the 10 gig specifically would fault a healthy master
|
||||
# during a genuine 10 gig outage and turn a degraded state into a total one --
|
||||
# the same reasoning as the default-route note above. Which one satisfied it is
|
||||
# recorded for the operator and the test harness, but does not affect the verdict.
|
||||
for ifc in bond0.53 pppoe0; do
|
||||
if ip -4 addr show dev "$ifc" 2>/dev/null | grep -q 'inet '; then
|
||||
echo "$ifc" > "$STATE/wan" 2>/dev/null
|
||||
# Re-stamp on every healthy tick, so the grace window below measures
|
||||
# time since this box last DEMONSTRABLY had a WAN.
|
||||
date +%s > "$STATE/since" 2>/dev/null
|
||||
exit 0
|
||||
fi
|
||||
done
|
||||
rm -f "$STATE/wan" 2>/dev/null
|
||||
|
||||
# No WAN right now, but there was one within GRACE: ride it out.
|
||||
#
|
||||
# `since` is re-stamped on every healthy tick, so this measures time since the
|
||||
# box last HAD a WAN -- not time since it was promoted. Measuring from promotion
|
||||
# was wrong in a way that only shows up on an established master: after hours of
|
||||
# uptime `now - since` far exceeds any grace, so the first moment bond0.53 went
|
||||
# down and pppoe0 was mid-redial, the master failed its own check, shed every
|
||||
# VIP, and the peer -- inheriting the same WAN outage -- did the same. Observed
|
||||
# in labsim: taking the 10 gig down flapped the pair instead of falling back to
|
||||
# PPPoE. A WAN gap must be survivable wherever it happens, not only just after a
|
||||
# promotion.
|
||||
since=$(cat "$STATE/since" 2>/dev/null || echo 0)
|
||||
[ $(( $(date +%s) - since )) -lt "$GRACE" ] && exit 0
|
||||
|
||||
# Master, past grace, still no WAN: release. This is the 2026-09-02 case.
|
||||
exit 1
|
||||
115
migration/vrrp-wan-install
Executable file
115
migration/vrrp-wan-install
Executable file
@@ -0,0 +1,115 @@
|
||||
#!/bin/bash
|
||||
# Install (or verify) the WAN-follows-VRRP mechanism on a VyOS router.
|
||||
#
|
||||
# This exists because on 2026-09-05 the sim's failover proof was obtained from
|
||||
# scripts that had been hand-`sed`-ed in place: /config/vrrp-wan-health and
|
||||
# -reconcile differed from git by an edited VIP, so the tested behaviour was not
|
||||
# the committed behaviour and any reinstall would have silently reverted it.
|
||||
# `--check` makes that class of drift a hard failure instead of a discovery.
|
||||
#
|
||||
# It installs ONLY the mechanism -- scripts, units, drop-in, settings. It never
|
||||
# touches VyOS configuration: the `interfaces pppoe` node, `vif 53 disable` and
|
||||
# the VRRP sync-group hooks are config and belong in the config model
|
||||
# (labsim/sim-*.py for the sim, infra/vyos/subtrees/overrides.json for
|
||||
# production), not in an installer.
|
||||
#
|
||||
# vrrp-wan-install --vip 192.168.1.1 [--host vyos@10.0.1.253]
|
||||
# vrrp-wan-install --check [--host ...] # exits non-zero on any drift
|
||||
#
|
||||
# With no --host it operates on the local machine, so it can be scp'd to a
|
||||
# router and run there.
|
||||
set -uo pipefail
|
||||
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
VIP=""; HOST=""; MODE=install; PW="${VYOS_PW:-vyos}"
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--vip) VIP="$2"; shift 2 ;;
|
||||
--host) HOST="$2"; shift 2 ;;
|
||||
--check) MODE=check; shift ;;
|
||||
*) echo "usage: $0 [--vip A.B.C.D] [--host user@ip] [--check]" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
SSH_OPTS=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=8 -o PreferredAuthentications=password)
|
||||
run() { # run a command on the target
|
||||
if [ -n "$HOST" ]; then timeout 60 sshpass -p "$PW" ssh "${SSH_OPTS[@]}" "$HOST" "$@"
|
||||
else bash -c "$*"; fi
|
||||
}
|
||||
put() { # copy a file to the target
|
||||
if [ -n "$HOST" ]; then timeout 60 sshpass -p "$PW" scp "${SSH_OPTS[@]}" "$1" "$HOST:$2" >/dev/null
|
||||
else cp "$1" "$2"; fi
|
||||
}
|
||||
|
||||
# script -> destination. take/release are hooks keepalived calls; both exec the
|
||||
# reconciler, so there is one code path.
|
||||
#
|
||||
# he-tunnel-follow is part of the mechanism, not a separate thing: the reconciler
|
||||
# owns tun0's link state and that script owns its source address and MTU. Listing
|
||||
# it here is what makes `--check` catch drift on it and what makes the VyOS
|
||||
# image-upgrade runbook reinstall it -- IPv6 was previously the one half of the
|
||||
# WAN story that no installer knew about.
|
||||
SCRIPTS="vrrp-wan-reconcile vrrp-wan-apply vrrp-wan-health vrrp-wan-guard vrrp-wan-take vrrp-wan-release he-tunnel-follow"
|
||||
UNITS="vrrp-wan-reconcile.service vrrp-wan-reconcile.timer vrrp-wan-guard.service vrrp-wan-guard.timer"
|
||||
GATE_DIR=/etc/systemd/system/ppp@pppoe0.service.d
|
||||
GATE=$GATE_DIR/10-vrrp-wan-gate.conf
|
||||
|
||||
if [ "$MODE" = check ]; then
|
||||
rc=0
|
||||
for f in $SCRIPTS; do
|
||||
local_sum=$(md5sum "$HERE/$f" | cut -d' ' -f1)
|
||||
remote_sum=$(run "md5sum /config/$f 2>/dev/null | cut -d' ' -f1")
|
||||
[ "$local_sum" = "$remote_sum" ] || { echo " DRIFT /config/$f"; rc=1; }
|
||||
done
|
||||
for f in $UNITS; do
|
||||
local_sum=$(md5sum "$HERE/$f" | cut -d' ' -f1)
|
||||
remote_sum=$(run "md5sum /etc/systemd/system/$f 2>/dev/null | cut -d' ' -f1")
|
||||
[ "$local_sum" = "$remote_sum" ] || { echo " DRIFT /etc/systemd/system/$f"; rc=1; }
|
||||
done
|
||||
gate_sum=$(md5sum "$HERE/ppp-vrrp-gate.conf" | cut -d' ' -f1)
|
||||
remote_gate=$(run "md5sum $GATE 2>/dev/null | cut -d' ' -f1")
|
||||
[ "$gate_sum" = "$remote_gate" ] || { echo " DRIFT $GATE (a VyOS upgrade wipes /etc -- both routers would dial)"; rc=1; }
|
||||
run "[ -r /config/vrrp-wan.conf ]" || { echo " MISSING /config/vrrp-wan.conf"; rc=1; }
|
||||
# Secrets are placed by Pulumi (infra/vyos/secretsFile.ts), never by this
|
||||
# installer, so check presence only -- there is no correct content to compare
|
||||
# against and printing a diff of credentials would be worse than useless.
|
||||
# Without it he-tunnel-follow cannot re-point the tunnel when the WAN falls
|
||||
# back to PPPoE, which fails silently: IPv4 keeps working and IPv6 goes dark.
|
||||
run "[ -r /config/he-secrets ]" || { echo " MISSING /config/he-secrets (IPv6 cannot follow a WAN change)"; rc=1; }
|
||||
for t in vrrp-wan-reconcile.timer vrrp-wan-guard.timer; do
|
||||
[ "$(run "systemctl is-enabled $t 2>/dev/null")" = enabled ] || { echo " NOT ENABLED $t"; rc=1; }
|
||||
done
|
||||
[ "$rc" -eq 0 ] && echo " vrrp-wan in sync"
|
||||
exit "$rc"
|
||||
fi
|
||||
|
||||
[ -n "$VIP" ] || { echo "--vip is required to install" >&2; exit 2; }
|
||||
|
||||
for f in $SCRIPTS; do
|
||||
put "$HERE/$f" "/tmp/$f"
|
||||
# root:vyattacfg 0775 -- vrrp-wan-apply enters config mode, which requires
|
||||
# membership of vyattacfg.
|
||||
run "sudo install -o root -g vyattacfg -m 0775 /tmp/$f /config/$f"
|
||||
done
|
||||
|
||||
# Settings, with the VIP substituted. One file, read by BOTH the reconciler and
|
||||
# the health check -- keepalived invokes the latter with no environment at all,
|
||||
# so an Environment= line in the unit would be read by one and not the other.
|
||||
sed "s|^VRRP_WAN_VIP=.*|VRRP_WAN_VIP=${VIP}|" "$HERE/vrrp-wan.conf" > /tmp/vrrp-wan.conf.gen
|
||||
put /tmp/vrrp-wan.conf.gen /tmp/vrrp-wan.conf.gen
|
||||
run "sudo install -o root -g vyattacfg -m 0664 /tmp/vrrp-wan.conf.gen /config/vrrp-wan.conf"
|
||||
|
||||
for f in $UNITS; do
|
||||
put "$HERE/$f" "/tmp/$f"
|
||||
run "sudo install -m 0644 /tmp/$f /etc/systemd/system/$f"
|
||||
done
|
||||
|
||||
put "$HERE/ppp-vrrp-gate.conf" /tmp/ppp-vrrp-gate.conf
|
||||
run "sudo mkdir -p $GATE_DIR && sudo install -m 0644 /tmp/ppp-vrrp-gate.conf $GATE"
|
||||
|
||||
run "sudo systemctl daemon-reload && sudo systemctl enable --now vrrp-wan-reconcile.timer vrrp-wan-guard.timer" >/dev/null 2>&1
|
||||
|
||||
echo " installed (vip=$VIP)"
|
||||
run "sudo /config/vrrp-wan-reconcile --status"
|
||||
278
migration/vrrp-wan-reconcile
Normal file
278
migration/vrrp-wan-reconcile
Normal file
@@ -0,0 +1,278 @@
|
||||
#!/bin/sh
|
||||
# Make the WAN match VRRP mastership. Idempotent; safe to run every 30s and on
|
||||
# every VRRP transition.
|
||||
#
|
||||
# Two WANs, two different control planes, for a reason:
|
||||
#
|
||||
# bond0.53 (10 gig, DHCP) -- CONFIG plane. Its lease is bound to a cloned MAC
|
||||
# (f0:9f:c2:12:9b:4f, the old USG's), and only VyOS
|
||||
# config can move a MAC. One commit per failover.
|
||||
# pppoe0 (Vodafone) -- SYSTEMD plane. Gated by a drop-in on
|
||||
# ppp@pppoe0; see migration/ppp-vrrp-gate.conf.
|
||||
# No commit, no config lock, no `save`.
|
||||
#
|
||||
# PPPoE used to be on the config plane too, via `set interfaces pppoe pppoe0
|
||||
# disable`. That could not work: `disable` unlinks /etc/ppp/peers/pppoe0, which
|
||||
# is pppd's options file, so the promotion path deleted the very thing it needed
|
||||
# and left the unit restart-looping (observed: 47 restarts, zero sessions at the
|
||||
# access concentrator).
|
||||
#
|
||||
# Why a reconciler and not just transition scripts: VyOS delivers
|
||||
# `transition-script` through keepalived-fifo.py, and on 2026-09-02 that helper
|
||||
# logged NOTHING for a promotion while Keepalived_vrrp logged all six instances
|
||||
# entering MASTER. A router held every VIP with no WAN -- the outage, recreated
|
||||
# by the mechanism meant to prevent it. Scripts give speed; the timer gives
|
||||
# correctness.
|
||||
#
|
||||
# vrrp-wan-reconcile reconcile once
|
||||
# vrrp-wan-reconcile --status what it thinks, changing nothing
|
||||
|
||||
CONF=/config/vrrp-wan.conf
|
||||
[ -r "$CONF" ] && . "$CONF"
|
||||
VIP="${VRRP_WAN_VIP:-192.168.1.1}"
|
||||
WAN_VIF="${WAN_VIF:-53}"
|
||||
FLAP_MAX="${FLAP_MAX:-6}"
|
||||
FLAP_WINDOW="${FLAP_WINDOW:-600}"
|
||||
FLAP_HOLDOFF="${FLAP_HOLDOFF:-900}"
|
||||
|
||||
STATE=/run/vrrp-wan
|
||||
LOCK=/run/vrrp-wan.lock
|
||||
APPLY=/config/vrrp-wan-apply
|
||||
DROPIN=/etc/systemd/system/ppp@pppoe0.service.d/10-vrrp-wan-gate.conf
|
||||
V6_TUNNEL="${V6_TUNNEL:-tun0}"
|
||||
RADVD_CONF="${RADVD_CONF:-/run/radvd/radvd.conf}"
|
||||
|
||||
cfg() { /opt/vyatta/bin/vyatta-op-cmd-wrapper show configuration commands 2>/dev/null; }
|
||||
holds_vip() { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; }
|
||||
wan_up() { ip -4 addr show "bond0.${WAN_VIF}" 2>/dev/null | grep -q 'inet '; }
|
||||
ppp_up() { ip -4 addr show pppoe0 2>/dev/null | grep -q 'inet '; }
|
||||
ppp_active() { systemctl is-active --quiet ppp@pppoe0 2>/dev/null; }
|
||||
lease_age() { s=$(stat -c %Y "$STATE/may-dial" 2>/dev/null) || return 1
|
||||
echo $(( $(date +%s) - s )); }
|
||||
|
||||
# NOTE: there is deliberately no ppp_disabled(). pppoe0 is now ENABLED in config
|
||||
# on both routers, so such a test would be permanently false and the backup
|
||||
# early-exit below would never fire -- entering config mode every 30s for ever,
|
||||
# committing nothing. That exact shape was already live on the sim secondary,
|
||||
# whose /tmp/vrrp-wan-commit.log read "No configuration changes to commit" while
|
||||
# the script reported success.
|
||||
wan_disabled(){ cfg | grep -q "vif ${WAN_VIF} disable"; }
|
||||
|
||||
if [ "${1:-}" = "--status" ]; then
|
||||
printf 'vip=%s holds_vip=%s wan_disabled=%s wan_up=%s ppp_up=%s ppp_active=%s may_dial=%s lease_age=%s dropin=%s role=%s tun=%s radvd=%s\n' \
|
||||
"$VIP" "$(holds_vip && echo yes || echo no)" \
|
||||
"$(wan_disabled && echo yes || echo no)" \
|
||||
"$(wan_up && echo yes || echo no)" \
|
||||
"$(ppp_up && echo yes || echo no)" \
|
||||
"$(ppp_active && echo yes || echo no)" \
|
||||
"$([ -f "$STATE/may-dial" ] && echo yes || echo no)" \
|
||||
"$(lease_age 2>/dev/null || echo -)" \
|
||||
"$([ -f "$DROPIN" ] && echo yes || echo MISSING)" \
|
||||
"$(cat "$STATE/role" 2>/dev/null || echo unset)" \
|
||||
"$(ip -br link show "$V6_TUNNEL" 2>/dev/null | awk '{print $2}' || echo absent)" \
|
||||
"$(systemctl is-active radvd 2>/dev/null || echo inactive)"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# One writer. The lock fd MUST be closed for children (`9>&-` on every call):
|
||||
# entering VyOS config mode spawns a long-lived unionfs-fuse for the session
|
||||
# which INHERITS the descriptor and never releases it, so from the first commit
|
||||
# onward every later run lost the flock and exited 0 having done nothing. That
|
||||
# is how a demoted router kept the WAN.
|
||||
exec 9>"$LOCK"
|
||||
flock -n 9 || exit 0
|
||||
|
||||
mkdir -p "$STATE"
|
||||
|
||||
# Reap config sessions whose owning process is gone. VyOS leaves a unionfs mount
|
||||
# per `configure`, one of them holds the commit lock, and after that EVERY commit
|
||||
# fails -- including the manual one you try in order to fix it.
|
||||
for d in /opt/vyatta/config/tmp/new_config_*; do
|
||||
[ -d "$d" ] || continue
|
||||
pid=${d##*_}
|
||||
kill -0 "$pid" 2>/dev/null && continue
|
||||
umount -l "$d" 2>/dev/null
|
||||
rm -rf "$d" 2>/dev/null
|
||||
done
|
||||
|
||||
# `show configuration commands` has been observed returning EMPTY transiently
|
||||
# under commit-lock contention. Every grep against it then reads false, which on
|
||||
# the master path looks like "the WAN is disabled" and triggers a pointless
|
||||
# commit -- one such spurious "releasing" was logged on a box where both WANs
|
||||
# were already in the right state. A real config is ~500 lines; refuse to act on
|
||||
# a suspiciously short one.
|
||||
if [ "$(cfg | wc -l)" -lt 50 ]; then
|
||||
logger -t vrrp-wan "config read returned <50 lines; skipping this tick"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# --- PPPoE: the systemd plane ---------------------------------------------
|
||||
ppp_dial() {
|
||||
# An ESTABLISHED session outranks every guard below, and this must be the
|
||||
# first thing here. `may-dial` is a lease the guard expires after
|
||||
# LEASE_TTL, so any early `return 1` before this renew silently hands the
|
||||
# guard a live session to kill.
|
||||
#
|
||||
# Observed in labsim: the flap damper tripped, returned early, the lease
|
||||
# went stale at 81s > 75s and the guard hung up pppoe0 ON THE MASTER --
|
||||
# a damper meant to suppress repeat DIALS tore down a working WAN instead.
|
||||
# Everything below only decides whether to start a NEW session.
|
||||
if ppp_active; then
|
||||
touch "$STATE/may-dial"
|
||||
return 0
|
||||
fi
|
||||
# Refuse to bless a box whose gate is missing. /etc is per-image, so a VyOS
|
||||
# upgrade silently drops the drop-in -- and without it BOTH routers dial on
|
||||
# the next commit that touches the pppoe subtree. Failing closed turns a
|
||||
# silent loss of protection into "PPPoE never comes up", the safe direction.
|
||||
if [ ! -f "$DROPIN" ]; then
|
||||
logger -t vrrp-wan "REFUSING to dial: gate drop-in $DROPIN is missing (VyOS upgrade?)"
|
||||
return 1
|
||||
fi
|
||||
# The peers file is pppd's options file AND the gate's second condition, so
|
||||
# without it this box silently never dials: systemd logs "skipped because of
|
||||
# an unmet condition check" once and nothing else complains. Only a commit
|
||||
# that touches the pppoe subtree re-renders it.
|
||||
#
|
||||
# Seen in labsim: config applied but never `save`d, the router rebooted, and
|
||||
# came back with no pppoe0 node at all -- so no peers file, and a master that
|
||||
# dialled every tick into silence. Say so loudly rather than looking healthy.
|
||||
if [ ! -f /etc/ppp/peers/pppoe0 ]; then
|
||||
if cfg | grep -q "interfaces pppoe pppoe0 source-interface"; then
|
||||
logger -t vrrp-wan "CANNOT dial: pppoe0 is configured but /etc/ppp/peers/pppoe0 is missing -- re-commit the pppoe subtree to re-render it"
|
||||
else
|
||||
logger -t vrrp-wan "CANNOT dial: no pppoe0 in config (did a reboot revert an unsaved commit?)"
|
||||
fi
|
||||
return 1
|
||||
fi
|
||||
now=$(date +%s)
|
||||
if [ -f "$STATE/holdoff" ] && [ "$now" -lt "$(cat "$STATE/holdoff" 2>/dev/null || echo 0)" ]; then
|
||||
return 1
|
||||
fi
|
||||
touch "$STATE/may-dial" # renew the lease every tick
|
||||
# Trim the dial log to the window, then decide.
|
||||
if [ -f "$STATE/dials" ]; then
|
||||
awk -v c="$((now - FLAP_WINDOW))" '$1 > c' "$STATE/dials" > "$STATE/dials.new" 2>/dev/null
|
||||
mv "$STATE/dials.new" "$STATE/dials" 2>/dev/null
|
||||
fi
|
||||
# `cat | wc`, not `wc -l < file`: the shell applies redirections left to
|
||||
# right, so a missing file fails the `<` BEFORE `2>/dev/null` is in effect
|
||||
# and dash prints "No such file or directory" on every first-ever dial.
|
||||
if [ "$(cat "$STATE/dials" 2>/dev/null | wc -l)" -ge "$FLAP_MAX" ]; then
|
||||
echo $((now + FLAP_HOLDOFF)) > "$STATE/holdoff"
|
||||
logger -t vrrp-wan "DIAL FLAP: >=${FLAP_MAX} attempts in ${FLAP_WINDOW}s -- holding off ${FLAP_HOLDOFF}s"
|
||||
return 1
|
||||
fi
|
||||
echo "$now" >> "$STATE/dials"
|
||||
logger -t vrrp-wan "MASTER: dialling pppoe0"
|
||||
systemctl reset-failed ppp@pppoe0 2>/dev/null
|
||||
# `systemctl start` exits 0 even when a Condition blocks the start, so its
|
||||
# return code proves nothing. is-active is the only honest answer.
|
||||
systemctl start ppp@pppoe0 2>/dev/null
|
||||
}
|
||||
|
||||
ppp_release() {
|
||||
# Order matters: revoke the lease FIRST, then stop. The file's absence blocks
|
||||
# any NEW start (including one a concurrent VyOS commit would trigger); the
|
||||
# stop kills the process that already exists. Stopping first leaves a window
|
||||
# in which a commit re-dials a box that is being demoted.
|
||||
rm -f "$STATE/may-dial"
|
||||
ppp_active || return 0
|
||||
logger -t vrrp-wan "not MASTER: hanging up pppoe0"
|
||||
systemctl stop ppp@pppoe0 2>/dev/null
|
||||
}
|
||||
|
||||
# --- IPv6: the kernel plane -------------------------------------------------
|
||||
# The HE 6in4 tunnel and the VLAN 9 router advertisements have to follow
|
||||
# mastership too, or a failover keeps IPv4 and silently drops IPv6 -- the
|
||||
# partial outage that presents as "some sites are broken".
|
||||
#
|
||||
# A THIRD plane, and deliberately not either of the other two. Not config,
|
||||
# because nothing here needs a commit (unlike the cloned MAC) and a commit per
|
||||
# transition is the cost the pppoe0 half exists to avoid. Not the systemd gate,
|
||||
# because there is no equivalent of a peers file to destroy.
|
||||
#
|
||||
# What makes this cheap: the tunnel is anchored to 87.192.101.48, the 10 gig
|
||||
# lease bound to the cloned MAC, so it follows the VIP to the other router
|
||||
# UNCHANGED. A router-level failover therefore needs no HE API call at all --
|
||||
# only the link brought up on the box that now owns the address.
|
||||
#
|
||||
# Note what is deliberately NOT done here: this does not invoke
|
||||
# he-tunnel-follow. That script has its own 1-minute task-scheduler cadence and
|
||||
# a 2-tick hysteresis, and on 2026-09-06 that hysteresis was the only thing that
|
||||
# stopped a routine `vif53-pin-boot-disable` run from pointing HE at a PPPoE
|
||||
# address -- by 22 seconds. Calling it from a 30s reconciler as well would halve
|
||||
# the window it needs. Its job is the WITHIN-box fall back to PPPoE; ours is
|
||||
# link state.
|
||||
v6_take() {
|
||||
# Absent on a router that has no tunnel in its config -- which is every
|
||||
# router until the model change lands. No-op there rather than complain.
|
||||
[ -e "/sys/class/net/$V6_TUNNEL" ] || return 0
|
||||
# Needs SOME WAN address to source from. Either line will do: if bond0.53 is
|
||||
# down but pppoe0 is up, he-tunnel-follow re-points the tunnel on its own
|
||||
# schedule, and holding the link down until then would turn a degraded path
|
||||
# into no path.
|
||||
wan_up || ppp_up || return 0
|
||||
ip link show "$V6_TUNNEL" 2>/dev/null | grep -q 'state DOWN' && {
|
||||
logger -t vrrp-wan "MASTER: bringing $V6_TUNNEL up"
|
||||
ip link set "$V6_TUNNEL" up 2>/dev/null
|
||||
}
|
||||
# radvd's config is rendered into /run by the VyOS commit, so on a box with
|
||||
# no router-advert node there is nothing to start.
|
||||
[ -f "$RADVD_CONF" ] || return 0
|
||||
systemctl is-active --quiet radvd 2>/dev/null && return 0
|
||||
logger -t vrrp-wan "MASTER: starting radvd"
|
||||
systemctl start radvd 2>/dev/null
|
||||
}
|
||||
|
||||
v6_release() {
|
||||
[ -e "/sys/class/net/$V6_TUNNEL" ] || return 0
|
||||
# radvd FIRST, and this ordering is the point: on a graceful stop it emits a
|
||||
# final advertisement with router-lifetime 0, which is what tells VLAN 9
|
||||
# hosts to stop using this box as their default router. Kill the daemon
|
||||
# after tearing things down and they keep a dead gateway until the RA
|
||||
# lifetime expires on its own.
|
||||
if systemctl is-active --quiet radvd 2>/dev/null; then
|
||||
logger -t vrrp-wan "not MASTER: stopping radvd (deprecates the v6 gateway)"
|
||||
systemctl stop radvd 2>/dev/null
|
||||
fi
|
||||
ip link show "$V6_TUNNEL" 2>/dev/null | grep -q 'state DOWN' && return 0
|
||||
logger -t vrrp-wan "not MASTER: bringing $V6_TUNNEL down"
|
||||
ip link set "$V6_TUNNEL" down 2>/dev/null
|
||||
}
|
||||
|
||||
# --- decide ----------------------------------------------------------------
|
||||
if holds_vip; then
|
||||
echo master > "$STATE/role"
|
||||
[ -f "$STATE/since" ] || date +%s > "$STATE/since"
|
||||
ppp_dial
|
||||
# The WAN block is now an `if` rather than an early exit, so the IPv6 plane
|
||||
# below is reached on EVERY tick and not only on the one that enables the
|
||||
# WAN. On the promotion tick bond0.53 has no address yet, so v6_take no-ops
|
||||
# and the next tick takes it.
|
||||
if wan_disabled; then
|
||||
logger -t vrrp-wan "MASTER with bond0.${WAN_VIF} disabled -> enabling"
|
||||
t0=$(date +%s)
|
||||
"$APPLY" enable 9>&-
|
||||
logger -t vrrp-wan "bond0.${WAN_VIF} enable commit took $(( $(date +%s) - t0 ))s"
|
||||
fi
|
||||
v6_take
|
||||
else
|
||||
echo backup > "$STATE/role"
|
||||
rm -f "$STATE/since" "$STATE/holdoff"
|
||||
ppp_release
|
||||
# Before the WAN goes, not after: once bond0.53 is disabled the source
|
||||
# address is gone and radvd's farewell advertisement has no path out.
|
||||
v6_release
|
||||
if ! wan_disabled; then
|
||||
logger -t vrrp-wan "not MASTER but bond0.${WAN_VIF} enabled -> releasing"
|
||||
t0=$(date +%s)
|
||||
"$APPLY" disable 9>&-
|
||||
logger -t vrrp-wan "bond0.${WAN_VIF} disable commit took $(( $(date +%s) - t0 ))s"
|
||||
fi
|
||||
fi
|
||||
|
||||
# No `save`, deliberately. config.boot keeps `vif 53 disable` on BOTH routers, so
|
||||
# a reboot in any order comes up unable to claim the cloned MAC. PPPoE needs no
|
||||
# such convention any more: with the gate, config.boot is safe by construction
|
||||
# and a stray `save` cannot make both boxes dial.
|
||||
12
migration/vrrp-wan-reconcile.service
Normal file
12
migration/vrrp-wan-reconcile.service
Normal file
@@ -0,0 +1,12 @@
|
||||
[Unit]
|
||||
# Belt to the transition scripts' braces. VyOS's keepalived-fifo.py helper was
|
||||
# observed dropping a MASTER transition silently, leaving a router holding every
|
||||
# VIP with no WAN. A timer cannot be dropped the same way.
|
||||
Description=Reconcile WAN interface state with VRRP mastership
|
||||
# vyos-router loads config at boot and its pppoe handler will try to dial; the
|
||||
# gate drop-in blocks that, but ordering after it keeps the logs readable.
|
||||
After=keepalived.service vyos-router.service
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/config/vrrp-wan-reconcile
|
||||
13
migration/vrrp-wan-reconcile.timer
Normal file
13
migration/vrrp-wan-reconcile.timer
Normal file
@@ -0,0 +1,13 @@
|
||||
[Unit]
|
||||
Description=Reconcile WAN with VRRP mastership every 30s
|
||||
|
||||
[Timer]
|
||||
# 30s: fast enough that a dropped transition is a blip rather than an outage,
|
||||
# slow enough that it is never the thing generating load. It only commits when
|
||||
# state actually disagrees, so a steady-state tick is two `ip` calls and a grep.
|
||||
OnBootSec=60
|
||||
OnUnitActiveSec=30
|
||||
AccuracySec=5
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
5
migration/vrrp-wan-release
Normal file
5
migration/vrrp-wan-release
Normal file
@@ -0,0 +1,5 @@
|
||||
#!/bin/sh
|
||||
# VRRP transition hook. One code path: the reconciler derives everything from
|
||||
# ground truth, so take and release are the same operation asked at different
|
||||
# moments. Speed comes from here; correctness comes from the timer.
|
||||
exec /config/vrrp-wan-reconcile
|
||||
5
migration/vrrp-wan-take
Normal file
5
migration/vrrp-wan-take
Normal file
@@ -0,0 +1,5 @@
|
||||
#!/bin/sh
|
||||
# VRRP transition hook. One code path: the reconciler derives everything from
|
||||
# ground truth, so take and release are the same operation asked at different
|
||||
# moments. Speed comes from here; correctness comes from the timer.
|
||||
exec /config/vrrp-wan-reconcile
|
||||
71
migration/vrrp-wan.conf
Normal file
71
migration/vrrp-wan.conf
Normal file
@@ -0,0 +1,71 @@
|
||||
# Settings for the vrrp-wan scripts. Installed to /config/vrrp-wan.conf.
|
||||
#
|
||||
# Why a file and not systemd Environment=: keepalived invokes vrrp-wan-health
|
||||
# with NO environment at all, so an Environment= line in the .service would be
|
||||
# read by the reconciler and ignored by the health check -- two sources of truth
|
||||
# for the one value that decides who is master. It is also what stops a repeat
|
||||
# of 2026-09-05, when the sim's proof was obtained from scripts hand-`sed`-ed in
|
||||
# place: /config/vrrp-wan-health differed from git, and a reinstall would have
|
||||
# silently reverted the tested behaviour.
|
||||
|
||||
# The management VIP. "Do I hold this address" IS the definition of master here
|
||||
# -- ground truth, not a marker written by a script that may not have run.
|
||||
VRRP_WAN_VIP=192.168.1.1
|
||||
|
||||
# The DHCP WAN sub-interface. Stays on the config plane because its lease is
|
||||
# bound to a cloned MAC, which only VyOS config can move.
|
||||
WAN_VIF=53
|
||||
|
||||
# Seconds a new master may go without any WAN before the health check fails it.
|
||||
#
|
||||
# Must exceed the ISP's stale-session hold-down, or a hard failover blows the
|
||||
# window and BOTH routers end up in FAULT -- worse than the outage the check
|
||||
# exists to prevent.
|
||||
#
|
||||
# MEASURED, labsim T4, master destroyed with `virsh destroy`, time until the
|
||||
# survivor held a PPPoE session (labsim/wan-failover-evidence/T4-*). Two
|
||||
# independent runs, so these are the AC's behaviour rather than one-offs:
|
||||
#
|
||||
# session-control=replace 26s / 26s
|
||||
# session-control=deny 148s / 141s <-- worst
|
||||
# session-control=disable 21s / 20s
|
||||
#
|
||||
# `deny` is the hostile case and the only one that matters for sizing: the AC
|
||||
# refuses the survivor until its own dead-peer timer frees the dead session.
|
||||
# The poller caught it happening -- the destroyed router's session stayed in the
|
||||
# table while the survivor's dial attempts appeared and were rejected, twice,
|
||||
# before it finally got in at 148s.
|
||||
#
|
||||
# 148s also lands well past the theoretical lcp-echo-interval(30) x
|
||||
# failure(3) = 90s budget that 180 was originally sized against, which left only
|
||||
# 32s of margin. 300 gives roughly 2x the worst observed, on IDLE 2-vCPU sim
|
||||
# VMs; the VP2440s under kea, BGP and conntrack will be slower, and Vodafone's
|
||||
# actual policy and timers are unknown.
|
||||
#
|
||||
# The cost is real and worth stating: this is also how long a master that is
|
||||
# alive but genuinely cannot route keeps holding every VIP before yielding --
|
||||
# the 2026-09-02 outage shape. That case is mostly covered by bond0.53, which
|
||||
# satisfies the check within seconds of getting a DHCP lease; GRACE only
|
||||
# dominates when PPPoE is the only path left.
|
||||
#
|
||||
# Do not lower this below the worst measured handover without re-running
|
||||
# `labsim/labsim-pppoe-ha-test.sh --hard`. Before 2026-09-06 that matrix never
|
||||
# actually set session-control and reported `deny` at 25s -- a number that did
|
||||
# not exist.
|
||||
GRACE=300
|
||||
|
||||
# may-dial is a LEASE, not a flag. vrrp-wan-reconcile renews its mtime every
|
||||
# tick; vrrp-wan-guard revokes it once it goes stale. A plain flag survives the
|
||||
# reconciler dying, and a router that stops reconciling while demoted would keep
|
||||
# dialling for ever.
|
||||
LEASE_TTL=75
|
||||
|
||||
# Flap damper. Two routers that both believe they hold the VIP (a VRRP
|
||||
# partition) will both dial; with the AC set to `replace` each dial kills the
|
||||
# other's session, the loser's pppd exits non-zero, systemd redials in 5s, and
|
||||
# the pair hammers the access concentrator indefinitely. Against a real ISP that
|
||||
# is how an account gets rate-limited. More than FLAP_MAX dials in FLAP_WINDOW
|
||||
# puts this box in hold-off and logs loudly.
|
||||
FLAP_MAX=6
|
||||
FLAP_WINDOW=600
|
||||
FLAP_HOLDOFF=900
|
||||
120
migration/vyos-known-good
Executable file
120
migration/vyos-known-good
Executable file
@@ -0,0 +1,120 @@
|
||||
#!/bin/vbash
|
||||
# Pin a config you have SEEN working, and get back to it with one command.
|
||||
#
|
||||
# Why this exists when VyOS already has rollback: `rollback 1` returns you to the
|
||||
# previous revision, which may itself be broken -- you can walk backwards through
|
||||
# several bad commits looking for the one that worked. This pins a state you
|
||||
# explicitly confirmed was good, so recovery is one step and does not require
|
||||
# remembering how many changes ago things last worked.
|
||||
#
|
||||
# It is deliberately NOT automatic. A config is only "known good" once a human
|
||||
# has used the network and found it working; a script cannot judge that, and a
|
||||
# snapshot taken automatically after every commit would faithfully preserve the
|
||||
# broken one.
|
||||
#
|
||||
# vyos-known-good save mark the running config as known-good
|
||||
# vyos-known-good status when it was taken, and how it differs from running
|
||||
# vyos-known-good restore go back to it (commit-confirmed, so even this is safe)
|
||||
# vyos-known-good diff what would change if you restored
|
||||
#
|
||||
# Lives in /config so it survives image upgrades, like vyos-unifi-switch.
|
||||
|
||||
# Capture the arguments BEFORE sourcing script-template: sourcing it resets the
|
||||
# positional parameters, so $1 is empty by the time the case statement runs and
|
||||
# every invocation silently falls through to the usage message.
|
||||
ACTION="${1:-status}"
|
||||
|
||||
source /opt/vyatta/etc/functions/script-template
|
||||
|
||||
GOOD="/config/known-good.boot"
|
||||
META="/config/known-good.meta"
|
||||
RUNNING="/config/config.boot"
|
||||
CONFIRM_MINUTES="${CONFIRM_MINUTES:-5}"
|
||||
|
||||
say() { printf '\033[0;36m[known-good]\033[0m %s\n' "$*"; }
|
||||
warn() { printf '\033[1;33m[known-good]\033[0m %s\n' "$*" >&2; }
|
||||
die() { printf '\033[0;31m[known-good]\033[0m %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
# The running config on disk is only current if nothing is uncommitted-and-unsaved.
|
||||
# Saving a snapshot that does not match what is actually running would be worse
|
||||
# than having no snapshot at all -- it would look like a safety net and not be one.
|
||||
require_saved() {
|
||||
if ! cli-shell-api sessionChanged >/dev/null 2>&1; then
|
||||
return 0
|
||||
fi
|
||||
die "there are uncommitted changes; commit and save first, or this snapshot would not match reality"
|
||||
}
|
||||
|
||||
cmd_save() {
|
||||
require_saved
|
||||
[ -r "$RUNNING" ] || die "cannot read $RUNNING"
|
||||
sudo cp "$RUNNING" "$GOOD"
|
||||
# 0660 root:vyattacfg, matching /config/config.boot. 0600 would make the
|
||||
# snapshot unreadable to the vyos user, so `status` and `diff` -- the two you
|
||||
# run while deciding whether to restore -- would silently show nothing.
|
||||
sudo chmod 0660 "$GOOD"; sudo chgrp vyattacfg "$GOOD"
|
||||
sudo chmod 0660 "$META" 2>/dev/null; sudo chgrp vyattacfg "$META" 2>/dev/null
|
||||
{
|
||||
echo "saved_at=$(date -Is)"
|
||||
echo "saved_by=${SUDO_USER:-$USER}"
|
||||
echo "hostname=$(hostname)"
|
||||
echo "lines=$(wc -l < "$RUNNING")"
|
||||
} | sudo tee "$META" >/dev/null
|
||||
say "pinned $(wc -l < "$GOOD") lines as known-good on $(hostname)"
|
||||
say "restore with: /config/vyos-known-good restore"
|
||||
}
|
||||
|
||||
cmd_status() {
|
||||
[ -r "$GOOD" ] || { warn "no known-good snapshot on $(hostname) -- run 'save' while things work"; return 1; }
|
||||
say "known-good on $(hostname):"
|
||||
sed 's/^/ /' "$META" 2>/dev/null
|
||||
local n
|
||||
n="$(diff <(grep -vE '^\s*$' "$GOOD") <(grep -vE '^\s*$' "$RUNNING") 2>/dev/null | grep -c '^[<>]')"
|
||||
if [ "${n:-0}" -eq 0 ]; then
|
||||
say "running config MATCHES known-good"
|
||||
else
|
||||
warn "running config differs from known-good by $n line(s) -- 'diff' to see them"
|
||||
fi
|
||||
}
|
||||
|
||||
cmd_diff() {
|
||||
[ -r "$GOOD" ] || die "no known-good snapshot"
|
||||
diff -u "$GOOD" "$RUNNING" | sed -E "s/(password|key|secret)[[:space:]]+\S+/\1 <REDACTED>/I" || true
|
||||
}
|
||||
|
||||
cmd_restore() {
|
||||
[ -r "$GOOD" ] || die "no known-good snapshot to restore"
|
||||
say "restoring known-good on $(hostname) (taken $(grep -m1 saved_at "$META" 2>/dev/null | cut -d= -f2-))"
|
||||
|
||||
# Commit-confirmed even here. If the known-good snapshot is itself somehow
|
||||
# wrong, or the restore cannot be confirmed because access is still broken,
|
||||
# the router undoes it rather than leaving you worse off. Silence reverts.
|
||||
local script; script="$(mktemp)"
|
||||
{
|
||||
echo 'source /opt/vyatta/etc/functions/script-template'
|
||||
echo 'configure'
|
||||
echo "load $GOOD"
|
||||
printf 'sudo sg vyattacfg "/usr/bin/config-mgmt commit_confirm -y -t=%s"\n' "$CONFIRM_MINUTES"
|
||||
echo 'export IN_COMMIT_CONFIRM=t'
|
||||
echo 'commit'
|
||||
echo 'unset IN_COMMIT_CONFIRM'
|
||||
echo 'exit'
|
||||
} > "$script"
|
||||
vbash "$script"; local rc=$?
|
||||
rm -f "$script"
|
||||
|
||||
[ $rc -eq 0 ] || die "restore failed (rc=$rc) -- nothing was committed"
|
||||
say ""
|
||||
say "RESTORED under a ${CONFIRM_MINUTES} minute timer."
|
||||
say "Check the network NOW. If it works, confirm it:"
|
||||
say " sudo sg vyattacfg '/usr/bin/config-mgmt confirm'"
|
||||
say "If you do nothing, the router reverts on its own."
|
||||
}
|
||||
|
||||
case "$ACTION" in
|
||||
save) cmd_save ;;
|
||||
status) cmd_status ;;
|
||||
diff) cmd_diff ;;
|
||||
restore) cmd_restore ;;
|
||||
*) die "usage: vyos-known-good {save|status|diff|restore}" ;;
|
||||
esac
|
||||
@@ -243,22 +243,38 @@ def build_delta(inv: dict, priority: int, wan_user: str, with_wan: bool,
|
||||
]
|
||||
|
||||
if not with_wan:
|
||||
# The backup carries the identical WAN and NAT config but with the
|
||||
# interfaces administratively DOWN. The cloned MAC is therefore never
|
||||
# live on two boxes at once, while everything needed to route and
|
||||
# masquerade is already present -- taking over is enabling two
|
||||
# interfaces, not rebuilding a config under pressure.
|
||||
# Both boxes carry the identical WAN and NAT config; only the RESTING
|
||||
# STATE differs, and only for the DHCP line. Takeover is no longer a
|
||||
# human deleting two lines under pressure -- vrrp-wan-reconcile does it,
|
||||
# driven by who holds the management VIP. See migration/PPPOE-HA.md.
|
||||
#
|
||||
# bond0.53 stays here, on the CONFIG plane, because its lease is bound
|
||||
# to a cloned MAC and only VyOS config can move a MAC between boxes.
|
||||
# This is the "nothing to follow" default: a freshly built or PXE'd box
|
||||
# has no live master to imitate, so it must come up unable to claim that
|
||||
# MAC. On a running pair the model follows reality instead -- see the
|
||||
# export-before-apply rule in migration/PPPOE-HA.md.
|
||||
#
|
||||
# pppoe0 is deliberately NOT disabled here any more. `disable` unlinks
|
||||
# /etc/ppp/peers/pppoe0, which is pppd's own options file, so it
|
||||
# destroys what the promotion path needs and leaves ppp@pppoe0
|
||||
# restart-looping. Dialling is gated at the systemd unit instead.
|
||||
#
|
||||
# ORDERING TRAP for a rebuilt box: because pppoe0 is left ENABLED,
|
||||
# interfaces_pppoe.py will try to dial on the first commit that touches
|
||||
# the pppoe subtree. Install the gate FIRST --
|
||||
# `migration/vrrp-wan-install --vip <mgmt VIP> --host vyos@<box>` --
|
||||
# or the new box will take the single ISP session off the live master.
|
||||
#
|
||||
# NAT rules naming a down interface are harmless: VyOS warns at commit
|
||||
# ("Interface ... does not exist!") and commits anyway, verified.
|
||||
out += [
|
||||
"",
|
||||
"# --- WAN held DOWN on this box -----------------------------",
|
||||
"# Enable these two to take over the internet path:",
|
||||
f"# set interfaces bonding bond0 vif {WAN_DHCP_VIF.split('.')[1]} disable <- delete this",
|
||||
f"# set interfaces pppoe {WAN_PPPOE_IF} disable <- and this",
|
||||
"# --- 10 gig held DOWN on this box --------------------------",
|
||||
"# Do NOT enable by hand: vrrp-wan-reconcile owns this, keyed on",
|
||||
"# whoever holds the management VIP. pppoe0 is gated at the unit",
|
||||
"# (ppp@pppoe0.service.d/10-vrrp-wan-gate.conf), not in config.",
|
||||
f"set interfaces bonding bond0 vif {WAN_DHCP_VIF.split('.')[1]} disable",
|
||||
f"set interfaces pppoe {WAN_PPPOE_IF} disable",
|
||||
]
|
||||
|
||||
if True:
|
||||
|
||||
85
migration/vyos002-catch.sh
Executable file
85
migration/vyos002-catch.sh
Executable file
@@ -0,0 +1,85 @@
|
||||
#!/bin/bash
|
||||
# Arm BEFORE powering vyos002 on. Strips the eth2 address the moment the box is
|
||||
# reachable, then installs the VRRP health-check.
|
||||
#
|
||||
# Why this exists: vyos002 boots with `interfaces ethernet eth2 address
|
||||
# 192.168.8.144/23` still in config.boot -- the same subnet as bond0.2. Linux
|
||||
# answers ARP for any local address out of any interface on that L2, so eth2
|
||||
# answers for addresses that bond0.2 is supposed to route, and traffic lands on
|
||||
# a port that does not route it. That is the 2026-09-02 cluster outage.
|
||||
#
|
||||
# A human "jumping on it fast" loses this race more often than not; the box is
|
||||
# reachable within a second or two of the interfaces coming up. This polls at
|
||||
# 1s and commits the moment it gets in.
|
||||
#
|
||||
# ./vyos002-catch.sh arm and wait (Ctrl-C to disarm)
|
||||
#
|
||||
# Bounded risk while you wait, worth knowing: vyos002 comes up BACKUP (priority
|
||||
# 100, no-preempt, vyos001 healthy MASTER), so it does NOT hold 192.168.8.1 and
|
||||
# the GATEWAY cannot be poisoned. The exposure is its own bond0.2 address, which
|
||||
# is survivable. The unbounded case is it becoming MASTER while eth2 is present
|
||||
# -- which is exactly what the health-check in step 2 prevents.
|
||||
set -uo pipefail
|
||||
|
||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
PW="${VYOS_PW:-vyos}"
|
||||
# Management first because it comes up with the box; LoT is the fallback and is
|
||||
# L2-direct from this workstation (see RECOVERY-CARD-vlan1-move.md).
|
||||
TARGETS=("192.168.1.253" "10.0.1.253")
|
||||
SSH_OPTS=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=2 -o PreferredAuthentications=password)
|
||||
|
||||
log() { printf '\033[36m[catch %s]\033[0m %s\n' "$(date +%T)" "$*"; }
|
||||
|
||||
on() { timeout 12 sshpass -p "$PW" ssh "${SSH_OPTS[@]}" "vyos@$1" "$@"; }
|
||||
|
||||
log "armed -- polling ${TARGETS[*]} every 1s. Power on vyos002 now."
|
||||
HOST=""
|
||||
while [ -z "$HOST" ]; do
|
||||
for t in "${TARGETS[@]}"; do
|
||||
if timeout 4 sshpass -p "$PW" ssh "${SSH_OPTS[@]}" "vyos@$t" true 2>/dev/null; then
|
||||
HOST="$t"; break
|
||||
fi
|
||||
done
|
||||
[ -z "$HOST" ] && sleep 1
|
||||
done
|
||||
log "CAUGHT on $HOST -- stripping eth2 address"
|
||||
|
||||
# Step 1, on its own commit: get the address off eth2 before anything else. Any
|
||||
# extra command in this commit is extra seconds of exposure.
|
||||
timeout 90 sshpass -p "$PW" ssh "${SSH_OPTS[@]}" "vyos@$HOST" 'vbash -s' <<'EOF' 2>&1 | tail -3
|
||||
source /opt/vyatta/etc/functions/script-template
|
||||
delete interfaces ethernet eth2 address
|
||||
commit
|
||||
echo "ETH2_RC=$?"
|
||||
save
|
||||
exit
|
||||
EOF
|
||||
log "eth2 address removed"
|
||||
|
||||
# Step 2: the health-check. Without it this box can hold every floating IP while
|
||||
# having no WAN -- the outage itself. Copy the script BEFORE referencing it, or
|
||||
# the commit succeeds and the check silently never passes.
|
||||
timeout 30 sshpass -p "$PW" scp "${SSH_OPTS[@]}" \
|
||||
"$HERE/vrrp-wan-health" "vyos@$HOST:/tmp/vrrp-wan-health" >/dev/null 2>&1 \
|
||||
&& log "health-check script copied" || log "WARN: scp failed -- step 2 will be skipped"
|
||||
|
||||
timeout 90 sshpass -p "$PW" ssh "${SSH_OPTS[@]}" "vyos@$HOST" 'vbash -s' <<'EOF' 2>&1 | tail -3
|
||||
sudo install -o root -g vyattacfg -m 0775 /tmp/vrrp-wan-health /config/vrrp-wan-health
|
||||
source /opt/vyatta/etc/functions/script-template
|
||||
set high-availability vrrp sync-group MAIN health-check script '/config/vrrp-wan-health'
|
||||
set high-availability vrrp sync-group MAIN health-check interval '5'
|
||||
set high-availability vrrp sync-group MAIN health-check failure-count '3'
|
||||
commit
|
||||
echo "HEALTH_RC=$?"
|
||||
save
|
||||
exit
|
||||
EOF
|
||||
log "health-check installed"
|
||||
|
||||
echo
|
||||
log "=== state ==="
|
||||
timeout 30 sshpass -p "$PW" ssh "${SSH_OPTS[@]}" "vyos@$HOST" \
|
||||
'echo "-- eth2 (must show no inet) --"; ip -4 addr show eth2 2>/dev/null | grep inet || echo " none"
|
||||
echo "-- health-check --"; sudo /config/vrrp-wan-health; echo " exit=$? (non-zero = no WAN = refuses the VIPs, which is CORRECT and safe)"
|
||||
echo "-- vrrp --"; /opt/vyatta/bin/vyatta-op-cmd-wrapper show vrrp' 2>&1
|
||||
37
migration/vyos002-return.conf
Normal file
37
migration/vyos002-return.conf
Normal file
@@ -0,0 +1,37 @@
|
||||
# vyos002 — everything that must land before it is trusted on the network.
|
||||
#
|
||||
# Apply order matters only in that this is ONE commit: the eth2 address and the
|
||||
# missing health-check are the two defects that caused the 2026-09-02 outage, and
|
||||
# neither should survive a single reboot window.
|
||||
#
|
||||
# ssh vyos@192.168.1.253 (or 10.0.1.253 — L2-direct, see RECOVERY-CARD)
|
||||
# configure; <paste>; commit; save
|
||||
|
||||
# 1. The ARP poisoner. eth2 is the 1G copper NIC on US24 port 16, native VLAN 2,
|
||||
# and it held 192.168.8.144/23 -- the same subnet as bond0.2. Two interfaces
|
||||
# answering for one subnet is what hijacked 192.168.8.1 and took the cluster
|
||||
# down: eth2's MAC answered while bond0.2's MAC routed.
|
||||
# Origin: eth2 was `address dhcp`, a kea reservation for the ROUTER'S OWN NIC
|
||||
# handed it .144, and a CLI commit froze it static.
|
||||
delete interfaces ethernet eth2 address
|
||||
|
||||
# 2. The health-check. Without it this box can hold every floating IP while
|
||||
# having no WAN at all -- the outage itself. It goes on the SYNC GROUP; VyOS
|
||||
# rejects it per-group.
|
||||
# /config/vrrp-wan-health must be copied over FIRST (from migration/) and be
|
||||
# chmod +x, or the commit succeeds and the check silently never passes.
|
||||
set high-availability vrrp sync-group MAIN health-check script '/config/vrrp-wan-health'
|
||||
set high-availability vrrp sync-group MAIN health-check interval '5'
|
||||
set high-availability vrrp sync-group MAIN health-check failure-count '3'
|
||||
|
||||
# 3. Management onto a tagged sub-interface, matching vyos001 (kea #1117).
|
||||
# Pair this with USW Aggregation port 3 (LAG 3+4) -> Native VLAN = None.
|
||||
# Until that switch change lands, leave these three commented out: vyos002
|
||||
# can run untagged on bond0 while vyos001 runs tagged -- one VLAN is one
|
||||
# broadcast domain, and the coexistence was proven in labsim.
|
||||
# set interfaces bonding bond0 vif 1 address '192.168.1.253/24'
|
||||
# set interfaces bonding bond0 vif 1 description 'management'
|
||||
# delete interfaces bonding bond0 address
|
||||
# set firewall group interface-group LAN interface 'bond0.1'
|
||||
# delete firewall group interface-group LAN interface 'bond0'
|
||||
# set high-availability vrrp group native interface 'bond0.1'
|
||||
159
migration/wan-drill
Executable file
159
migration/wan-drill
Executable file
@@ -0,0 +1,159 @@
|
||||
#!/usr/bin/env bash
|
||||
# Controlled WAN failover drill. Runs from the workstation over the LAN.
|
||||
#
|
||||
# It is written as ONE unattended script on purpose. The drill takes the
|
||||
# household's internet down, so anything driving it step-by-step from off-site
|
||||
# -- a person on a laptop, or an agent that needs the internet to think --
|
||||
# stops being able to act at exactly the moment it matters. This script only
|
||||
# needs the LAN, and it always runs its cleanup.
|
||||
#
|
||||
# ./wan-drill run it
|
||||
# ./wan-drill --dry show what it would do, touch nothing
|
||||
#
|
||||
# Recovery if this script itself dies: see migration/RECOVERY-CARD-wan-panic.md
|
||||
# or /config/RECOVERY-CARD.md on either router. Short version, on vyos002:
|
||||
# sudo /config/wan-panic
|
||||
set -uo pipefail
|
||||
|
||||
P1=10.0.1.252 # vyos001, normally MASTER
|
||||
P2=10.0.1.253 # vyos002, normally BACKUP
|
||||
PW=vyos
|
||||
SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null
|
||||
-o LogLevel=ERROR -o ConnectTimeout=5)
|
||||
TAKEOVER_BUDGET=240 # give vyos002 this long to raise a WAN
|
||||
FAILBACK_BUDGET=180 # and vyos001 this long to take it back
|
||||
WATCHDOG_HOLD=150 # vyos002 stands down by itself after this with no WAN
|
||||
|
||||
LOG="${LOG:-/tmp/wan-drill-$(date +%H%M%S).log}"
|
||||
DRY=0; [ "${1:-}" = "--dry" ] && DRY=1
|
||||
|
||||
say() { printf '%s %s\n' "$(date +%T)" "$*" | tee -a "$LOG"; }
|
||||
r() { timeout 25 sshpass -p "$PW" ssh "${SSH[@]}" "vyos@$1" "${@:2}" 2>/dev/null; }
|
||||
|
||||
holder() { for h in "$P1" "$P2"; do
|
||||
[ "$(r "$h" 'ip -4 -o addr show | grep -c " 192.168.1.1/"' | tr -d ' \n')" != 0 ] \
|
||||
&& { echo "$h"; return; }; done; echo none; }
|
||||
wan_of() { r "$1" 'for i in bond0.53 pppoe0; do a=$(ip -4 addr show dev $i 2>/dev/null | sed -n "s/.*inet \([0-9.]*\).*/\1/p"); [ -n "$a" ] && printf "%s=%s " $i $a; done'; }
|
||||
online() { [ "$(r "$1" 'ping -c1 -W2 9.9.9.9 >/dev/null 2>&1 && echo y' | tr -d ' \n')" = y ]; }
|
||||
|
||||
# --- IPv6 -------------------------------------------------------------------
|
||||
# Until 2026-09-06 this drill measured IPv4 only, and PPPOE-HA.md recorded that
|
||||
# "IPv6 stayed up" through a failover on the strength of a reading taken outside
|
||||
# the window. It cannot have: the reconciler disables bond0.53 on the demoted
|
||||
# box, so 87.192.101.48 moves to the survivor, and inbound protocol 41 from HE
|
||||
# then lands on whichever router owns the tunnel. Measure it rather than assume.
|
||||
#
|
||||
# Quad9 again, so the v6 result is comparable with the v4 one on the line above.
|
||||
online6() { [ "$(r "$1" 'ping -6 -c1 -W2 2620:fe::fe >/dev/null 2>&1 && echo y' | tr -d ' \n')" = y ]; }
|
||||
# Tunnel link state and source address. The source is the interesting half: it
|
||||
# should be IDENTICAL before and after a router-level failover, because the 10
|
||||
# gig lease follows the cloned MAC. A changed source means something called the
|
||||
# HE API during the drill, which a router failover must never need to do.
|
||||
tun_of() { r "$1" 'ip tunnel show tun0 2>/dev/null | sed -nE "s/.* local ([0-9.]+).*/\1/p"' | tr -d ' \n'; }
|
||||
he_calls(){ r "$1" 'sudo journalctl -t he-tunnel-follow --since "'"$2"'" --no-pager 2>/dev/null | grep -c "HE endpoint set to"' | tr -d ' \n'; }
|
||||
|
||||
cleanup() {
|
||||
say "--- cleanup (always runs) ---"
|
||||
r "$P2" 'sudo /config/wan-drill-watchdog disarm' >/dev/null
|
||||
r "$P1" 'sudo rm -f /run/vrrp-wan/force-fault' >/dev/null
|
||||
r "$P2" 'sudo rm -f /run/vrrp-wan/force-fault' >/dev/null
|
||||
sleep 20
|
||||
say "final holder : $(holder)"
|
||||
say "final WAN : $P1 [$(wan_of $P1)] $P2 [$(wan_of $P2)]"
|
||||
online "$P1" && say "final internet: UP via $P1" || {
|
||||
online "$P2" && say "final internet: UP via $P2" \
|
||||
|| say "final internet: *** DOWN -- run: sudo /config/wan-panic on $P2 ***"; }
|
||||
say "log: $LOG"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
say "=== pre-flight ==="
|
||||
DRILL_START="$(date '+%Y-%m-%d %H:%M:%S')"
|
||||
start_holder="$(holder)"
|
||||
say "holder now : $start_holder"
|
||||
say "$P1 WAN : $(wan_of $P1)"
|
||||
say "$P2 WAN : $(wan_of $P2)"
|
||||
online "$P1" && say "internet : UP via $P1" || { say "internet ALREADY DOWN -- refusing to drill"; exit 1; }
|
||||
tun_before="$(tun_of $P1)"; tun2_before="$(tun_of $P2)"
|
||||
say "$P1 tun0 src : ${tun_before:-<no tunnel>}"
|
||||
say "$P2 tun0 src : ${tun2_before:-<no tunnel -- IPv6 cannot survive a failover>}"
|
||||
if online6 "$P1"; then say "IPv6 : UP via $P1"
|
||||
else say "IPv6 : DOWN on $P1 before we start -- v6 figures below are not meaningful"; fi
|
||||
[ "$start_holder" = "$P1" ] || { say "expected $P1 to hold the VIP, got $start_holder -- refusing"; exit 1; }
|
||||
|
||||
if [ "$DRY" = 1 ]; then
|
||||
say "--dry: would arm the watchdog on $P2 (${WATCHDOG_HOLD}s) and force-fault $P1"
|
||||
trap - EXIT; exit 0
|
||||
fi
|
||||
|
||||
say "=== arming auto-abort on $P2 ==="
|
||||
r "$P2" "sudo /config/wan-drill-watchdog arm $WATCHDOG_HOLD" | tee -a "$LOG"
|
||||
|
||||
say "=== DRILL: force-faulting $P1 ==="
|
||||
t0=$(date +%s)
|
||||
r "$P1" 'sudo touch /run/vrrp-wan/force-fault'
|
||||
|
||||
took=""; took6=""
|
||||
while [ $(( $(date +%s) - t0 )) -lt "$TAKEOVER_BUDGET" ]; do
|
||||
sleep 5
|
||||
h="$(holder)"; w="$(wan_of $P2)"
|
||||
# v6 keeps being probed after v4 comes back, because the two recover
|
||||
# independently and the gap between them IS the number this drill exists to
|
||||
# produce. Stop only when both are up, or the budget runs out.
|
||||
[ -z "$took6" ] && online6 "$P2" && took6=$(( $(date +%s) - t0 ))
|
||||
say " t+$(( $(date +%s) - t0 ))s holder=$h vyos002_wan=[$w] v6=$([ -n "$took6" ] && echo up || echo down)"
|
||||
if [ -z "$took" ] && [ "$h" = "$P2" ] && [ -n "$w" ] && online "$P2"; then
|
||||
took=$(( $(date +%s) - t0 ))
|
||||
fi
|
||||
[ -n "$took" ] && [ -n "$took6" ] && break
|
||||
done
|
||||
|
||||
if [ -n "$took" ]; then
|
||||
say "*** TAKEOVER OK: $P2 held the VIP and reached the internet in ${took}s ***"
|
||||
else
|
||||
say "*** TAKEOVER FAILED within ${TAKEOVER_BUDGET}s -- failing back ***"
|
||||
fi
|
||||
if [ -n "$took6" ] && [ -n "$took" ]; then
|
||||
say "*** IPv6 followed in ${took6}s (v4 ${took}s, gap $(( took6 - took ))s) ***"
|
||||
elif [ -n "$took6" ]; then
|
||||
say "*** IPv6 followed in ${took6}s, but IPv4 never did ***"
|
||||
else
|
||||
say "*** IPv6 did NOT return within ${TAKEOVER_BUDGET}s on $P2 -- the v6 estate is down for the whole takeover ***"
|
||||
fi
|
||||
|
||||
say "=== failing back to $P1 ==="
|
||||
t1=$(date +%s)
|
||||
r "$P2" 'sudo /config/wan-drill-watchdog disarm' >/dev/null
|
||||
r "$P1" 'sudo rm -f /run/vrrp-wan/force-fault'
|
||||
r "$P2" 'sudo touch /run/vrrp-wan/force-fault'
|
||||
back=""; back6=""
|
||||
while [ $(( $(date +%s) - t1 )) -lt "$FAILBACK_BUDGET" ]; do
|
||||
sleep 5
|
||||
h="$(holder)"; w="$(wan_of $P1)"
|
||||
[ -z "$back6" ] && online6 "$P1" && back6=$(( $(date +%s) - t1 ))
|
||||
say " t+$(( $(date +%s) - t1 ))s holder=$h vyos001_wan=[$w] v6=$([ -n "$back6" ] && echo up || echo down)"
|
||||
if [ -z "$back" ] && [ "$h" = "$P1" ] && [ -n "$w" ] && online "$P1"; then
|
||||
back=$(( $(date +%s) - t1 ))
|
||||
fi
|
||||
[ -n "$back" ] && [ -n "$back6" ] && break
|
||||
done
|
||||
[ -n "$back" ] && say "*** FAILBACK OK in ${back}s ***" \
|
||||
|| say "*** FAILBACK FAILED -- cleanup will clear both levers ***"
|
||||
[ -n "$back6" ] && say "*** IPv6 back in ${back6}s ***" \
|
||||
|| say "*** IPv6 did NOT return within ${FAILBACK_BUDGET}s on $P1 ***"
|
||||
|
||||
# The invariant a router-level failover must satisfy: the HE endpoint is never
|
||||
# touched. 87.192.101.48 follows the cloned MAC to the other box, so the tunnel
|
||||
# source is the same address on either router and there is nothing to tell HE.
|
||||
# A non-zero count here means something re-pointed the tunnel at a PPPoE address
|
||||
# -- which Vodafone re-issues on every dial, so it would be wrong within minutes.
|
||||
say "=== IPv6 invariants ==="
|
||||
tun_after="$(tun_of $P1)"
|
||||
say "tun0 src : ${tun_before:-none} -> ${tun_after:-none}"
|
||||
[ "$tun_before" = "$tun_after" ] && say " OK: tunnel source unchanged across the drill" \
|
||||
|| say " *** CHANGED -- a router failover should never move the HE endpoint ***"
|
||||
for h in "$P1" "$P2"; do
|
||||
n="$(he_calls "$h" "$DRILL_START")"
|
||||
say "HE updates from $h since ${DRILL_START}: ${n:-?}"
|
||||
[ "${n:-0}" = 0 ] || say " *** $h called the HE API during a router failover -- it should not need to ***"
|
||||
done
|
||||
83
migration/wan-drill-watchdog
Normal file
83
migration/wan-drill-watchdog
Normal file
@@ -0,0 +1,83 @@
|
||||
#!/bin/sh
|
||||
# Auto-abort for a WAN failover drill. Armed on the router that is EXPECTED TO
|
||||
# TAKE OVER, before the drill starts.
|
||||
#
|
||||
# The drill takes the internet down for as long as the new master needs to
|
||||
# raise a WAN. If it never does, whoever is running the drill has no internet
|
||||
# either -- and if that is an agent, it simply stops responding mid-incident.
|
||||
# So the abort cannot depend on anyone being there.
|
||||
#
|
||||
# Rule: if I hold the VIP and have had NO WAN for HOLD consecutive seconds,
|
||||
# stand down. force-fault sheds every VIP and the peer -- which is healthy and
|
||||
# merely non-preempting -- takes them straight back.
|
||||
#
|
||||
# This is deliberately SHORTER than GRACE in vrrp-wan.conf (300s). GRACE is
|
||||
# sized for a real hostile-ISP takeover that is still making progress; this is
|
||||
# sized for "the drill failed, give the house its internet back". Anything the
|
||||
# drill proves after two and a half minutes of downtime is not worth the
|
||||
# downtime.
|
||||
#
|
||||
# wan-drill-watchdog arm [seconds] background it, then run the drill
|
||||
# wan-drill-watchdog disarm cancel it (drill succeeded)
|
||||
|
||||
STATE=/run/vrrp-wan
|
||||
VIP=192.168.1.1
|
||||
[ -r /config/vrrp-wan.conf ] && . /config/vrrp-wan.conf
|
||||
VIP="${VRRP_WAN_VIP:-$VIP}"
|
||||
HOLD="${2:-150}"
|
||||
PIDF=/run/wan-drill-watchdog.pid
|
||||
|
||||
have_vip() { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; }
|
||||
have_wan() {
|
||||
ip -4 addr show dev bond0.53 2>/dev/null | grep -q 'inet ' && return 0
|
||||
ip -4 addr show dev pppoe0 2>/dev/null | grep -q 'inet ' && return 0
|
||||
return 1
|
||||
}
|
||||
|
||||
case "${1:-arm}" in
|
||||
disarm)
|
||||
if [ -f "$PIDF" ]; then
|
||||
kill "$(cat "$PIDF")" 2>/dev/null
|
||||
rm -f "$PIDF"
|
||||
echo " watchdog disarmed"
|
||||
else
|
||||
echo " no watchdog armed"
|
||||
fi
|
||||
exit 0
|
||||
;;
|
||||
arm)
|
||||
[ -f "$PIDF" ] && kill "$(cat "$PIDF")" 2>/dev/null
|
||||
mkdir -p "$STATE" 2>/dev/null
|
||||
# setsid so it survives the ssh session that armed it going away -- the
|
||||
# whole point is that it outlives whoever started the drill.
|
||||
setsid sh -c '
|
||||
bad=0
|
||||
while :; do
|
||||
sleep 5
|
||||
if ip -4 -o addr show 2>/dev/null | grep -q " '"$VIP"'/"; then
|
||||
if ip -4 addr show dev bond0.53 2>/dev/null | grep -q "inet " ||
|
||||
ip -4 addr show dev pppoe0 2>/dev/null | grep -q "inet "; then
|
||||
bad=0
|
||||
else
|
||||
bad=$((bad + 5))
|
||||
fi
|
||||
else
|
||||
bad=0
|
||||
fi
|
||||
if [ "$bad" -ge '"$HOLD"' ]; then
|
||||
logger -t wan-drill "ABORT: held the VIP with no WAN for '"$HOLD"'s -- standing down"
|
||||
touch '"$STATE"'/force-fault
|
||||
rm -f '"$PIDF"'
|
||||
exit 0
|
||||
fi
|
||||
done
|
||||
' >/dev/null 2>&1 &
|
||||
echo $! > "$PIDF"
|
||||
echo " watchdog armed (pid $(cat "$PIDF")): stand down after ${HOLD}s holding the VIP with no WAN"
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
echo "usage: wan-drill-watchdog [arm [seconds]|disarm]" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
109
migration/wan-panic
Normal file
109
migration/wan-panic
Normal file
@@ -0,0 +1,109 @@
|
||||
#!/bin/sh
|
||||
# wan-panic -- get the internet back. No Claude, no internet, no thinking.
|
||||
#
|
||||
# Run it on EITHER router. It works out which box it is and does the right
|
||||
# thing. Safe to run twice, safe to run on both, safe to run when nothing is
|
||||
# wrong.
|
||||
#
|
||||
# sudo /config/wan-panic hand the WAN back to vyos001
|
||||
# sudo /config/wan-panic status show who has what, change nothing
|
||||
# sudo /config/wan-panic undo also stop the whole vrrp-wan mechanism
|
||||
#
|
||||
# WHY THIS EXISTS: the WAN now follows VRRP mastership. If a failover leaves
|
||||
# the wrong box holding the VIPs, or the new master cannot raise a WAN, the
|
||||
# house has no internet -- and whoever is debugging it has no internet either.
|
||||
# So the recovery path must be a single command that is already on the box.
|
||||
#
|
||||
# HOW TO REACH THE ROUTERS WITH THE NETWORK BROKEN: ssh the LoT leg,
|
||||
# ssh vyos@10.0.1.252 (vyos001)
|
||||
# ssh vyos@10.0.1.253 (vyos002)
|
||||
# It is L2-direct on bond0.10 and survives Management, VRRP and routing being
|
||||
# broken. Console via the JetKVMs is the fallback.
|
||||
|
||||
STATE=/run/vrrp-wan
|
||||
HOST="$(cat /etc/hostname 2>/dev/null || hostname)"
|
||||
VIP=192.168.1.1
|
||||
[ -r /config/vrrp-wan.conf ] && . /config/vrrp-wan.conf
|
||||
VIP="${VRRP_WAN_VIP:-$VIP}"
|
||||
|
||||
have_vip() { ip -4 -o addr show 2>/dev/null | grep -q " ${VIP}/"; }
|
||||
wan_list() {
|
||||
for i in bond0.53 pppoe0; do
|
||||
a=$(ip -4 addr show dev "$i" 2>/dev/null | sed -n 's/.*inet \([0-9.]*\).*/\1/p')
|
||||
[ -n "$a" ] && printf '%s=%s ' "$i" "$a"
|
||||
done
|
||||
}
|
||||
|
||||
show() {
|
||||
printf ' host : %s\n' "$HOST"
|
||||
printf ' holds VIP : %s\n' "$(have_vip && echo YES || echo no)"
|
||||
printf ' WAN : %s\n' "$(wan_list)"
|
||||
printf ' route : %s\n' "$(ip route show default 2>/dev/null | head -1)"
|
||||
printf ' force-fault: %s\n' "$([ -f "$STATE/force-fault" ] && echo SET || echo clear)"
|
||||
}
|
||||
|
||||
case "${1:-failback}" in
|
||||
status)
|
||||
show
|
||||
exit 0
|
||||
;;
|
||||
|
||||
undo)
|
||||
# Full stop. Leaves whatever WAN is currently up exactly as it is and
|
||||
# stops anything from moving it again. Use when the mechanism itself is
|
||||
# suspect. `vyos-known-good restore` is the heavier hammer below.
|
||||
systemctl disable --now vrrp-wan-reconcile.timer vrrp-wan-guard.timer 2>/dev/null
|
||||
rm -f "$STATE/force-fault"
|
||||
echo " vrrp-wan timers stopped. Nothing will move the WAN now."
|
||||
echo " The 10 gig / PPPoE stay exactly as they are this second."
|
||||
echo
|
||||
echo " If the config itself is wrong, the heavier hammer is:"
|
||||
echo " sudo /config/vyos-known-good restore"
|
||||
echo " on BOTH routers (it reboots them onto the pinned config)."
|
||||
echo
|
||||
show
|
||||
exit 0
|
||||
;;
|
||||
|
||||
failback)
|
||||
# The common case: put vyos001 back in charge.
|
||||
#
|
||||
# It is done with the force-fault lever, not by restarting keepalived,
|
||||
# because failing the health check is the SUPPORTED way to shed
|
||||
# mastership -- the sync group goes FAULT, releases every VIP, and the
|
||||
# peer takes over by the same path a genuine WAN loss uses. `restart
|
||||
# vrrp` is not dependable: with advert_int 1 the peer declares the master
|
||||
# dead in ~3.6s and a restart usually finishes inside that window.
|
||||
case "$HOST" in
|
||||
*002|*2)
|
||||
# This is the secondary. Stand down so vyos001 can have it back.
|
||||
mkdir -p "$STATE" 2>/dev/null
|
||||
touch "$STATE/force-fault"
|
||||
echo " $HOST: standing down (force-fault SET)."
|
||||
echo " vyos001 should take the VIPs and raise the WAN within ~30s."
|
||||
echo
|
||||
echo " When the dust settles and you WANT this box eligible again:"
|
||||
echo " sudo rm /run/vrrp-wan/force-fault"
|
||||
;;
|
||||
*)
|
||||
# This is the primary. Make sure nothing is holding it back.
|
||||
rm -f "$STATE/force-fault"
|
||||
echo " $HOST: cleared force-fault -- eligible for MASTER."
|
||||
echo " NOTE: VRRP is no-preempt, so if the peer currently holds the"
|
||||
echo " VIPs it KEEPS them. To actually take them back, run this on"
|
||||
echo " the OTHER box (vyos002 / 10.0.1.253):"
|
||||
echo " sudo /config/wan-panic"
|
||||
;;
|
||||
esac
|
||||
# Reconcile now rather than waiting up to 30s for the timer.
|
||||
[ -x /config/vrrp-wan-reconcile ] && /config/vrrp-wan-reconcile 9>&- 2>/dev/null
|
||||
echo
|
||||
show
|
||||
exit 0
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "usage: wan-panic [failback|status|undo]" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
35
migration/window-evidence/2026-09-06-baseline.txt
Normal file
35
migration/window-evidence/2026-09-06-baseline.txt
Normal file
@@ -0,0 +1,35 @@
|
||||
=== W0 baseline 2026-09-06T22:25:20+01:00 ===
|
||||
--- 10.0.1.252 ---
|
||||
vyos001
|
||||
holds VIP : YES
|
||||
WAN : bond0.53=87.192.101.48 pppoe0=90.251.152.236
|
||||
route : default via 87.192.96.1 dev bond0.53 proto failover metric 1
|
||||
internet: UP
|
||||
ipv6 : UP
|
||||
vrrp : 6 MASTER
|
||||
vlan2 v6 present already? : 0
|
||||
--- 10.0.1.253 ---
|
||||
vyos002
|
||||
holds VIP : no
|
||||
WAN :
|
||||
route :
|
||||
internet: DOWN
|
||||
ipv6 : DOWN
|
||||
vrrp : 6 BACKUP
|
||||
vlan2 v6 present already? : 0
|
||||
--- cluster ---
|
||||
aitopatom-3a1c Ready
|
||||
spark-2935 Ready
|
||||
worker0-k8s0.ad.itaz.eu Ready
|
||||
worker1-k8s0.ad.itaz.eu Ready
|
||||
worker2-k8s0.ad.itaz.eu Ready
|
||||
--- known-good save ---
|
||||
[0;36m[known-good][0m pinned 1118 lines as known-good on vyos001
|
||||
[0;36m[known-good][0m restore with: /config/vyos-known-good restore
|
||||
[0;36m[known-good][0m pinned 1119 lines as known-good on vyos002
|
||||
[0;36m[known-good][0m restore with: /config/vyos-known-good restore
|
||||
--- vrrp-wan-install --check ---
|
||||
10.0.1.252: vrrp-wan in sync
|
||||
10.0.1.253: vrrp-wan in sync
|
||||
vyos001: in sync (533 nodes)
|
||||
vyos002: in sync (534 nodes)
|
||||
97
migration/window-evidence/2026-09-06-dhcpv6.txt
Normal file
97
migration/window-evidence/2026-09-06-dhcpv6.txt
Normal file
@@ -0,0 +1,97 @@
|
||||
=== W2: does a production node take a DHCPv6 reservation? ===
|
||||
Window of 2026-09-06, operator offline.
|
||||
|
||||
SHORT ANSWER: YES for the x86_64 Fedora nodes -- MAC reservations work. My first
|
||||
write-up of this file said "MAC reservations do NOT match" and that was WRONG:
|
||||
I read the result before a full RA/DHCPv6 cycle had completed. The timed
|
||||
observation I set running (660s, one MaxRtrAdvInterval) then showed the opposite,
|
||||
and a direct check confirmed it. Recording the mistake because committing the
|
||||
premature version to git (3c933b9) is exactly the "assert before measuring"
|
||||
failure this session has been about.
|
||||
|
||||
CONFIRMED, direct check 2026-09-06 ~22:47:
|
||||
worker0-k8s0 192.168.8.23 -> 2001:470:187e:2::23/128 MATCHES reservation
|
||||
worker2-k8s0 192.168.8.25 -> 2001:470:187e:2::25/128 MATCHES reservation
|
||||
Both are the EXACT reserved addresses, /128, DHCPv6-assigned. The MAC-keyed
|
||||
scheme -- one source of truth with IPv4 -- works for these nodes.
|
||||
|
||||
WHY IT WORKS DESPITE "[no hwaddr info]" IN THE KEA LOG
|
||||
The DHCP6_QUERY_LABEL "[no hwaddr info]" is about whether the client sent an
|
||||
explicit hardware-address option; it is NOT the reservation-matching path. For a
|
||||
`hw-address` host reservation kea extracts the MAC from the client's DUID when
|
||||
the DUID type carries one (DUID-LLT / DUID-LL embed the link-layer address).
|
||||
NetworkManager on the Fedora nodes uses such a DUID, so kea matched ::23 and ::25
|
||||
by MAC even though the query label showed no explicit hwaddr. The earlier
|
||||
DUID-UUID packets I saw were from other clients, and led me to over-generalise.
|
||||
|
||||
ROOT CAUSE OF THE TWO NON-BINDING NODES (proven, and it is NOT architecture)
|
||||
Corrects an earlier claim in git that called this an "arm64" / "per-node client"
|
||||
issue. worker2 is aarch64 and bound fine, so arch was a coincidence. The real
|
||||
chain, proven by the kea ALLOC_ENGINE log:
|
||||
|
||||
1. The VLAN 2 subnet is RESERVATIONS-ONLY -- no dynamic pool (confirmed in the
|
||||
rendered kea6 config: pools NONE, 5 reservations).
|
||||
2. Every node sends DUID-UUID (type 00:04), which carries no MAC -- hence
|
||||
"[no hwaddr info]" on every packet. kea therefore cannot match the
|
||||
hw-address reservation from the DUID.
|
||||
3. kea falls back to deriving the MAC from the packet's SOURCE LINK-LOCAL,
|
||||
which only works when that address is EUI-64 (embeds the MAC).
|
||||
4. worker0/worker2 have ipv6.addr-gen-mode=eui64, so their link-local is
|
||||
EUI-64 (fe80::7a55:36ff:fe08:28fb embeds 78:55:36:08:28:fb) -> MAC derived
|
||||
-> reservation matches -> they get ::23 / ::25.
|
||||
5. worker1 (end0) and spark (enP7s7) use addr-gen-mode=default =
|
||||
STABLE-PRIVACY (RFC 7217): fe80::ed79:7863:8c2c:1d4c embeds no MAC -> kea
|
||||
derives nothing -> no reservation match -> and with no dynamic pool there
|
||||
is nothing else to hand out:
|
||||
|
||||
ALLOC_ENGINE_V6_ALLOC_FAIL_SHARED_NETWORK: 1 subnets have no matching pools
|
||||
ALLOC_ENGINE_V6_ALLOC_FAIL_NO_POOLS: no pools were available
|
||||
|
||||
So the MAC-reservation scheme is really a LINK-LOCAL-EUI-64 scheme in disguise.
|
||||
It works only where every node uses EUI-64 link-locals, which is NOT the modern
|
||||
NetworkManager default.
|
||||
|
||||
FIX OPTIONS (attended decision)
|
||||
a) Enforce ipv6.addr-gen-mode=eui64 on the cluster NICs fleet-wide and at
|
||||
provision time. Smallest change, keeps one-source-of-truth-with-IPv4, and
|
||||
is a node setting we already control via NM/labctl. Cost: EUI-64 leaks the
|
||||
MAC into the address (irrelevant for infra nodes) and it must be enforced or
|
||||
a future node silently fails to bind -- the exact trap that produced this.
|
||||
b) Key reservations on DUID instead. Robust to link-local mode, but a DUID is
|
||||
client-generated, a second source of truth, and changes on reinstall.
|
||||
c) Dynamic pool + labctl discovery (node-ip refuses an absent address already,
|
||||
914135c), giving up the "address knowable before boot" property.
|
||||
Not (d) a dynamic pool ALONGSIDE reservations: an unmatched node would then get
|
||||
SOME address, not its reserved one, so node-ip becomes unpredictable -- worse
|
||||
than failing loudly.
|
||||
|
||||
RA HALF (unchanged, and it was always the safe finding)
|
||||
RA on bond0.2 carries AdvManagedFlag on, AdvAutonomous off, AdvLinkMTU 1472,
|
||||
AdvDefaultLifetime 0. NetworkManager (ipv6.method=auto) follows the managed flag
|
||||
and starts DHCPv6. "ipv6.method=auto will do DHCPv6" is evidence, not inference.
|
||||
|
||||
A SECOND FINDING, unrelated, found live and FIXED in-window
|
||||
`service dhcpv6-server` with no listen-interface renders kea6 with
|
||||
interfaces: [ "*" ] -- a DHCPv6 server on EVERY VLAN. Observed answering an
|
||||
unrelated device on bond0.10 (LoT) within seconds of the first apply. Same
|
||||
family as the kea IPv4 cross-VLAN bug (ISC #1117). Fixed by pinning
|
||||
listen-interface bond0.2 + a subnet-level interface bond0.2; both routers now
|
||||
render interfaces: ["bond0.2"].
|
||||
|
||||
OPTION (b) mac-sources: VyOS dhcpv6-server global-parameters accepts only
|
||||
name-server, so kea mac-sources cannot be passed through config, and editing
|
||||
/run/kea/*.conf is banned drift (regenerated every commit). Moot now that the
|
||||
default matching works for the Fedora nodes.
|
||||
|
||||
STILL OPEN FOR THE ATTENDED SESSION
|
||||
- the two arm64 nodes: why the DHCPv6 transaction does not complete.
|
||||
- whether to keep MAC keys (works for x86 Fedora, unproven for arm) or move to
|
||||
DUID keys / a dynamic range + labctl discovery for uniformity.
|
||||
|
||||
=== CLOSED 2026-09-06: all five nodes bound, aitopatom done ===
|
||||
addr-gen-mode=eui64 rolled out fleet-wide + baked into provisioning (6c94371).
|
||||
aitopatom-3a1c reached via michal@ (root@ refused key), took ::27 immediately on
|
||||
the flip. Final: worker0 ::23, worker2 ::25, worker1 ::13, spark ::12,
|
||||
aitopatom ::27 -- all match their reservations, cluster 5/5 Ready.
|
||||
The keying scheme (MAC + enforced EUI-64) is proven and enforced; remaining IPv6
|
||||
work is egress + the cluster conversion, both attended.
|
||||
54
migration/window-evidence/2026-09-06-final.txt
Normal file
54
migration/window-evidence/2026-09-06-final.txt
Normal file
@@ -0,0 +1,54 @@
|
||||
=== MAINTENANCE WINDOW 2026-09-06 (unattended) — HANDOFF ===
|
||||
|
||||
WHAT HAPPENED
|
||||
VLAN 2 (the k8s VLAN) now has IPv6, applied to both routers and pinned in the
|
||||
Pulumi model. Addressing only, NOT egress (default-lifetime 0 -- routers are
|
||||
not v6 default routers yet). IPv4 untouched the whole time; a watchdog was
|
||||
armed on both routers and never reverted (0 reverts).
|
||||
|
||||
The window's open question is ANSWERED: production NetworkManager nodes take a
|
||||
DHCPv6 lease from the managed flag, and MAC-keyed reservations match. worker0
|
||||
and worker2 hold their exact reserved ::23 / ::25.
|
||||
|
||||
WHAT DID NOT WORK (for you to decide, attended)
|
||||
The two arm64 nodes -- worker1 (Asahi) and spark-2935 (DGX) -- ran a DHCPv6
|
||||
transaction but did not bind an address. Per-node client issue, not the
|
||||
reservation scheme (two x86 nodes prove the scheme). Untouched, deliberately.
|
||||
|
||||
A mistake worth knowing about: I first committed "MAC reservations do NOT work"
|
||||
after checking before a full RA cycle. My own timed observation caught it and
|
||||
it is corrected in git (lab@d9f74aa, k8s-deployment@75af36d) and in
|
||||
2026-09-06-dhcpv6.txt. The scheme works.
|
||||
|
||||
WHAT IS NEXT
|
||||
- decide the two arm64 nodes' DHCPv6 (why they don't bind), and whether to keep
|
||||
MAC keys or move to DUID / dynamic+discovery for uniformity.
|
||||
- THEN egress: flip default-lifetime 0 -> 1800 + a default-preference pair,
|
||||
WITH the tunnel throughput measured (1480/1472). One line, attended.
|
||||
- the cluster steps (Cilium IPAM switch, k3s dual CIDRs, Cilium IPv6) were
|
||||
deliberately NOT started -- they need someone watching and the 3-server etcd
|
||||
rehearsal harness still to be built.
|
||||
|
||||
Nothing here is load-bearing yet: no node depends on its v6 address, so this is
|
||||
fully reversible with migration/vlan2-v6-apply revert.
|
||||
|
||||
=== FINAL STATE 2026-09-06T22:51:14+01:00 ===
|
||||
--- 10.0.1.252 ---
|
||||
vyos001
|
||||
internet/ipv6: UP/UP
|
||||
vrrp: 6 MASTER
|
||||
route: default via 87.192.96.1 dev bond0.53 proto failover metric 1
|
||||
bond0.2 v6: 2001:470:187e:2::1/64
|
||||
--- 10.0.1.253 ---
|
||||
vyos002
|
||||
internet/ipv6: DOWN/DOWN
|
||||
vrrp: 6 BACKUP
|
||||
route:
|
||||
bond0.2 v6: 2001:470:187e:2::2/64
|
||||
--- nodes v6 ---
|
||||
192.168.8.23 2001:470:187e:2::23
|
||||
192.168.8.13
|
||||
192.168.8.25 2001:470:187e:2::25
|
||||
192.168.8.12
|
||||
--- cluster ---
|
||||
5 Ready
|
||||
Reference in New Issue
Block a user