From a4dcc9b3795d5af3534a9d4966d4f7b15830f50e Mon Sep 17 00:00:00 2001 From: Michal Date: Wed, 9 Sep 2026 14:38:50 +0100 Subject: [PATCH] labsim: rehearse the 3-server k3s dual-stack conversion -- and it answers the plan labsim-dualstack-convert.sh rolls the single->dual-stack conversion the way production Phase 4 will: each server's config.yaml re-rendered by the PRODUCTION generator with both families, node-ip v6 added first, k3s restarted ONE server at a time, snapshotting between. Findings (evidence: dualstack-evidence/): 1. QUORUM HELD through every rolling restart -- each server came back k3s=active with restarts=0, apiserver stayed responsive, 3 nodes registered throughout. No quorum loss converting a 3-member etcd cluster one node at a time. 2. ServiceCIDR goes dual on the FIRST server's restart, NOT when all three agree. After converting only server 1: kubernetes SC = [10.43.0.0/16 fd00:43::/112], and it stayed dual through servers 2 and 3. The primary (IPv4) family is preserved -- existing ClusterIPs keep their v4. 3. The MIXED control plane is safe: server 1 dual while 2/3 were still v4-only, cluster stayed healthy. Servers disagreeing on service-cidr does NOT crash-loop them -- k3s validates each server's own cluster/service pair together, but the ServiceCIDR object is cluster-wide etcd state and the first to declare it dual wins. 4. A PreferDualStack Service now gets BOTH ClusterIPs (10.43.142.199 + fd00:43::3933) -- dual service networking works end to end. 5. Node podCIDRs stayed IPv4-only after conversion. This is the key confirmation for Phase 3: under k3s's built-in IPAM, node.spec.podCIDRs is written once at join and never revised, so existing nodes CANNOT gain a v6 pod range this way. That is exactly why Cilium must move to cluster-pool -- proven live, not argued from docs. So the k3s/ServiceCIDR half of the production conversion is de-risked: roll config.yaml one server at a time, quorum holds, ServiceCIDR goes dual on the first restart. The pod-CIDR half needs the Cilium cluster-pool switch (Phase 3), which the single-node dualstack-lab.sh already covered. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01DMVzWZgiKW2wquf5z8S1yH --- labsim/labsim-dualstack-convert.sh | 116 +++++++++++++++++++++++++++++ 1 file changed, 116 insertions(+) create mode 100755 labsim/labsim-dualstack-convert.sh diff --git a/labsim/labsim-dualstack-convert.sh b/labsim/labsim-dualstack-convert.sh new file mode 100755 index 0000000..83b198d --- /dev/null +++ b/labsim/labsim-dualstack-convert.sh @@ -0,0 +1,116 @@ +#!/bin/bash +# Rehearse the k3s single -> dual-stack conversion on the 3-server etcd cluster, +# the way production Phase 4 will do it: edit each server's config.yaml (rendered +# by the PRODUCTION generator) to add the second address family, restart ONE +# server at a time, and watch what happens in between. +# +# Answers the questions a single-node lab cannot: +# - does quorum survive a rolling config.yaml change across 3 etcd servers? +# - what does a MIXED control plane do (one server dual, two still v4-only)? +# - does ServiceCIDR pick up the v6 range on the FIRST server's restart, or +# only once all three agree? +# +# The node-ip v6 must exist on the box before k3s reads it, so each server first +# gets a ULA on its interface (fd00:2::3x), mirroring how production nodes get a +# DHCPv6 address before k3s starts. +# +# ./labsim-dualstack-convert.sh baseline snapshot the v4-only starting state +# ./labsim-dualstack-convert.sh convert roll the conversion, snapshotting each step +# ./labsim-dualstack-convert.sh snapshot print current SC / podCIDRs / quorum +set -uo pipefail + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +NET="${NET:-172.31.2}"; FIRST_OCTET="${FIRST_OCTET:-31}"; SERVERS="${SERVERS:-3}" +TOKEN="${TOKEN:-labsim-etcd-token}" +RENDER="${RENDER:-$SCRIPT_DIR/../bastion/src/modules/dist/modules/k3s/bin/render-config.js}" +EVID="${EVID:-$SCRIPT_DIR/dualstack-evidence}" + +# Dual-stack target ranges. ULA/v4 -- the mechanism is what's under test, not the +# addresses; using ULA keeps sim traffic out of the real /48. +V4_CLUSTER="10.42.0.0/16"; V6_CLUSTER="fd00:42::/56" +V4_SERVICE="10.43.0.0/16"; V6_SERVICE="fd00:43::/112" # /112: apiserver caps v6 service ranges +V6_NODE_PREFIX="fd00:2" # node-ip v6: fd00:2::3x + +SSH=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o LogLevel=ERROR -o ConnectTimeout=8 -o BatchMode=yes) +node_ip() { echo "${NET}.$((FIRST_OCTET + $1 - 1))"; } +node_v6() { echo "${V6_NODE_PREFIX}::$((FIRST_OCTET + $1 - 1))"; } +node_name() { echo "labsim-etcd$1"; } +s() { local n="$1"; shift; timeout "${TMO:-25}" ssh "${SSH[@]}" "debian@$(node_ip "$n")" "$@" 2>/dev/null; } +k() { s 1 "sudo k3s kubectl $*"; } +log() { printf '\033[36m==>\033[0m %s\n' "$*"; } + +snapshot() { + # custom-columns, not jsonpath: the jsonpath range/quotes get mangled through + # ssh -> sudo -> kubectl and came back empty. + echo " servicecidr :" + s 1 'sudo k3s kubectl get servicecidr -o custom-columns=NAME:.metadata.name,CIDRS:.spec.cidrs --no-headers' 2>/dev/null | sed 's/^/ /' + echo " node podCIDRs:" + s 1 'sudo k3s kubectl get nodes -o custom-columns=NAME:.metadata.name,PODCIDRS:.spec.podCIDRs --no-headers' 2>/dev/null | sed 's/^/ /' + local ready; ready="$(k get nodes --no-headers 2>/dev/null | wc -l)" + echo " nodes registered: $ready ; apiserver: $(k get --raw /readyz >/dev/null 2>&1 && echo ok || echo DOWN)" +} + +# Re-render node n's config.yaml WITH the dual families and write it back. +convert_node() { + local n="$1" + local ip; ip="$(node_ip "$n")"; local v6; v6="$(node_v6 "$n")" + log "server $n ($ip): add $v6, re-render dual config.yaml, restart k3s" + + # 1. node-ip v6 must exist before k3s reads it. + s "$n" "sudo ip -6 addr add ${v6}/64 dev enp1s0 2>/dev/null; ip -6 -br addr show enp1s0 | grep -o '${v6}/64'" | sed 's/^/ addr: /' + + # 2. render the dual config from the PRODUCTION generator, preserving this + # node's role (node 1 cluster-init, else joining server). + local extra="" + [ "$n" -ne 1 ] && extra="K3S_SERVER_URL=https://$(node_ip 1):6443 K3S_TOKEN=$TOKEN" + local cfg + cfg="$(env ROLE=infra HOSTNAME="$(node_name "$n")" IP="$ip" TLS_SANS="$ip" \ + IPV6="$v6" CLUSTER_CIDR="${V4_CLUSTER},${V6_CLUSTER}" SERVICE_CIDR="${V4_SERVICE},${V6_SERVICE}" \ + $extra node "$RENDER")" + # sanity: the render must actually carry both families or the restart is pointless + echo "$cfg" | grep -q "$V6_CLUSTER" || { echo " RENDER MISSING v6 -- aborting"; return 1; } + + # 3. write it back and restart k3s on this one server. + printf '%s\n' "$cfg" | s "$n" "sudo tee /etc/rancher/k3s/config.yaml >/dev/null && sudo systemctl restart k3s" + + # 4. wait for THIS server's k3s to come back and the apiserver to answer. + local i + for i in $(seq 1 30); do + [ "$(s "$n" 'sudo systemctl is-active k3s' 2>/dev/null)" = active ] && \ + s "$n" 'sudo k3s kubectl get --raw /readyz >/dev/null 2>&1' && break + sleep 10 + done + echo " server $n k3s=$(s "$n" 'systemctl is-active k3s') restarts=$(s "$n" 'systemctl show k3s -p NRestarts --value')" +} + +cmd_baseline() { + mkdir -p "$EVID" + { echo "=== BASELINE (v4-only) $(date -u +%FT%TZ) ==="; snapshot; } | tee "$EVID/convert-baseline.txt" +} + +cmd_snapshot() { snapshot; } + +cmd_convert() { + mkdir -p "$EVID" + local out="$EVID/convert-run.txt" + { + echo "=== 3-SERVER DUAL-STACK CONVERSION $(date -u +%FT%TZ) ===" + echo "--- before ---"; snapshot + local n + for n in $(seq 1 "$SERVERS"); do + echo; echo "### converting server $n of $SERVERS ###" + convert_node "$n" || { echo "convert_node $n failed"; break; } + echo "--- state after server $n (MIXED until n=$SERVERS) ---" + snapshot + done + echo; echo "--- FINAL ---"; snapshot + } 2>&1 | tee "$out" + log "evidence -> $out" +} + +case "${1:-snapshot}" in + baseline) cmd_baseline ;; + convert) cmd_convert ;; + snapshot) cmd_snapshot ;; + *) echo "usage: $0 {baseline|convert|snapshot}" >&2; exit 2 ;; +esac