labsim: 3-server embedded-etcd k3s harness, driven by the production generator
Some checks failed
Some checks failed
The rehearsal vehicle for cluster dual-stack. Unlike k8s-up.sh (1 server + 2 agents, inline INSTALL_K3S_EXEC), this builds 3 control-plane servers with embedded etcd, each configured through /etc/rancher/k3s/config.yaml rendered by labctl's own generate*Config via bin/render-config.js -- the exact production code path. Node 1 cluster-inits; 2/3 join as etcd members (role=infra + server URL, same shape as production worker1/worker2). Carries the audit-policy.yaml the generated config references (byte-identical to audit-policy.ts) -- without it the apiserver silently crash-loops. Validates the embedded config.yaml as YAML before building each seed ISO. Next: rehearse the dual-stack conversion on it (rolling config.yaml change, quorum, mixed control plane, ServiceCIDR pickup). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01DMVzWZgiKW2wquf5z8S1yH
This commit is contained in:
257
labsim/labsim-k8s-etcd.sh
Executable file
257
labsim/labsim-k8s-etcd.sh
Executable file
@@ -0,0 +1,257 @@
|
||||
#!/bin/bash
|
||||
# A 3-SERVER embedded-etcd k3s cluster in labsim, configured the way PRODUCTION
|
||||
# is -- through /etc/rancher/k3s/config.yaml rendered by labctl's own generator.
|
||||
#
|
||||
# WHY THIS EXISTS, separate from k8s-up.sh. k8s-up.sh is 1 server + 2 agents and
|
||||
# drives k3s with inline INSTALL_K3S_EXEC flags. Production is 3 control-plane
|
||||
# servers with embedded etcd, configured by config.yaml from k3s-config.ts. The
|
||||
# dual-stack conversion touches etcd quorum on all three at once and rolls a
|
||||
# config.yaml change one server at a time -- a failure mode a single-server lab
|
||||
# structurally cannot show, driven by a mechanism k8s-up.sh does not use. So this
|
||||
# harness models the real shape and the real code path, or the rehearsal is
|
||||
# theatre.
|
||||
#
|
||||
# ./labsim-k8s-etcd.sh up build the 3-server cluster
|
||||
# ./labsim-k8s-etcd.sh kubeconfig fetch ./labsim-etcd.kubeconfig
|
||||
# ./labsim-k8s-etcd.sh status node + etcd health
|
||||
# ./labsim-k8s-etcd.sh down destroy the three VMs
|
||||
# ./labsim-k8s-etcd.sh render N print the config.yaml node N would get
|
||||
#
|
||||
# The config.yaml is produced by:
|
||||
# bastion .../k3s/bin/render-config.js (the production generator, exported)
|
||||
# so if that generator changes shape, this cluster moves with it.
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
source "$SCRIPT_DIR/lib.sh"
|
||||
source "$SCRIPT_DIR/ovs.sh"
|
||||
|
||||
K8S_VLAN="${K8S_VLAN:-2}"
|
||||
NET="${NET:-172.31.2}"
|
||||
FIRST_OCTET="${FIRST_OCTET:-31}" # .31/.32/.33 -- clear of k8s-up.sh's .11-.13
|
||||
SERVERS="${SERVERS:-3}"
|
||||
MEM="${MEM:-4096}"; CPUS="${CPUS:-2}"; DISK_GB="${DISK_GB:-12}"
|
||||
TOKEN="${TOKEN:-labsim-etcd-token}"
|
||||
CILIUM_VERSION="${CILIUM_VERSION:-1.19.1}"
|
||||
|
||||
DEB_URL="${DEB_URL:-https://cloud.debian.org/images/cloud/trixie/latest/debian-13-genericcloud-amd64.qcow2}"
|
||||
DEB_BASE="${DEB_BASE:-$IMG_DIR/debian-13-genericcloud-amd64.qcow2}"
|
||||
|
||||
# The production generator, compiled. Built by `npm --prefix bastion/src/modules run build`.
|
||||
RENDER="${RENDER:-$SCRIPT_DIR/../bastion/src/modules/dist/modules/k3s/bin/render-config.js}"
|
||||
|
||||
node_name() { echo "labsim-etcd$1"; }
|
||||
node_ip() { echo "${NET}.$((FIRST_OCTET + $1 - 1))"; }
|
||||
|
||||
# The audit policy the generated config.yaml references. Without the file the
|
||||
# apiserver refuses to start (audit-policy-file points at a missing path), which
|
||||
# is a silent-looking crash loop. Kept byte-identical to labctl's audit-policy.ts.
|
||||
AUDIT_POLICY='apiVersion: audit.k8s.io/v1
|
||||
kind: Policy
|
||||
rules:
|
||||
- level: Metadata
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["secrets", "configmaps"]
|
||||
- level: RequestResponse
|
||||
verbs: ["create", "update", "patch", "delete"]
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["pods", "services", "deployments"]
|
||||
- level: None
|
||||
resources:
|
||||
- group: ""
|
||||
resources: ["endpoints", "events"]
|
||||
users: ["system:kube-proxy", "system:apiserver"]
|
||||
- level: Metadata
|
||||
omitStages:
|
||||
- "RequestReceived"'
|
||||
|
||||
# Render node N's config.yaml with the production generator. Node 1 is
|
||||
# cluster-init; 2..N join as SERVERS (not agents) -- role=infra with a server
|
||||
# URL is exactly a joining etcd member, the same as production worker1/worker2.
|
||||
render_config() {
|
||||
local n="$1"
|
||||
local ip; ip="$(node_ip "$n")"
|
||||
local server1; server1="$(node_ip 1)"
|
||||
if [ "$n" -eq 1 ]; then
|
||||
ROLE=infra HOSTNAME="$(node_name 1)" IP="$ip" TLS_SANS="$ip" \
|
||||
node "$RENDER"
|
||||
else
|
||||
ROLE=infra HOSTNAME="$(node_name "$n")" IP="$ip" TLS_SANS="$ip" \
|
||||
K3S_SERVER_URL="https://${server1}:6443" K3S_TOKEN="$TOKEN" \
|
||||
node "$RENDER"
|
||||
fi
|
||||
}
|
||||
|
||||
ensure_base_image() {
|
||||
[ -f "$DEB_BASE" ] && { log "base image present"; return; }
|
||||
log "fetching Debian cloud image -> $DEB_BASE"
|
||||
sudo mkdir -p "$IMG_DIR"
|
||||
sudo curl -fsSL --retry 3 -o "${DEB_BASE}.tmp" "$DEB_URL" || die "fetch failed"
|
||||
sudo mv "${DEB_BASE}.tmp" "$DEB_BASE"
|
||||
}
|
||||
|
||||
build_seed() {
|
||||
local iso="$1" n="$2" pubkey="$3"
|
||||
local vm; vm="$(node_name "$n")"; local ip; ip="$(node_ip "$n")"
|
||||
local server1; server1="$(node_ip 1)"
|
||||
local tmp; tmp="$(mktemp -d)"
|
||||
local cfg; cfg="$(render_config "$n")"
|
||||
|
||||
cat > "$tmp/meta-data" <<EOF
|
||||
instance-id: $vm
|
||||
local-hostname: $vm
|
||||
EOF
|
||||
cat > "$tmp/network-config" <<EOF
|
||||
version: 2
|
||||
ethernets:
|
||||
enp1s0:
|
||||
match: { name: "en*" }
|
||||
addresses: [$ip/24]
|
||||
routes: [{ to: default, via: ${NET}.1 }]
|
||||
nameservers: { addresses: [8.8.8.8, 1.1.1.1] }
|
||||
EOF
|
||||
|
||||
# config.yaml and audit policy embedded via write_files, indented for YAML.
|
||||
local cfg_ind audit_ind
|
||||
cfg_ind="$(printf '%s\n' "$cfg" | sed 's/^/ /')"
|
||||
audit_ind="$(printf '%s\n' "$AUDIT_POLICY" | sed 's/^/ /')"
|
||||
|
||||
cat > "$tmp/user-data" <<EOF
|
||||
#cloud-config
|
||||
hostname: $vm
|
||||
users:
|
||||
- name: debian
|
||||
groups: [sudo]
|
||||
shell: /bin/bash
|
||||
sudo: ["ALL=(ALL) NOPASSWD:ALL"]
|
||||
lock_passwd: false
|
||||
plain_text_passwd: labsim
|
||||
ssh_authorized_keys: [ $pubkey ]
|
||||
ssh_pwauth: true
|
||||
ssh_authorized_keys: [ $pubkey ]
|
||||
|
||||
package_update: true
|
||||
packages: [curl, jq, iproute2, tcpdump, etcd-client]
|
||||
|
||||
write_files:
|
||||
- path: /etc/rancher/k3s/config.yaml
|
||||
content: |
|
||||
$cfg_ind
|
||||
- path: /etc/rancher/k3s/audit-policy.yaml
|
||||
content: |
|
||||
$audit_ind
|
||||
- path: /etc/modules-load.d/cilium.conf
|
||||
content: |
|
||||
br_netfilter
|
||||
overlay
|
||||
- path: /etc/sysctl.d/99-k8s.conf
|
||||
content: |
|
||||
net.ipv4.ip_forward = 1
|
||||
net.ipv6.conf.all.forwarding = 1
|
||||
net.bridge.bridge-nf-call-iptables = 1
|
||||
net.bridge.bridge-nf-call-ip6tables = 1
|
||||
|
||||
runcmd:
|
||||
- [ modprobe, br_netfilter ]
|
||||
- [ modprobe, overlay ]
|
||||
- [ sysctl, --system ]
|
||||
- |
|
||||
# Joining servers must wait for the cluster-init server's API, or the join
|
||||
# races etcd bootstrap and the unit backs off for minutes.
|
||||
if [ "$n" -ne 1 ]; then
|
||||
for i in \$(seq 1 90); do
|
||||
curl -sk --max-time 3 https://${server1}:6443/ping >/dev/null 2>&1 && break
|
||||
sleep 5
|
||||
done
|
||||
fi
|
||||
- |
|
||||
# INSTALL_K3S_EXEC=server (bare) -- everything else comes from config.yaml,
|
||||
# exactly as production. The token is passed via env for the join; on node 1
|
||||
# it seeds the cluster token.
|
||||
curl -sfL https://get.k3s.io | INSTALL_K3S_EXEC="server" K3S_TOKEN="$TOKEN" sh -
|
||||
EOF
|
||||
|
||||
# Guard the generated YAML before building the ISO -- a bad indent in the
|
||||
# embedded config.yaml would fail on the node, minutes later and opaquely.
|
||||
python3 -c "import yaml,sys; yaml.safe_load(open(sys.argv[1]))" "$tmp/user-data" \
|
||||
|| die "generated user-data is not valid YAML for node $n"
|
||||
|
||||
sudo mkdir -p "$(dirname "$iso")"
|
||||
sudo genisoimage -quiet -output "$iso" -volid cidata -joliet -rock \
|
||||
"$tmp/user-data" "$tmp/meta-data" "$tmp/network-config"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
create_node() {
|
||||
local n="$1" pubkey="$2"
|
||||
local vm; vm="$(node_name "$n")"; local ip; ip="$(node_ip "$n")"
|
||||
if virsh_q dominfo "$vm" >/dev/null 2>&1; then
|
||||
local st; st="$(virsh_q domstate "$vm" 2>/dev/null | head -1 | tr -d '\n')"
|
||||
[ "$st" = running ] && { log "$vm already running ($ip)"; return; }
|
||||
log "$vm is $st -- starting"; virsh_q start "$vm" >/dev/null; return
|
||||
fi
|
||||
local disk="$IMG_DIR/${vm}.qcow2" seed="$IMG_DIR/${vm}-seed.iso"
|
||||
log "creating $vm ($ip, server ${n}, ${MEM}MB/${CPUS}cpu)"
|
||||
sudo qemu-img create -q -f qcow2 -F qcow2 -b "$DEB_BASE" "$disk" "${DISK_GB}G" >/dev/null
|
||||
build_seed "$seed" "$n" "$pubkey"
|
||||
sudo virt-install --connect "$LIBVIRT_URI" --name "$vm" \
|
||||
--memory "$MEM" --vcpus "$CPUS" \
|
||||
--disk "path=$disk,format=qcow2,bus=virtio" \
|
||||
--disk "path=$seed,device=cdrom" \
|
||||
--network "network=$OVS_NET,portgroup=vlan${K8S_VLAN},model=virtio" \
|
||||
--os-variant debian12 --graphics none --noautoconsole --import >/dev/null
|
||||
}
|
||||
|
||||
ssh_node() { ssh -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
|
||||
-o LogLevel=ERROR -o ConnectTimeout=8 "debian@$1" "$2" 2>/dev/null; }
|
||||
|
||||
cmd_up() {
|
||||
require_tools
|
||||
command -v genisoimage >/dev/null || die "genisoimage missing"
|
||||
[ -f "$RENDER" ] || die "render CLI not built: $RENDER (run: npm --prefix bastion/src/modules run build)"
|
||||
local pubkey; pubkey="$(find_ssh_pubkey)"
|
||||
ensure_base_image
|
||||
selected_vlans; log "ensuring OVS fabric"; ovs_up
|
||||
local n; for n in $(seq 1 "$SERVERS"); do create_node "$n" "$pubkey"; done
|
||||
echo
|
||||
log "3 servers booting. Node 1 cluster-inits; 2/3 join as etcd members."
|
||||
log "watch: ssh debian@$(node_ip 1) 'sudo k3s kubectl get nodes'"
|
||||
log "then: $0 kubeconfig && $0 status"
|
||||
}
|
||||
|
||||
cmd_kubeconfig() {
|
||||
local s1; s1="$(node_ip 1)"; local out="$SCRIPT_DIR/labsim-etcd.kubeconfig"
|
||||
ssh_node "$s1" "sudo cat /etc/rancher/k3s/k3s.yaml" | sed "s|127.0.0.1|$s1|" > "$out"
|
||||
chmod 600 "$out"; log "wrote $out"
|
||||
}
|
||||
|
||||
cmd_status() {
|
||||
local s1; s1="$(node_ip 1)"
|
||||
echo "=== nodes ==="
|
||||
ssh_node "$s1" "sudo k3s kubectl get nodes -o wide 2>/dev/null" | sed 's/^/ /'
|
||||
echo "=== etcd members (quorum needs 2 of 3) ==="
|
||||
ssh_node "$s1" 'sudo k3s kubectl get nodes -l node-role.kubernetes.io/etcd=true --no-headers 2>/dev/null | wc -l' | sed 's/^/ etcd nodes: /'
|
||||
echo "=== servicecidr (both families once dual-stack) ==="
|
||||
ssh_node "$s1" "sudo k3s kubectl get servicecidr -o jsonpath='{range .items[*]}{.metadata.name}={.spec.cidrs}{\"\\n\"}{end}' 2>/dev/null" | sed 's/^/ /'
|
||||
}
|
||||
|
||||
cmd_down() {
|
||||
local n vm
|
||||
for n in $(seq 1 "$SERVERS"); do
|
||||
vm="$(node_name "$n")"
|
||||
virsh_q destroy "$vm" >/dev/null 2>&1 || true
|
||||
virsh_q undefine "$vm" --remove-all-storage >/dev/null 2>&1 || true
|
||||
log "removed $vm"
|
||||
done
|
||||
}
|
||||
|
||||
case "${1:-up}" in
|
||||
up) cmd_up ;;
|
||||
kubeconfig) cmd_kubeconfig ;;
|
||||
status) cmd_status ;;
|
||||
down) cmd_down ;;
|
||||
render) render_config "${2:-1}" ;;
|
||||
*) die "usage: $0 {up|kubeconfig|status|down|render N}" ;;
|
||||
esac
|
||||
Reference in New Issue
Block a user