feat(k8s): deploy 3-node k3s HA control plane over Tailscale

Bootstrap a regular (non-ITAR) k3s cluster on cnode1/2/3 with embedded
etcd. All cluster communication — node registration, API server, etcd
peering, flannel VXLAN — runs exclusively over Tailscale IPs. Zero LAN
addresses appear in node status or TLS certificates.

Scripts (k8s/):
- env.sh: shared config (Tailscale IPs, SSH opts, k3s version)
- wipe.sh: remove existing k3s from all cnodes
- install-cp.sh: full bootstrap (cnode1 --cluster-init, then cnode2/3 join)
- join-servers.sh: re-join cnode2/3 only (fixes broken join state)
- post-setup.sh: apply NoSchedule taints, fetch kubeconfig, verify
- verify.sh: 13-point health check (nodes, Tailscale IPs, taints, etcd,
  CoreDNS, API server, workload isolation)
- probe-nodes.sh: SSH + Tailscale reachability check

All 3 cnodes are tainted control-plane:NoSchedule so no user workloads
can schedule on the control plane. 13/13 health checks pass.

Docs updated: k8s README TL;DR reflects k3s (not Talos) as the deployed
choice, with Talos preserved for the future ITAR cluster.

💘 Generated with Crush

Assisted-by: Crush:glm-5.2
This commit is contained in:
2026-07-28 12:21:33 -05:00
parent 174deea2a9
commit 8f1642bf96
13 changed files with 693 additions and 26 deletions
+121
View File
@@ -0,0 +1,121 @@
#!/usr/bin/bash
#
# k8s/join-servers.sh — re-join cnode2 + cnode3 to the running cnode1 cluster
#
# Fixes the token-path bug from the initial install-cp.sh run. cnode1 is
# already healthy; this script only touches cnode2 and cnode3:
# 1. Uninstalls the broken k3s on each
# 2. Fetches the CORRECT token from cnode1
# 3. Re-installs both as HA server nodes joining the etcd cluster
#
set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
source "$SCRIPT_DIR/env.sh"
# Build the shared TLS-SAN flags
tls_san_flags=""
for san in "${TLS_SANS[@]}"; do
tls_san_flags+=" --tls-san=$san"
done
SERVER_URL="https://${BOOTSTRAP_IP}:${K3S_API_PORT}"
echo "============================================"
echo " Re-joining cnode2 + cnode3 to cluster"
echo " Bootstrap server: $BOOTSTRAP_NAME ($BOOTSTRAP_IP)"
echo "============================================"
# -------------------------------------------------------
# 1. Fetch the CORRECT token from cnode1
# -------------------------------------------------------
echo ""
echo "--- [1/4] Fetching join token from cnode1 ---"
JOIN_TOKEN=$(cn "$CNODE1_IP" 'cat /var/lib/rancher/k3s/server/token')
if [ -z "$JOIN_TOKEN" ] || [[ "$JOIN_TOKEN" == cat:* ]]; then
echo "FATAL: could not fetch token. Got: ${JOIN_TOKEN:0:40}"
exit 1
fi
echo " Token OK (masked: ${JOIN_TOKEN:0:12}***)"
# -------------------------------------------------------
# 2. Uninstall broken k3s from cnode2 + cnode3
# -------------------------------------------------------
for node_ip in "$CNODE2_IP" "$CNODE3_IP"; do
echo ""
echo "--- [2/4] Wiping broken k3s on $node_ip ---"
cn "$node_ip" '
systemctl stop k3s 2>/dev/null || true
if [ -x /usr/local/bin/k3s-uninstall.sh ]; then
/usr/local/bin/k3s-uninstall.sh
else
echo "no k3s to remove"
fi
rm -rf /etc/rancher/k3s /var/lib/rancher/k3s /var/lib/kubelet /var/lib/cni
rm -f /etc/systemd/system/k3s.service /etc/systemd/system/k3s.service.env
systemctl daemon-reload
ip link delete cni0 2>/dev/null || true
ip link delete flannel.1 2>/dev/null || true
echo "WIPE DONE"
' | tail -3
done
# -------------------------------------------------------
# 3. Install cnode2 + cnode3 with correct token
# -------------------------------------------------------
for node_ip in "$CNODE2_IP" "$CNODE3_IP"; do
case "$node_ip" in
"$CNODE2_IP") node_name="$CNODE2_NAME" ;;
"$CNODE3_IP") node_name="$CNODE3_NAME" ;;
esac
echo ""
echo "--- [3/4] Joining server: $node_name ($node_ip) ---"
ssh "${SSH_OPTS[@]}" "${SSH_USER}@${node_ip}" "sudo -n bash -s" <<REMOTE_JOIN
set -euo pipefail
export INSTALL_K3S_VERSION="$K3S_VERSION"
export K3S_TOKEN="$JOIN_TOKEN"
curl -sfL https://get.k3s.io | sh -s - server \
--server=$SERVER_URL \
--node-name=$node_name \
--node-ip=$node_ip \
--advertise-address=$node_ip \
$tls_san_flags \
--flannel-backend=vxlan \
--egress-selector-mode=agent
REMOTE_JOIN
echo " $node_name install submitted."
done
# -------------------------------------------------------
# 4. Wait for all 3 nodes Ready
# -------------------------------------------------------
echo ""
echo "--- [4/4] Waiting for all 3 nodes Ready ---"
for i in $(seq 1 30); do
READY_NODES=$(cn "$CNODE1_IP" 'k3s kubectl get nodes --no-headers 2>/dev/null | grep -c " Ready"' 2>/dev/null || echo 0)
if [ "$READY_NODES" = "3" ]; then
echo " All 3 nodes Ready."
break
fi
echo " ...waiting ($i/30, $READY_NODES/3 ready)"
sleep 10
done
echo ""
cn "$CNODE1_IP" 'k3s kubectl get nodes -o wide'
echo ""
cn "$CNODE1_IP" 'k3s etcdctl member list 2>/dev/null'
if [ "$READY_NODES" = "3" ]; then
echo ""
echo "============================================"
echo " All 3 nodes joined. HA control plane active."
echo " Run: bash $SCRIPT_DIR/post-setup.sh"
echo "============================================"
else
echo ""
echo "WARN: $READY_NODES/3 ready. Check journalctl -u k3s on the failing node."
exit 1
fi