Establish shellcheck as a mandatory pre-commit quality gate and bring all 93
shell scripts to a clean state.
- tests/shellcheck.sh: wrapper that runs koalaman/shellcheck:stable via Docker
(no native binary needed), skips vendored + upstream librenms-agent scripts.
- .shellcheckrc: documents intentional codebase-wide disables (dynamic source
paths SC1090/SC1091, client-side ssh expansion SC2029).
- AGENTS.md: new Git Policy rule mandating clean shellcheck for every shell
script before commit.
Fixes applied (real bugs + quality): missing quote in netinfra/gather-configs.sh
(caused cascading parse errors), unquoted expansions, declare-and-assign masking,
egrep -> grep -E, $FUNCNAME array indexing, unused variable removal, cd || exit.
Intentional patterns (sourced config, sysfs/ps diagnostics, ssh heredocs that
expand local config) get justified targeted disables.
💘 Generated with Crush
Assisted-by: Crush:glm-5.2
110 lines
4.1 KiB
Bash
110 lines
4.1 KiB
Bash
#!/usr/bin/bash
|
|
#
|
|
# k8s/verify.sh — health check for the pfv-k8s control plane
|
|
#
|
|
set -uo pipefail
|
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
# shellcheck source=./env.sh
|
|
source "$SCRIPT_DIR/env.sh"
|
|
|
|
export KUBECONFIG="${KUBECONFIG:-$HOME/.kube/config.pfv-k8s}"
|
|
|
|
PASS=0
|
|
FAIL=0
|
|
ok() { echo " [PASS] $1"; PASS=$((PASS+1)); }
|
|
fail() { echo " [FAIL] $1"; FAIL=$((FAIL+1)); }
|
|
|
|
echo "============================================"
|
|
echo " pfv-k8s Control Plane Health Check"
|
|
echo "============================================"
|
|
|
|
# 1. All 3 nodes Ready
|
|
echo ""
|
|
echo "--- Nodes Ready ---"
|
|
READY=$(kubectl get nodes --no-headers 2>/dev/null | grep -c " Ready" || echo 0)
|
|
if [ "$READY" = "3" ]; then ok "All 3 nodes Ready"; else fail "Expected 3 Ready nodes, got $READY"; fi
|
|
|
|
kubectl get nodes -o wide 2>&1 | sed 's/^/ /'
|
|
|
|
# 2. Nodes use Tailscale IPs
|
|
echo ""
|
|
echo "--- Tailscale IPs ---"
|
|
for name in "${ALL_CNODE_NAMES[@]}"; do
|
|
IP=$(kubectl get node "$name" -o jsonpath='{.status.addresses[?(@.type=="InternalIP")].address}' 2>/dev/null)
|
|
case "$IP" in
|
|
100.*) ok "$name uses Tailscale IP ($IP)" ;;
|
|
*) fail "$name uses non-Tailscale IP ($IP)" ;;
|
|
esac
|
|
done
|
|
|
|
# 3. Taints applied (NoSchedule on all cnodes)
|
|
echo ""
|
|
echo "--- Control-plane taints ---"
|
|
for name in "${ALL_CNODE_NAMES[@]}"; do
|
|
TAINT=$(kubectl get node "$name" -o jsonpath='{.spec.taints[*].key}' 2>/dev/null)
|
|
if echo "$TAINT" | grep -q "control-plane"; then
|
|
ok "$name has control-plane taint"
|
|
else
|
|
fail "$name missing control-plane taint"
|
|
fi
|
|
done
|
|
|
|
# 4. etcd members = 3 (k3s v1.36 embeds etcdctl; verify via node roles + API)
|
|
echo ""
|
|
echo "--- etcd quorum ---"
|
|
# All 3 nodes must have the etcd role label
|
|
ETCD_NODES=$(kubectl get nodes -o jsonpath='{range .items[*]}{.metadata.name}{"\t"}{.metadata.labels.node-role\.kubernetes\.io/etcd}{"\n"}{end}' 2>/dev/null | grep -c "true" || echo 0)
|
|
if [ "$ETCD_NODES" = "3" ]; then ok "3 nodes have etcd role (embedded HA etcd)"; else fail "Only $ETCD_NODES/3 nodes have etcd role"; fi
|
|
|
|
# Verify etcd is the backing store via the API (if etcd is down, this fails)
|
|
LEASE_COUNT=$(kubectl get leases -A --no-headers 2>/dev/null | wc -l)
|
|
if [ "$LEASE_COUNT" -gt "0" ]; then
|
|
ok "etcd backing store active ($LEASE_COUNT leases found)"
|
|
else
|
|
fail "No leases found — etcd may not be accepting writes"
|
|
fi
|
|
|
|
# Check etcd leader via metrics on cnode1
|
|
LEADER=$(cn "$CNODE1_IP" 'ETCDCTL_API=3 /var/lib/rancher/k3s/data/current/bin/etcdctl \
|
|
--endpoints=https://127.0.0.1:2379 \
|
|
--cacert=/var/lib/rancher/k3s/server/tls/etcd/server-ca.crt \
|
|
--cert=/var/lib/rancher/k3s/server/tls/etcd/server-client.crt \
|
|
--key=/var/lib/rancher/k3s/server/tls/etcd/server-client.key \
|
|
endpoint status 2>/dev/null' 2>/dev/null)
|
|
if [ -n "$LEADER" ]; then
|
|
ok "etcd endpoint reachable ($LEADER)"
|
|
else
|
|
# etcdctl not on disk in k3s v1.36; rely on node roles + leases above
|
|
ok "etcd health confirmed via 3 node roles + active leases (etcdctl not standalone in k3s v1.36)"
|
|
fi
|
|
|
|
# 5. CoreDNS running
|
|
echo ""
|
|
echo "--- System components ---"
|
|
COREDNS=$(kubectl get pods -n kube-system -l k8s-app=kube-dns --no-headers 2>/dev/null | grep -c "Running" || echo 0)
|
|
if [ "$COREDNS" -ge "1" ]; then ok "CoreDNS running"; else fail "CoreDNS not running"; fi
|
|
|
|
# 6. API server reachable over Tailscale
|
|
echo ""
|
|
echo "--- API server (Tailscale) ---"
|
|
if kubectl get --raw=/readyz 2>/dev/null | grep -q "ok"; then
|
|
ok "API server healthy over Tailscale"
|
|
else
|
|
fail "API server not reachable"
|
|
fi
|
|
|
|
# 7. No user workloads on cnodes
|
|
echo ""
|
|
echo "--- Workload isolation ---"
|
|
USER_PODS=$(kubectl get pods -A --field-selector spec.nodeName="${CNODE1_NAME}" -o jsonpath='{.items[*].metadata.name}' 2>/dev/null | wc -w)
|
|
# Subtract system pods
|
|
SYSTEM_PODS=$(kubectl get pods -A --field-selector spec.nodeName="${CNODE1_NAME}" -l k8s-app --no-headers 2>/dev/null | wc -l)
|
|
if [ "$((USER_PODS - SYSTEM_PODS))" -le 0 ]; then ok "Only system pods on cnodes (expected)"; else fail "Unexpected pods on $CNODE1_NAME"; fi
|
|
|
|
echo ""
|
|
echo "============================================"
|
|
echo " Results: $PASS passed, $FAIL failed"
|
|
if [ "$FAIL" -gt 0 ]; then exit 1; fi
|
|
echo " All checks passed."
|
|
echo "============================================"
|