ultix perf/ops workbench: complete 2026-08-31 optimization pass for VM 5111
Everything for the ultix-streaming (VM 5111, pfv-tsys5) performance pass: full report + host audit results, staged/gated configs, guest prep + host one-shot + post-reboot-fix + netcheck lifecycle scripts, grow-root manual runbook, rolling tracking HUD, questions v1, and the gateway boot-race hardening units. Applied and verified live 2026-08-31; open work is tracked in Redmine project 55 as #601-#607. [#602] 💘 Generated with Crush Assisted-by: Crush:glm-5.2
This commit is contained in:
Executable
+95
@@ -0,0 +1,95 @@
|
||||
#!/usr/bin/env bash
|
||||
# ONE-SHOT host pass v3 — does EVERYTHING except rebooting VM 5111.
|
||||
# ssh root@pfv-tsys5.knel.net 'bash -s' < ~/optimize/2-host-one-shot.sh
|
||||
# ssh root@pfv-tsys5.knel.net 'bash -s -- --go' < ~/optimize/2-host-one-shot.sh
|
||||
# ssh root@pfv-tsys5.knel.net 'bash -s -- --go --with-vm-restart' < ... # optional
|
||||
#
|
||||
# 1. VM 5111's disk/net/onboot flags are written as PENDING config: they
|
||||
# activate automatically at the VM's NEXT reboot (tonight, your last step
|
||||
# today, Friday — whenever the VM naturally restarts). Nothing forces it.
|
||||
# 2. CPU priority (live): VM 5111 = 9000, sectestbed/preprod fleet = 50.
|
||||
# 3. Reboot wave: every OTHER running VM.
|
||||
# Optional --with-vm-restart: also gracefully bounce VM 5111 at the end.
|
||||
# THE HOST ITSELF IS NEVER REBOOTED (NFS server). NO DISK IS RESIZED
|
||||
# (storage ops are manual-only: GROW-ROOT-RUNBOOK.md).
|
||||
set -euo pipefail
|
||||
GO=0; VMR=0; SKIPWAVE=0
|
||||
for a in "$@"; do
|
||||
case "$a" in
|
||||
--go) GO=1 ;;
|
||||
--with-vm-restart) VMR=1 ;;
|
||||
--skip-wave) SKIPWAVE=1 ;;
|
||||
esac
|
||||
done
|
||||
VM=5111
|
||||
FLEET="5000 5101 5102 5103 5104 5105 5106 5107 5108 5109 51011 51012 51013 51014 51015 51016 515 53100 53101 53102 53103 53104 53105 53106 53107 53108"
|
||||
run() { if [ "$GO" = 1 ]; then "$@"; else echo "DRY: $*"; fi; }
|
||||
staget() { # pending-stage a qm set; tolerate refusal on a running VM
|
||||
if [ "$GO" = 1 ]; then
|
||||
qm set "$@" || echo " NOTE: not staged while running; use --with-vm-restart (or set while stopped) later"
|
||||
else
|
||||
echo "DRY(pending): qm set $*"
|
||||
fi
|
||||
}
|
||||
|
||||
echo "== 1. VM $VM flags as PENDING config (live at its next reboot) =="
|
||||
staget "$VM" --onboot 1 --startup order=10,up=180
|
||||
staget "$VM" -scsi0 NVME:5111/vm-5111-disk-0.qcow2,iothread=1,ssd=1,discard=on,size=438G
|
||||
staget "$VM" -scsi1 ssd2:5111/vm-5111-disk-0.qcow2,iothread=1,ssd=1,discard=on,size=400G
|
||||
staget "$VM" -scsi2 SSD:5111/vm-5111-disk-0.qcow2,iothread=1,ssd=1,discard=on,size=200G
|
||||
staget "$VM" --net0 virtio=BC:24:11:1A:8F:6F,bridge=vmbr0,queues=4
|
||||
staget "$VM" --net1 virtio=BC:24:11:E3:32:D9,bridge=datanet,queues=2
|
||||
staget "$VM" --delete ide2
|
||||
|
||||
echo "== 2. CPU priority (live, reversible) =="
|
||||
run qm set "$VM" --cpuunits 9000
|
||||
for id in $FLEET; do run qm set "$id" --cpuunits 50; done
|
||||
|
||||
echo "== 3. reboot wave: every other running VM =="
|
||||
if [ "$SKIPWAVE" = 1 ]; then
|
||||
echo "skipped (--skip-wave)"
|
||||
else
|
||||
OTHERS=$(qm list | awk 'NR>1 && $3=="running" && $1!="'"$VM"'" {print $1}')
|
||||
echo "targets: ${OTHERS//$'\n'/ }"
|
||||
for id in $OTHERS; do
|
||||
if [ "$GO" = 1 ]; then
|
||||
qm reboot "$id" 2>/dev/null || { qm shutdown "$id" --timeout 120 || true; sleep 2; qm start "$id" 2>/dev/null || true; }
|
||||
else
|
||||
echo "DRY: reboot $id"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
if [ "$GO" = 1 ]; then
|
||||
echo; echo "== pending queue for VM $VM (activates at its next restart) =="
|
||||
qm pending "$VM" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
LOG=/var/log/ukrrs-vm5111-bounce.log
|
||||
if [ "$VMR" = 1 ]; then
|
||||
echo "== graceful bounce of VM $VM (detached; flags applied while STOPPED) =="
|
||||
qm status "$VM"
|
||||
if [ "$GO" = 1 ]; then
|
||||
setsid bash -c "
|
||||
echo bounce-start \$(date -Is)
|
||||
qm shutdown $VM --timeout 120
|
||||
qm wait $VM --timeout 180 || true
|
||||
qm set $VM --onboot 1 --startup order=10,up=180
|
||||
qm set $VM -scsi0 NVME:5111/vm-5111-disk-0.qcow2,iothread=1,ssd=1,discard=on,size=438G
|
||||
qm set $VM -scsi1 ssd2:5111/vm-5111-disk-0.qcow2,iothread=1,ssd=1,discard=on,size=400G
|
||||
qm set $VM -scsi2 SSD:5111/vm-5111-disk-0.qcow2,iothread=1,ssd=1,discard=on,size=200G
|
||||
qm set $VM --net0 virtio=BC:24:11:1A:8F:6F,bridge=vmbr0,queues=4
|
||||
qm set $VM --net1 virtio=BC:24:11:E3:32:D9,bridge=datanet,queues=2
|
||||
qm set $VM --delete ide2 || true
|
||||
qm start $VM
|
||||
echo bounce-done \$(date -Is)
|
||||
" </dev/null >"$LOG" 2>&1 &
|
||||
echo "dispatched. Downtime ~3-4 min; it comes back on its own."
|
||||
echo "Watch from anywhere: ssh root@pfv-tsys5.knel.net tail -f $LOG"
|
||||
else
|
||||
echo "DRY: would detach-bounce $VM, applying flags while stopped, log $LOG"
|
||||
fi
|
||||
else
|
||||
echo
|
||||
echo "VM $VM was NOT rebooted. Flags sit pending and go live at its next restart."
|
||||
fi
|
||||
Reference in New Issue
Block a user