Update pve RustDesk GA sync scripts
This commit is contained in:
@@ -0,0 +1,39 @@
|
||||
#!/bin/bash
|
||||
# Install GPU boot ready-check on pve (detect mode by default).
|
||||
set -euo pipefail
|
||||
|
||||
BASE_URL="${BLAIR_BOOT_WATCH_BASE_URL:-https://git.easygoingaming.com/Davoguha/blair-pve-sync-temp/raw/main}"
|
||||
TARGET='/usr/local/sbin/watch-blair-gpu-boot.sh'
|
||||
SHARE_COPY='/tank/blair-share/tools/golden/host/watch-blair-gpu-boot.sh'
|
||||
CRON='/etc/cron.d/blair-gpu-boot-watch'
|
||||
STATE='/var/lib/blair-gpu-boot-watch.json'
|
||||
|
||||
echo "=== Download boot watch script ==="
|
||||
curl -fsSL "$BASE_URL/watch-blair-gpu-boot.sh" -o "$TARGET"
|
||||
chmod +x "$TARGET"
|
||||
sed -i 's/\r$//' "$TARGET" 2>/dev/null || true
|
||||
|
||||
mkdir -p "$(dirname "$SHARE_COPY")" /tank/blair-share/fleet/monitor
|
||||
cp -f "$TARGET" "$SHARE_COPY"
|
||||
chmod +x "$SHARE_COPY"
|
||||
touch /var/log/blair-gpu-boot-watch.log
|
||||
|
||||
echo "=== cron (every 3 min, detect-only) ==="
|
||||
cat > "$CRON" <<'EOF'
|
||||
# Detect hung Windows boot: running but QEMU GA never responds.
|
||||
# Enable recovery: add BLAIR_BOOT_WATCH_MODE=recover to the line below (max 1 restart/6h/VM).
|
||||
*/3 * * * * root BLAIR_BOOT_WATCH_MODE=detect /usr/local/sbin/watch-blair-gpu-boot.sh >> /var/log/blair-gpu-boot-watch.log 2>&1
|
||||
EOF
|
||||
chmod 644 "$CRON"
|
||||
cat "$CRON"
|
||||
|
||||
echo "=== test run (detect) ==="
|
||||
BLAIR_BOOT_WATCH_MODE=detect "$TARGET"
|
||||
echo "=== state ==="
|
||||
cat "$STATE" 2>/dev/null || echo '(no state yet)'
|
||||
echo "=== share status ==="
|
||||
cat /tank/blair-share/fleet/monitor/boot-watch.json 2>/dev/null || true
|
||||
echo "=== done ==="
|
||||
echo ""
|
||||
echo "To enable auto-recovery after burn-in:"
|
||||
echo " sed -i 's/MODE=detect/MODE=recover/' $CRON"
|
||||
@@ -0,0 +1,212 @@
|
||||
#!/bin/bash
|
||||
# GPU seat boot ready-check via QEMU guest agent (standard Proxmox/KVM signal).
|
||||
# Detects: hypervisor "running" but Windows never reached GA-ready (hung boot).
|
||||
# Optional gentle recovery: one ACPI shutdown+start per VM per cooldown window.
|
||||
#
|
||||
# Mode: BLAIR_BOOT_WATCH_MODE=detect|recover (default: detect — log only)
|
||||
# Install cron every 3-5 min alongside monitor-blair-fleet.sh.
|
||||
set -euo pipefail
|
||||
|
||||
export BLAIR_SHARE="${BLAIR_SHARE:-/tank/blair-share}"
|
||||
export BLAIR_BOOT_WATCH_STATE="${BLAIR_BOOT_WATCH_STATE:-/var/lib/blair-gpu-boot-watch.json}"
|
||||
export BLAIR_BOOT_WATCH_LOG="${BLAIR_BOOT_WATCH_LOG:-/var/log/blair-gpu-boot-watch.log}"
|
||||
|
||||
if ! command -v python3 >/dev/null 2>&1; then
|
||||
echo 'python3 required' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
exec python3 - <<'PY'
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
|
||||
NODE = os.environ.get('BLAIR_PVE_NODE', 'pve')
|
||||
GPU_VMIDS = [230, 231, 232, 233, 234, 235]
|
||||
STATE_PATH = os.environ.get('BLAIR_BOOT_WATCH_STATE', '/var/lib/blair-gpu-boot-watch.json')
|
||||
LOG_PATH = os.environ.get('BLAIR_BOOT_WATCH_LOG', '/var/log/blair-gpu-boot-watch.log')
|
||||
SHARE_STATUS = os.path.join(os.environ.get('BLAIR_SHARE', '/tank/blair-share'), 'fleet/monitor/boot-watch.json')
|
||||
|
||||
MODE = os.environ.get('BLAIR_BOOT_WATCH_MODE', 'detect').lower() # detect | recover
|
||||
GRACE_SEC = int(os.environ.get('BLAIR_BOOT_GRACE_SEC', '600')) # 10 min after start
|
||||
MAX_UPTIME_SEC = int(os.environ.get('BLAIR_BOOT_MAX_UPTIME_SEC', '3600')) # only watch first hour
|
||||
RECOVER_COOLDOWN_SEC = int(os.environ.get('BLAIR_BOOT_RECOVER_COOLDOWN', '21600')) # 6h
|
||||
MAX_RECOVER_PER_BOOT = int(os.environ.get('BLAIR_BOOT_MAX_RECOVER_PER_BOOT', '1'))
|
||||
|
||||
|
||||
def log(msg: str) -> None:
|
||||
line = f"{datetime.now().strftime('%Y-%m-%d %H:%M:%S')} {msg}"
|
||||
print(line, flush=True)
|
||||
with open(LOG_PATH, 'a', encoding='utf-8') as fh:
|
||||
fh.write(line + '\n')
|
||||
|
||||
|
||||
def load_state() -> dict:
|
||||
if not os.path.isfile(STATE_PATH):
|
||||
return {'vms': {}}
|
||||
with open(STATE_PATH, encoding='utf-8') as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def save_state(state: dict) -> None:
|
||||
os.makedirs(os.path.dirname(STATE_PATH), exist_ok=True)
|
||||
tmp = STATE_PATH + '.tmp'
|
||||
with open(tmp, 'w', encoding='utf-8') as fh:
|
||||
json.dump(state, fh, indent=2)
|
||||
fh.write('\n')
|
||||
os.replace(tmp, STATE_PATH)
|
||||
|
||||
|
||||
def qm_status(vmid: int) -> dict:
|
||||
proc = subprocess.run(
|
||||
['qm', 'status', str(vmid), '--verbose'],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
out = proc.stdout
|
||||
data = {'status': 'unknown', 'uptime': 0}
|
||||
for line in out.splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith('status:'):
|
||||
data['status'] = line.split(':', 1)[1].strip()
|
||||
elif line.startswith('uptime:'):
|
||||
try:
|
||||
data['uptime'] = int(line.split(':', 1)[1].strip())
|
||||
except ValueError:
|
||||
pass
|
||||
return data
|
||||
|
||||
|
||||
def ga_ping(vmid: int) -> bool:
|
||||
proc = subprocess.run(
|
||||
['qm', 'guest', 'cmd', str(vmid), 'ping'],
|
||||
capture_output=True, text=True, check=False,
|
||||
)
|
||||
return proc.returncode == 0
|
||||
|
||||
|
||||
def recover_vm(vmid: int) -> None:
|
||||
log(f'VM {vmid}: recovery — ACPI shutdown (forceStop) then start')
|
||||
subprocess.run(
|
||||
['qm', 'shutdown', str(vmid), '--timeout', '90', '--forceStop', '1'],
|
||||
check=False,
|
||||
)
|
||||
deadline = time.time() + 120
|
||||
while time.time() < deadline:
|
||||
if qm_status(vmid)['status'] == 'stopped':
|
||||
break
|
||||
time.sleep(3)
|
||||
if qm_status(vmid)['status'] != 'stopped':
|
||||
log(f'VM {vmid}: still not stopped; trying qm stop')
|
||||
subprocess.run(['qm', 'stop', str(vmid)], check=False)
|
||||
time.sleep(5)
|
||||
subprocess.run(['qm', 'start', str(vmid)], check=True)
|
||||
log(f'VM {vmid}: start issued after recovery')
|
||||
|
||||
|
||||
def main() -> int:
|
||||
state = load_state()
|
||||
vms = state.setdefault('vms', {})
|
||||
report = {'timestamp': datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ'), 'mode': MODE, 'vms': []}
|
||||
|
||||
for vmid in GPU_VMIDS:
|
||||
key = str(vmid)
|
||||
st = qm_status(vmid)
|
||||
entry = vms.setdefault(key, {
|
||||
'last_uptime': 0,
|
||||
'boot_marked': False,
|
||||
'ga_ready': False,
|
||||
'recoveries_this_boot': 0,
|
||||
'last_recovery_ts': 0,
|
||||
'last_ready_ts': 0,
|
||||
})
|
||||
|
||||
# New boot cycle detected (uptime went down)
|
||||
if st['uptime'] < entry.get('last_uptime', 0) - 30:
|
||||
entry['boot_marked'] = False
|
||||
entry['ga_ready'] = False
|
||||
entry['recoveries_this_boot'] = 0
|
||||
log(f'VM {vmid}: new boot detected (uptime={st["uptime"]}s)')
|
||||
|
||||
entry['last_uptime'] = st['uptime']
|
||||
|
||||
row = {'vmid': vmid, 'status': st['status'], 'uptime_sec': st['uptime']}
|
||||
|
||||
if st['status'] != 'running':
|
||||
row['ready'] = None
|
||||
row['action'] = 'skip_not_running'
|
||||
report['vms'].append(row)
|
||||
continue
|
||||
|
||||
if ga_ping(vmid):
|
||||
if not entry.get('ga_ready'):
|
||||
log(f'VM {vmid}: GA ready (uptime={st["uptime"]}s) — Windows boot OK')
|
||||
entry['ga_ready'] = True
|
||||
entry['boot_marked'] = True
|
||||
entry['last_ready_ts'] = int(time.time())
|
||||
entry['recoveries_this_boot'] = 0
|
||||
row['ready'] = True
|
||||
row['action'] = 'ok'
|
||||
report['vms'].append(row)
|
||||
continue
|
||||
|
||||
row['ready'] = False
|
||||
uptime = st['uptime']
|
||||
|
||||
if uptime < GRACE_SEC:
|
||||
row['action'] = 'grace_period'
|
||||
report['vms'].append(row)
|
||||
continue
|
||||
|
||||
if uptime > MAX_UPTIME_SEC:
|
||||
row['action'] = 'outside_boot_window'
|
||||
report['vms'].append(row)
|
||||
continue
|
||||
|
||||
# Hung boot: running, past grace, GA still down, within boot window
|
||||
log(f'VM {vmid}: NOT READY — running {uptime}s, GA ping failed (hung boot?)')
|
||||
row['action'] = 'not_ready'
|
||||
|
||||
if MODE != 'recover':
|
||||
row['recovery'] = 'detect_only'
|
||||
report['vms'].append(row)
|
||||
continue
|
||||
|
||||
now = int(time.time())
|
||||
last_rec = int(entry.get('last_recovery_ts') or 0)
|
||||
rec_count = int(entry.get('recoveries_this_boot') or 0)
|
||||
|
||||
if rec_count >= MAX_RECOVER_PER_BOOT:
|
||||
log(f'VM {vmid}: recovery skipped (already {rec_count}x this boot)')
|
||||
row['recovery'] = 'skipped_max_per_boot'
|
||||
elif now - last_rec < RECOVER_COOLDOWN_SEC:
|
||||
log(f'VM {vmid}: recovery skipped (cooldown)')
|
||||
row['recovery'] = 'skipped_cooldown'
|
||||
else:
|
||||
try:
|
||||
recover_vm(vmid)
|
||||
entry['last_recovery_ts'] = now
|
||||
entry['recoveries_this_boot'] = rec_count + 1
|
||||
entry['ga_ready'] = False
|
||||
entry['boot_marked'] = False
|
||||
row['recovery'] = 'restarted'
|
||||
except Exception as exc: # noqa: BLE001
|
||||
log(f'VM {vmid}: recovery failed: {exc}')
|
||||
row['recovery'] = f'failed:{exc}'
|
||||
|
||||
report['vms'].append(row)
|
||||
|
||||
save_state(state)
|
||||
os.makedirs(os.path.dirname(SHARE_STATUS), exist_ok=True)
|
||||
tmp = SHARE_STATUS + '.tmp'
|
||||
with open(tmp, 'w', encoding='utf-8') as fh:
|
||||
json.dump(report, fh, indent=2)
|
||||
fh.write('\n')
|
||||
os.replace(tmp, SHARE_STATUS)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main())
|
||||
PY
|
||||
Reference in New Issue
Block a user