From 730ce94fac5bae5f05731a5ad92b33e252d79783 Mon Sep 17 00:00:00 2001 From: Marius Date: Thu, 27 Aug 2026 14:04:50 +0300 Subject: [PATCH] feat(cluster): varianta PowerShell a scripturilor de oprire/repornire Statia de admin e Windows si nu ruleaza .sh direct: `bash` din PATH e cel din WSL, cu alt filesystem si alta configuratie de chei SSH. Adaug echivalentele native PowerShell. - cluster-shutdown.ps1 / cluster-startup.ps1, aceeasi logica si aceleasi garantii ca variantele bash - folosesc clientul OpenSSH din Windows, deja prezent in System32 - compatibile PowerShell 5.1: fara &&/||, fara ternar, fara ?., fara -AsHashtable; ping prin System.Net.NetworkInformation.Ping, nu Test-Connection (WMI, lent si des blocat) - nu redirecteaza stderr-ul lui ssh: in 5.1 asta transforma fiecare linie intr-un ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit - parsarea `pct list` / `qm list` se face in PowerShell, nu prin awk remote, ca sa nu se incurce interpolarea $1/$2 din stringurile PowerShell - scriu acelasi format de fisier de stare ca variantele bash, deci poti opri cu una si porni cu cealalta Ambele testate cu -DryRun pe clusterul live, rezultat identic cu bash. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01RRyaDj39hPQ89SZS6URRpS --- .../docs/oprire-planificata-cluster.md | 33 +- proxmox/cluster/scripts/cluster-shutdown.ps1 | 417 ++++++++++++++++++ proxmox/cluster/scripts/cluster-startup.ps1 | 401 +++++++++++++++++ 3 files changed, 845 insertions(+), 6 deletions(-) create mode 100644 proxmox/cluster/scripts/cluster-shutdown.ps1 create mode 100644 proxmox/cluster/scripts/cluster-startup.ps1 diff --git a/proxmox/cluster/docs/oprire-planificata-cluster.md b/proxmox/cluster/docs/oprire-planificata-cluster.md index 14a9e49..92f640f 100644 --- a/proxmox/cluster/docs/oprire-planificata-cluster.md +++ b/proxmox/cluster/docs/oprire-planificata-cluster.md @@ -13,19 +13,40 @@ mutare rack, mentenanță UPS, intervenții la switch. ## Varianta automată Toată procedura de mai jos e implementată în două scripturi, de rulat **de pe -stația de admin** (nu de pe un nod): +stația de admin** (nu de pe un nod). Există în două variante echivalente — +folosește-o pe cea potrivită shell-ului tău. -```bash -cd proxmox/cluster/scripts +**PowerShell (Windows, varianta obișnuită aici):** -./cluster-shutdown.sh --dry-run # arată exact ce ar face, nu atinge nimic -./cluster-shutdown.sh # oprire completă, cu confirmări +```powershell +cd E:\proiecte\ROMFASTSQL\proxmox\cluster\scripts + +.\cluster-shutdown.ps1 -DryRun # arată exact ce ar face, nu atinge nimic +.\cluster-shutdown.ps1 # oprire completă, cu confirmări # ... după revenirea curentului, cu nodurile pornite fizic ... -./cluster-startup.sh # repornire completă +.\cluster-startup.ps1 # repornire completă ``` +Cere doar clientul OpenSSH din Windows (`C:\Windows\System32\OpenSSH\ssh.exe`), +deja prezent, și cheia SSH pentru root pe noduri. Testate pe PowerShell 5.1. + +Dacă politica de execuție blochează scriptul: +`powershell -ExecutionPolicy Bypass -File .\cluster-shutdown.ps1 -DryRun` + +**Bash (Git Bash, WSL sau de pe CT 171 claude-agent):** + +```bash +cd proxmox/cluster/scripts +./cluster-shutdown.sh --dry-run +./cluster-shutdown.sh +./cluster-startup.sh +``` + +Cele două variante scriu același fișier de stare, deci poți opri cu una și porni +cu cealaltă. + `cluster-shutdown.sh` salvează local ce guest-uri rulau, iar `cluster-startup.sh` pornește **exact** ce era pornit — nu repornește VM-uri oprite intenționat. diff --git a/proxmox/cluster/scripts/cluster-shutdown.ps1 b/proxmox/cluster/scripts/cluster-shutdown.ps1 new file mode 100644 index 0000000..3e23478 --- /dev/null +++ b/proxmox/cluster/scripts/cluster-shutdown.ps1 @@ -0,0 +1,417 @@ +<# +.SYNOPSIS + Oprire controlata a intregului cluster Proxmox `romfast`. + +.DESCRIPTION + Opreste guest-urile in ordinea corecta de dependente, apoi nodurile, fara ca + HA sa relocheze resursele si fara ca watchdog-ul sa reseteze hard vreun nod. + + Procedura completa si explicatiile: docs/oprire-planificata-cluster.md + + RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod Proxmox. Motivul: nodul care + se opreste ultimul nu poate raporta rezultatul propriei opriri. + + Necesita clientul OpenSSH din Windows (C:\Windows\System32\OpenSSH\ssh.exe) + si o cheie SSH acceptata de root pe toate cele trei noduri. + +.PARAMETER Yes + Nu cere confirmari. + +.PARAMETER DryRun + Arata ce ar face, fara sa execute nimic care modifica starea. + +.PARAMETER NoNodes + Opreste doar guest-urile, lasa nodurile pornite. + +.EXAMPLE + .\cluster-shutdown.ps1 -DryRun +.EXAMPLE + .\cluster-shutdown.ps1 +#> +[CmdletBinding()] +param( + [switch]$Yes, + [switch]$DryRun, + [switch]$NoNodes +) + +$ErrorActionPreference = 'Stop' + +# ---------------------------------------------------------------- configurare + +$NodeIp = [ordered]@{ + pve1 = '10.0.20.200' + pvemini = '10.0.20.201' + pveelite = '10.0.20.202' +} + +# Ordinea de OPRIRE a nodurilor. pvemini ultimul - e nodul principal si serverul NFS. +$NodeShutdownOrder = @('pveelite', 'pve1', 'pvemini') + +# Ordinea de OPRIRE a guest-urilor: consumatorii intai, baza de date ULTIMA. +# VM 201 (roacentral) si CT 104 (flowise) folosesc Oracle din CT 108. +# Orice guest pornit care nu apare aici e oprit la final, intr-o maturare. +$GuestShutdownOrder = @( + 303, # Win11-Adina - desktop, fara dependente + 302, # oracle-test-302 - VM de test + 310, # Win11-Template - template + 201, # roacentral - IIS reverse proxy, consumator Oracle + 101, # minecraft + 110, # moltbot + 104, # flowise - consumator Oracle + 106, # gitea + 103, # dokploy + 102, # docker.romfast.ro + 100, # portainer + 171, # claude-agent + 109, # oracle-dr-windows + 301, # docker-portainer-template + 108 # central-oracle - ULTIMUL +) + +$OracleCt = 108 +$OracleDocker = 'oracle-xe' + +$GuestTimeout = 300 # secunde asteptate pentru shutdown-ul unui guest +$NodeTimeout = 600 # secunde asteptate pana un nod nu mai raspunde la ping +$CronBackup = '/root/crontab.backup-mentenanta.txt' + +# Cron-uri de dezactivat cat timp clusterul e jos (altfel mail storm, iar +# vm109-watchdog.sh chiar porneste VM 109). +$CronPattern = 'oom-alert|pvemini-down-alert|pveelite-down-alert|vm109-watchdog' + +$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path +$StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' } + +# ------------------------------------------------------------------- utilitare + +function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green } +function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green } +function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow } +function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray } +function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 } + +function Test-Alive { + param([string]$Address, [int]$TimeoutMs = 2000) + try { + $p = New-Object System.Net.NetworkInformation.Ping + return ($p.Send($Address, $TimeoutMs).Status -eq 'Success') + } catch { + return $false + } +} + +# Executa o comanda pe un nod. Intoarce liniile de output. +# NU redirectam stderr: in PS 5.1 asta transforma fiecare linie intr-un +# ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit. +function Invoke-Node { + param([string]$Node, [string]$Command) + $out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command + return $out +} + +# Ca Invoke-Node, dar respecta -DryRun (pentru comenzi care modifica starea). +function Invoke-NodeChange { + param([string]$Node, [string]$Command) + if ($DryRun) { + Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray + return @() + } + return Invoke-Node -Node $Node -Command $Command +} + +function Confirm-Step { + param([string]$Message) + if ($Yes -or $DryRun) { return } + Write-Host "" + Write-Host $Message -ForegroundColor Yellow + $reply = Read-Host "Scrie DA ca sa continui" + if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." } +} + +# Parseaza `pct list` / `qm list` in obiecte. +function Get-NodeGuests { + param([string]$Node) + $result = @() + + $ctLines = Invoke-Node -Node $Node -Command 'pct list' + foreach ($line in ($ctLines | Select-Object -Skip 1)) { + $f = ($line -split '\s+') | Where-Object { $_ -ne '' } + if ($f.Count -ge 2) { + $result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] } + } + } + + $vmLines = Invoke-Node -Node $Node -Command 'qm list' + foreach ($line in ($vmLines | Select-Object -Skip 1)) { + $f = ($line -split '\s+') | Where-Object { $_ -ne '' } + if ($f.Count -ge 3) { + $result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] } + } + } + return $result +} + +function Get-GuestStatus { + param([string]$Node, [string]$Type, [int]$Id) + $cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" } + $out = Invoke-Node -Node $Node -Command $cmd + if ($out -match 'status:\s*(\w+)') { return $Matches[1] } + return 'unknown' +} + +# ------------------------------------------------------------ pas 1: preflight + +Write-Step "Preflight" + +foreach ($node in $NodeIp.Keys) { + if (-not (Test-Alive $NodeIp[$node])) { + Stop-WithError "$node ($($NodeIp[$node])) nu raspunde la ping." + } + $h = Invoke-Node -Node $node -Command 'hostname' + if ($LASTEXITCODE -ne 0 -or -not $h) { Stop-WithError "$node nu accepta SSH cu cheie." } + Write-Ok "$node ($($NodeIp[$node])) - reachable" +} + +$quorum = Invoke-Node -Node 'pvemini' -Command 'pvecm status' +if (($quorum -join "`n") -notmatch 'Quorate:\s*Yes') { + Stop-WithError "Clusterul NU are quorum. Opreste-te si investigheaza." +} +$haStatus = Invoke-Node -Node 'pvemini' -Command 'ha-manager status' +$master = ($haStatus | Where-Object { $_ -match '^master\s+(\S+)' } | ForEach-Object { $Matches[1] }) +Write-Ok "quorum OK, HA master: $master" + +$runningTasks = 0 +foreach ($node in $NodeIp.Keys) { + $t = Invoke-Node -Node $node -Command "pvesh get /nodes/$node/tasks --limit 20 --output-format json" + $runningTasks += ([regex]::Matches(($t -join ''), '"status":"running"')).Count +} +if ($runningTasks -gt 0) { + Write-Warn "$runningTasks task-uri in executie pe cluster (backup? migrare?)." + Confirm-Step "Sunt task-uri active. Continui oricum?" +} else { + Write-Ok "niciun task in executie" +} + +# ------------------------------------------- pas 2: inventar + salvare stare + +Write-Step "Inventar guest-uri" + +$haSids = @() +foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) { + if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" } +} + +$inventory = @() +foreach ($node in $NodeIp.Keys) { + foreach ($g in (Get-NodeGuests -Node $node)) { + $sid = "$($g.Type):$($g.Id)" + $g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' }) + $inventory += $g + } +} + +$running = @($inventory | Where-Object { $_.Status -eq 'running' }) + +if ($running.Count -eq 0) { + Write-Warn "Niciun guest pornit. Trec direct la oprirea nodurilor." +} else { + foreach ($g in $running) { + " {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host + } + Write-Ok "$($running.Count) guest-uri pornite" +} + +# Salveaza starea LOCAL - cluster-startup porneste exact ce era pornit, ca sa nu +# reporneasca VM-uri oprite intentionat (302, 310, 301...). +# Format identic cu varianta bash, ca cele doua sa fie interschimbabile. +if (-not $DryRun) { + $lines = @("# stare cluster salvata la $(Get-Date -Format 'yyyy-MM-dd HH:mm:ss')") + foreach ($g in $running) { $lines += "$($g.Type) $($g.Node) $($g.Id) $($g.Status) $($g.Ha)" } + Set-Content -Path $StateFile -Value $lines -Encoding utf8 + Write-Ok "stare salvata in $StateFile" +} + +Confirm-Step "Se opresc $($running.Count) guest-uri, apoi nodurile: $($NodeShutdownOrder -join ', ')." + +# -------------------------------------------- pas 3: shutdown_policy=freeze + +Write-Step "Shutdown policy -> freeze" + +$dcCfg = Invoke-Node -Node 'pvemini' -Command 'cat /etc/pve/datacenter.cfg 2>/dev/null || true' +$currentPolicy = $null +if (($dcCfg -join "`n") -match 'shutdown_policy=(\w+)') { $currentPolicy = $Matches[1] } + +if ($currentPolicy -eq 'freeze') { + Write-Ok "deja pe freeze" +} else { + if ($null -ne $currentPolicy -or ($dcCfg | Measure-Object).Count -gt 0) { + Invoke-NodeChange -Node 'pvemini' -Command 'cp /etc/pve/datacenter.cfg /root/datacenter.cfg.backup-mentenanta' | Out-Null + Invoke-NodeChange -Node 'pvemini' -Command "sed -i '/^ha:/d' /etc/pve/datacenter.cfg; printf 'ha: shutdown_policy=freeze\n' >> /etc/pve/datacenter.cfg" | Out-Null + } else { + Invoke-NodeChange -Node 'pvemini' -Command "printf 'ha: shutdown_policy=freeze\n' > /etc/pve/datacenter.cfg" | Out-Null + } + $was = if ($currentPolicy) { $currentPolicy } else { 'nesetat' } + Write-Ok "setat freeze (era: $was)" +} + +# ---------------------------------------- pas 4: dezactivare cron-uri alerta + +Write-Step "Dezactivare cron-uri de alerta" + +$sedCron = "crontab -l | sed -E '/^[^#]/ s%^(.*($CronPattern)\.sh.*)`$%#MENTENANTA \1%' | crontab -" + +foreach ($node in $NodeIp.Keys) { + Invoke-NodeChange -Node $node -Command "crontab -l > $CronBackup 2>/dev/null || true" | Out-Null + Invoke-NodeChange -Node $node -Command $sedCron | Out-Null + if (-not $DryRun) { + $n = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true" + Write-Ok "$node - $n linii dezactivate, backup in $CronBackup" + } +} + +Write-Warn "Monitorizarea e acum OPRITA pe tot clusterul. cluster-startup o restaureaza." + +# --------------------------------------------- pas 5: shutdown curat Oracle + +Write-Step "Oracle - shutdown immediate" + +$oracleGuest = $running | Where-Object { $_.Type -eq 'ct' -and $_.Id -eq $OracleCt } | Select-Object -First 1 + +if (-not $oracleGuest) { + Write-Warn "CT $OracleCt nu ruleaza - sar peste shutdown-ul bazei." +} else { + Write-Log "CT $OracleCt pe $($oracleGuest.Node) - opresc instanta..." + if ($DryRun) { + Write-Host " [dry-run] shutdown immediate in $OracleDocker" -ForegroundColor DarkGray + } else { + $sqlCmd = 'pct exec ' + $OracleCt + ' -- docker exec ' + $OracleDocker + ' bash -c "printf ''shutdown immediate\nexit\n'' | sqlplus -s / as sysdba"' + $out = Invoke-Node -Node $oracleGuest.Node -Command $sqlCmd + if ($LASTEXITCODE -ne 0) { + Write-Warn "Shutdown-ul bazei a raportat eroare. Verifica manual inainte de a continua!" + $out | Select-Object -Last 5 | Write-Host + } else { + Write-Ok "instanta Oracle oprita" + } + } +} + +# ------------------------------------------------ pas 6: oprire guest-uri + +Write-Step "Oprire guest-uri" + +function Stop-Guest { + param([pscustomobject]$Guest) + + $sid = "$($Guest.Type):$($Guest.Id)" + $cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' } + + if ($Guest.Ha -eq 'ha') { + # Pentru resursele HA se foloseste ha-manager, NU pct/qm shutdown: din CLI + # acestea nu actualizeaza state-ul HA, iar CRM-ul ar reporni guest-ul. + Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state stopped" | Out-Null + } else { + Invoke-NodeChange -Node $Guest.Node -Command "$cmd shutdown $($Guest.Id) --timeout $GuestTimeout" | Out-Null + } + + if ($DryRun) { return } + + $waited = 0 + while ($waited -lt $GuestTimeout) { + if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'stopped') { + Write-Ok "$sid oprit (${waited}s)" + return + } + Start-Sleep -Seconds 5 + $waited += 5 + } + + Write-Warn "$sid nu s-a oprit in ${GuestTimeout}s - fortez stop" + Invoke-Node -Node $Guest.Node -Command "$cmd stop $($Guest.Id)" | Out-Null + Start-Sleep -Seconds 5 + if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -ne 'stopped') { + Stop-WithError "$sid REFUZA sa se opreasca. Rezolva manual inainte de a opri nodurile." + } + Write-Ok "$sid oprit fortat" +} + +# intai in ordinea definita... +foreach ($id in $GuestShutdownOrder) { + $g = $running | Where-Object { $_.Id -eq $id } | Select-Object -First 1 + if (-not $g) { continue } + Write-Log "opresc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))" + Stop-Guest -Guest $g +} + +# ...apoi orice a mai ramas pornit si nu era in lista +foreach ($g in $running) { + if ($GuestShutdownOrder -contains $g.Id) { continue } + Write-Warn "guest neplanificat inca pornit: $($g.Type):$($g.Id) pe $($g.Node) - il opresc" + Stop-Guest -Guest $g +} + +# ------------------------------------------- pas 7: verificare obligatorie + +Write-Step "Verificare inainte de oprirea nodurilor" + +if ($DryRun) { + Write-Warn "dry-run - sar peste verificare" +} else { + $ha = Invoke-Node -Node 'pvemini' -Command 'ha-manager status' + $bad = @($ha | Where-Object { $_ -match 'started|migrate|relocate|fence' }) + if ($bad.Count -gt 0) { + $bad | Write-Host + Stop-WithError "Exista servicii HA inca active. NU opri nodurile." + } + Write-Ok "toate serviciile HA sunt in state stopped" + + $still = 0 + foreach ($node in $NodeIp.Keys) { + $still += @((Get-NodeGuests -Node $node) | Where-Object { $_.Status -eq 'running' }).Count + } + if ($still -ne 0) { Stop-WithError "$still guest-uri inca pornite. NU opri nodurile." } + Write-Ok "0 guest-uri pornite pe cluster" + Write-Ok "watchdog-ul nu mai e armat - pierderea quorumului e inofensiva" +} + +if ($NoNodes) { + Write-Step "-NoNodes: nodurile raman pornite. Gata." + exit 0 +} + +Confirm-Step "Se opresc nodurile in ordinea: $($NodeShutdownOrder -join ', ') (pvemini ultimul)." + +# ------------------------------------------------- pas 8: oprire noduri + +Write-Step "Oprire noduri" + +foreach ($node in $NodeShutdownOrder) { + Write-Log "opresc $node ($($NodeIp[$node]))..." + Invoke-NodeChange -Node $node -Command 'systemctl poweroff --no-block' | Out-Null + if ($DryRun) { continue } + + $waited = 0 + $down = $false + while ($waited -lt $NodeTimeout) { + if (-not (Test-Alive $NodeIp[$node])) { + Write-Ok "$node OPRIT (${waited}s)" + $down = $true + break + } + Start-Sleep -Seconds 10 + $waited += 10 + } + if (-not $down) { + Write-Warn "$node inca raspunde la ping dupa ${NodeTimeout}s." + Write-Warn "Verifica manual - sshd poate fi deja jos desi reteaua e sus." + } +} + +Write-Step "Cluster oprit" +Write-Host "" +Write-Host " Stare salvata: $StateFile" +Write-Host " Backup crontab: $CronBackup (pe fiecare nod)" +Write-Host " Shutdown policy: freeze" +Write-Host "" +Write-Host " La revenirea curentului: .\cluster-startup.ps1" +Write-Host "" diff --git a/proxmox/cluster/scripts/cluster-startup.ps1 b/proxmox/cluster/scripts/cluster-startup.ps1 new file mode 100644 index 0000000..43e7c5b --- /dev/null +++ b/proxmox/cluster/scripts/cluster-startup.ps1 @@ -0,0 +1,401 @@ +<# +.SYNOPSIS + Repornirea clusterului Proxmox `romfast` dupa o oprire planificata. + +.DESCRIPTION + Asteapta nodurile, asteapta quorumul, porneste guest-urile in ordinea corecta + de dependente (Oracle primul), restaureaza cron-urile de monitorizare. + + Porneste DOAR guest-urile care rulau inainte de oprire, citite din fisierul de + stare scris de cluster-shutdown. Fara fisier de stare, foloseste lista + implicita, minus guest-urile din NeverAutostart. + + RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod. + + NODURILE TREBUIE PORNITE FIZIC INTAI. Scriptul le asteapta. + Porneste-le pe TOATE TREI aproximativ simultan: un singur nod pornit nu are + quorum, /etc/pve ramane read-only si nu se poate porni niciun guest. + +.PARAMETER Yes + Nu cere confirmari. + +.PARAMETER DryRun + Arata ce ar face, fara sa execute nimic care modifica starea. + +.PARAMETER All + Ignora fisierul de stare, foloseste lista implicita. + +.EXAMPLE + .\cluster-startup.ps1 -DryRun +.EXAMPLE + .\cluster-startup.ps1 +#> +[CmdletBinding()] +param( + [switch]$Yes, + [switch]$DryRun, + [switch]$All +) + +$ErrorActionPreference = 'Stop' + +# ---------------------------------------------------------------- configurare + +$NodeIp = [ordered]@{ + pve1 = '10.0.20.200' + pvemini = '10.0.20.201' + pveelite = '10.0.20.202' +} + +# Ordinea de PORNIRE: baza de date intai, consumatorii dupa. +# Inversul ordinii de oprire din cluster-shutdown. +$GuestStartOrder = @( + 108, # central-oracle - PRIMUL, restul depind de el + 201, # roacentral - IIS reverse proxy, consumator Oracle + 104, # flowise - consumator Oracle + 100, # portainer + 102, # docker.romfast.ro + 103, # dokploy + 106, # gitea + 171, # claude-agent + 101, # minecraft + 110, # moltbot + 303, # Win11-Adina + 302, # oracle-test-302 + 109, # oracle-dr-windows + 301, # docker-portainer-template + 310 # Win11-Template +) + +# Guest-uri care NU se pornesc automat cand LIPSESTE fisierul de stare. +# Daca fisierul de stare exista, el decide - daca erau pornite, se repornesc. +# 109 - VM DR. Pornirea lui nedorita a declansat bucla OOM din incidentul +# 2026-04-20; in mod normal sta oprit si e pornit doar de testul DR. +# 301, 310 - template-uri, nu servicii. +$NeverAutostart = @(109, 301, 310) + +$OracleCt = 108 +$OracleDocker = 'oracle-xe' +$OracleWait = 600 # secunde asteptate pana baza raspunde la interogari + +$NodeWait = 1800 # secunde asteptate pana apar toate nodurile +$QuorumWait = 300 +$GuestWait = 300 +$CronBackup = '/root/crontab.backup-mentenanta.txt' + +$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path +$StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' } + +# ------------------------------------------------------------------- utilitare + +function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green } +function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green } +function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow } +function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray } +function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 } + +function Test-Alive { + param([string]$Address, [int]$TimeoutMs = 2000) + try { + $p = New-Object System.Net.NetworkInformation.Ping + return ($p.Send($Address, $TimeoutMs).Status -eq 'Success') + } catch { + return $false + } +} + +function Invoke-Node { + param([string]$Node, [string]$Command) + $out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command + return $out +} + +function Invoke-NodeChange { + param([string]$Node, [string]$Command) + if ($DryRun) { + Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray + return @() + } + return Invoke-Node -Node $Node -Command $Command +} + +function Confirm-Step { + param([string]$Message) + if ($Yes -or $DryRun) { return } + Write-Host "" + Write-Host $Message -ForegroundColor Yellow + $reply = Read-Host "Scrie DA ca sa continui" + if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." } +} + +function Get-NodeGuests { + param([string]$Node) + $result = @() + + foreach ($line in ((Invoke-Node -Node $Node -Command 'pct list') | Select-Object -Skip 1)) { + $f = ($line -split '\s+') | Where-Object { $_ -ne '' } + if ($f.Count -ge 2) { + $result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] } + } + } + foreach ($line in ((Invoke-Node -Node $Node -Command 'qm list') | Select-Object -Skip 1)) { + $f = ($line -split '\s+') | Where-Object { $_ -ne '' } + if ($f.Count -ge 3) { + $result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] } + } + } + return $result +} + +function Get-GuestStatus { + param([string]$Node, [string]$Type, [int]$Id) + $cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" } + $out = Invoke-Node -Node $Node -Command $cmd + if ($out -match 'status:\s*(\w+)') { return $Matches[1] } + return 'unknown' +} + +# ------------------------------------------------- pas 1: asteapta nodurile + +Write-Step "Astept nodurile" +Write-Warn "Nodurile trebuie pornite FIZIC. Porneste-le pe toate trei aproximativ simultan." + +$waited = 0 +$missing = @() +while ($waited -lt $NodeWait) { + $missing = @() + foreach ($node in $NodeIp.Keys) { + if (-not (Test-Alive $NodeIp[$node])) { $missing += $node } + } + if ($missing.Count -eq 0) { break } + Write-Host "`r astept: $($missing -join ', ') (${waited}s) " -NoNewline -ForegroundColor DarkGray + Start-Sleep -Seconds 10 + $waited += 10 +} +Write-Host "`r `r" -NoNewline +if ($missing.Count -ne 0) { Stop-WithError "Nodurile $($missing -join ', ') nu au aparut in ${NodeWait}s." } + +foreach ($node in $NodeIp.Keys) { + $w = 0 + while ($true) { + $h = Invoke-Node -Node $node -Command 'uptime -p' + if ($LASTEXITCODE -eq 0 -and $h) { break } + $w += 10 + if ($w -gt 300) { Stop-WithError "$node raspunde la ping dar nu la SSH." } + Start-Sleep -Seconds 10 + } + Write-Ok "$node - up, SSH OK ($h)" +} + +# --------------------------------------------------- pas 2: asteapta quorum + +Write-Step "Astept quorumul" + +$waited = 0 +while ($true) { + $q = Invoke-Node -Node 'pvemini' -Command 'pvecm status' + if (($q -join "`n") -match 'Quorate:\s*Yes') { break } + if ($waited -ge $QuorumWait) { Stop-WithError "Clusterul nu a atins quorumul in ${QuorumWait}s." } + Write-Host "`r astept quorum (${waited}s)... " -NoNewline -ForegroundColor DarkGray + Start-Sleep -Seconds 10 + $waited += 10 +} +Write-Host "`r `r" -NoNewline + +$votes = 0 +if (($q -join "`n") -match 'Total votes:\s*(\d+)') { $votes = [int]$Matches[1] } +if ($votes -ne 3) { Write-Warn "Quorum atins, dar doar $votes/3 voturi. Un nod lipseste din cluster." } +Write-Ok "quorum OK, $votes/3 voturi" + +Invoke-Node -Node 'pvemini' -Command 'ha-manager status' | Select-Object -First 4 | Write-Host + +# ------------------------------------------------- pas 3: ce trebuie pornit + +Write-Step "Ce se porneste" + +$toStart = @() + +if ((-not $All) -and (Test-Path $StateFile)) { + foreach ($line in (Get-Content $StateFile)) { + if ($line -match '^\s*#' -or $line -match '^\s*$') { continue } + $f = ($line -split '\s+') | Where-Object { $_ -ne '' } + if ($f.Count -ge 5) { + $toStart += [pscustomobject]@{ Type=$f[0]; Node=$f[1]; Id=[int]$f[2]; Status=$f[3]; Ha=$f[4] } + } + } + Write-Ok "citit din $StateFile ($($toStart.Count) guest-uri)" +} else { + if (-not $All) { Write-Warn "Nu exista $StateFile - folosesc lista implicita." } + + $haSids = @() + foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) { + if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" } + } + + $allGuests = @() + foreach ($node in $NodeIp.Keys) { + foreach ($g in (Get-NodeGuests -Node $node)) { + $sid = "$($g.Type):$($g.Id)" + $g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' }) + $allGuests += $g + } + } + + # Fara fisier de stare pornim doar ce e explicit in lista de ordine SI nu e + # in NeverAutostart. Altfel am porni DR-ul si template-urile, care erau + # oprite intentionat. + foreach ($id in $GuestStartOrder) { + if ($NeverAutostart -contains $id) { + Write-Warn "$id sarit (NeverAutostart) - porneste-l manual daca chiar il vrei" + continue + } + $g = $allGuests | Where-Object { $_.Id -eq $id } | Select-Object -First 1 + if ($g) { $toStart += $g } + } +} + +if ($toStart.Count -eq 0) { Stop-WithError "Nimic de pornit." } + +foreach ($g in $toStart) { + " {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host +} + +Confirm-Step "Se pornesc guest-urile de mai sus, Oracle primul." + +# ------------------------------------------------- pas 4: pornire guest-uri + +Write-Step "Pornire guest-uri" + +$failed = @() + +function Start-Guest { + param([pscustomobject]$Guest) + + $sid = "$($Guest.Type):$($Guest.Id)" + $cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' } + + if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'running') { + Write-Ok "$sid deja pornit" + return $true + } + + if ($Guest.Ha -eq 'ha') { + # Pentru resursele HA, --state started repune resursa sub controlul CRM. + Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state started" | Out-Null + } else { + Invoke-NodeChange -Node $Guest.Node -Command "$cmd start $($Guest.Id)" | Out-Null + } + + if ($DryRun) { return $true } + + $waited = 0 + while ($waited -lt $GuestWait) { + if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'running') { + Write-Ok "$sid pornit (${waited}s)" + return $true + } + Start-Sleep -Seconds 5 + $waited += 5 + } + Write-Warn "$sid nu a pornit in ${GuestWait}s - verifica manual" + return $false +} + +function Wait-Oracle { + param([string]$Node) + Write-Log "astept ca instanta Oracle sa accepte interogari..." + if ($DryRun) { return } + + $sql = 'pct exec ' + $OracleCt + ' -- docker exec ' + $OracleDocker + ' bash -c "printf ''set pagesize 0 feedback off\nselect global_name from global_name;\nexit\n'' | sqlplus -s / as sysdba"' + $waited = 0 + while ($waited -lt $OracleWait) { + $out = Invoke-Node -Node $Node -Command $sql + if ($LASTEXITCODE -eq 0 -and (($out -join '') -match '[A-Z]')) { + Write-Host "`r `r" -NoNewline + Write-Ok "Oracle deschis si interogabil (${waited}s)" + return + } + Write-Host "`r Oracle inca nu raspunde (${waited}s)... " -NoNewline -ForegroundColor DarkGray + Start-Sleep -Seconds 15 + $waited += 15 + } + Write-Host "`r `r" -NoNewline + Write-Warn "Oracle nu a raspuns in ${OracleWait}s. Continui, dar VERIFICA MANUAL." + Write-Warn " ssh root@$($NodeIp[$Node]) `"pct exec $OracleCt -- docker logs --tail 50 $OracleDocker`"" +} + +foreach ($id in $GuestStartOrder) { + $g = $toStart | Where-Object { $_.Id -eq $id } | Select-Object -First 1 + if (-not $g) { continue } + + Write-Log "pornesc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))" + if (-not (Start-Guest -Guest $g)) { $failed += "$($g.Type):$($g.Id)" } + + # Oracle e blocant: nimic care depinde de el nu porneste pana nu e deschis. + if ($g.Id -eq $OracleCt) { Wait-Oracle -Node $g.Node } +} + +# orice a mai ramas in lista si nu era in ordinea definita +foreach ($g in $toStart) { + if ($GuestStartOrder -contains $g.Id) { continue } + Write-Warn "guest neplanificat: $($g.Type):$($g.Id) pe $($g.Node) - il pornesc" + if (-not (Start-Guest -Guest $g)) { $failed += "$($g.Type):$($g.Id)" } +} + +# ------------------------------------------------ pas 5: restaurare cron-uri + +Write-Step "Restaurare cron-uri de monitorizare" + +foreach ($node in $NodeIp.Keys) { + $exists = Invoke-Node -Node $node -Command "test -f $CronBackup; echo `$?" + if (($exists -join '') -notmatch '^0') { + Write-Warn "$node - nu exista $CronBackup, sar peste" + continue + } + Invoke-NodeChange -Node $node -Command "crontab $CronBackup" | Out-Null + if (-not $DryRun) { + $left = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true" + if (($left -join '').Trim() -eq '0') { + Write-Ok "$node - cron-uri restaurate" + } else { + Write-Warn "$node - au ramas $left linii #MENTENANTA, verifica manual" + } + } +} + +# ------------------------------------------------------- pas 6: verificare + +Write-Step "Verificare finala" + +if ($DryRun) { + Write-Warn "dry-run - sar peste verificare" +} else { + Invoke-Node -Node 'pvemini' -Command 'ha-manager status' | Write-Host + Write-Host "" + $rep = Invoke-Node -Node 'pvemini' -Command 'pvesr status' + $rep | Write-Host + Write-Host "" + + $badRep = @($rep | Select-Object -Skip 1 | Where-Object { $_ -notmatch 'OK\s*$' -and $_.Trim() -ne '' }) + if ($badRep.Count -eq 0) { + Write-Ok "replicare: toate job-urile OK" + } else { + Write-Warn "job-uri de replicare cu probleme:" + $badRep | Write-Host + } +} + +Write-Step "Cluster pornit" + +if ($failed.Count -gt 0) { + Write-Warn "Guest-uri care NU au pornit: $($failed -join ', ')" + Write-Warn "Verifica-le manual inainte de a considera repornirea terminata." +} else { + Write-Ok "toate guest-urile au pornit" +} + +Write-Host "" +Write-Host " De verificat manual:" +Write-Host " - shutdown_policy e inca 'freeze' (recomandat sa ramana asa)" +Write-Host " - daca lipsesc date recente, compara cu ultima replicare din pvesr status" +Write-Host ""