<# .SYNOPSIS Oprire controlata a intregului cluster Proxmox `romfast`. .DESCRIPTION Opreste guest-urile in ordinea corecta de dependente, apoi nodurile, fara ca HA sa relocheze resursele si fara ca watchdog-ul sa reseteze hard vreun nod. Procedura completa si explicatiile: docs/oprire-planificata-cluster.md RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod Proxmox. Motivul: nodul care se opreste ultimul nu poate raporta rezultatul propriei opriri. Necesita clientul OpenSSH din Windows (C:\Windows\System32\OpenSSH\ssh.exe) si o cheie SSH acceptata de root pe toate cele trei noduri. .PARAMETER Yes Nu cere confirmari. .PARAMETER DryRun Arata ce ar face, fara sa execute nimic care modifica starea. .PARAMETER NoNodes Opreste doar guest-urile, lasa nodurile pornite. .EXAMPLE .\cluster-shutdown.ps1 -DryRun .EXAMPLE .\cluster-shutdown.ps1 #> [CmdletBinding()] param( [switch]$Yes, [switch]$DryRun, [switch]$NoNodes ) $ErrorActionPreference = 'Stop' # ---------------------------------------------------------------- configurare $NodeIp = [ordered]@{ pve1 = '10.0.20.200' pvemini = '10.0.20.201' pveelite = '10.0.20.202' } # Ordinea de OPRIRE a nodurilor. pvemini ultimul - e nodul principal si serverul NFS. $NodeShutdownOrder = @('pveelite', 'pve1', 'pvemini') # Ordinea de OPRIRE a guest-urilor: consumatorii intai, baza de date ULTIMA. # VM 201 (roacentral) si CT 104 (flowise) folosesc Oracle din CT 108. # Orice guest pornit care nu apare aici e oprit la final, intr-o maturare. $GuestShutdownOrder = @( 303, # Win11-Adina - desktop, fara dependente 304, # Win11-Marius - desktop, fara dependente 302, # oracle-test-302 - VM de test 310, # Win11-Template - template 201, # roacentral - IIS reverse proxy, consumator Oracle 101, # minecraft 110, # moltbot 104, # flowise - consumator Oracle 106, # gitea 103, # dokploy 102, # docker.romfast.ro 100, # portainer 171, # claude-agent 109, # oracle-dr-windows 301, # docker-portainer-template 108 # central-oracle - ULTIMUL ) $OracleCt = 108 $OracleDocker = 'oracle-xe' $GuestTimeout = 300 # secunde asteptate pentru shutdown-ul unui guest $NodeTimeout = 600 # secunde asteptate pana un nod nu mai raspunde la ping $CronBackup = '/root/crontab.backup-mentenanta.txt' # Cron-uri de dezactivat cat timp clusterul e jos (altfel mail storm, iar # vm109-watchdog.sh chiar porneste VM 109). $CronPattern = 'oom-alert|pvemini-down-alert|pveelite-down-alert|vm109-watchdog' $ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path $StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' } # ------------------------------------------------------------------- utilitare function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green } function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green } function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow } function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray } function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 } function Test-Alive { param([string]$Address, [int]$TimeoutMs = 2000) try { $p = New-Object System.Net.NetworkInformation.Ping return ($p.Send($Address, $TimeoutMs).Status -eq 'Success') } catch { return $false } } # Executa o comanda pe un nod. Intoarce liniile de output. # NU redirectam stderr: in PS 5.1 asta transforma fiecare linie intr-un # ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit. function Invoke-Node { param([string]$Node, [string]$Command) $out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command return $out } # Comenzi cu ghilimele imbricate NU se trimit prin Invoke-Node. # # PowerShell 5.1 pierde ghilimelele duble cand paseaza un argument catre un # executabil nativ: `bash -lc "printf '...' | sqlplus"` ajunge pe nod ca # `bash -lc printf '...' | sqlplus`, adica printf fara format, iar sqlplus rulat # pe NOD, unde nu exista. Esecul e usor de citit gresit ca "lipseste sqlplus". # In loc sa ne luptam cu nivelurile de citare, trimitem scriptul codificat. # Verificat pe 2026-08-29, dupa ce varianta directa a esuat pe cluster live. function Invoke-NodeScript { param([string]$Node, [string]$Script) $lf = $Script -replace "`r`n", "`n" $b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($lf)) return Invoke-Node -Node $Node -Command "echo $b64 | base64 -d | bash" } # Ca Invoke-Node, dar respecta -DryRun (pentru comenzi care modifica starea). function Invoke-NodeChange { param([string]$Node, [string]$Command) if ($DryRun) { Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray return @() } return Invoke-Node -Node $Node -Command $Command } function Confirm-Step { param([string]$Message) if ($Yes -or $DryRun) { return } Write-Host "" Write-Host $Message -ForegroundColor Yellow $reply = Read-Host "Scrie DA ca sa continui" if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." } } # Parseaza `pct list` / `qm list` in obiecte. function Get-NodeGuests { param([string]$Node) $result = @() $ctLines = Invoke-Node -Node $Node -Command 'pct list' foreach ($line in ($ctLines | Select-Object -Skip 1)) { $f = ($line -split '\s+') | Where-Object { $_ -ne '' } if ($f.Count -ge 2) { $result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] } } } $vmLines = Invoke-Node -Node $Node -Command 'qm list' foreach ($line in ($vmLines | Select-Object -Skip 1)) { $f = ($line -split '\s+') | Where-Object { $_ -ne '' } if ($f.Count -ge 3) { $result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] } } } return $result } function Get-GuestStatus { param([string]$Node, [string]$Type, [int]$Id) $cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" } $out = Invoke-Node -Node $Node -Command $cmd if ($out -match 'status:\s*(\w+)') { return $Matches[1] } return 'unknown' } # ------------------------------------------------------------ pas 1: preflight Write-Step "Preflight" foreach ($node in $NodeIp.Keys) { if (-not (Test-Alive $NodeIp[$node])) { Stop-WithError "$node ($($NodeIp[$node])) nu raspunde la ping." } $h = Invoke-Node -Node $node -Command 'hostname' if ($LASTEXITCODE -ne 0 -or -not $h) { Stop-WithError "$node nu accepta SSH cu cheie." } Write-Ok "$node ($($NodeIp[$node])) - reachable" } $quorum = Invoke-Node -Node 'pvemini' -Command 'pvecm status' if (($quorum -join "`n") -notmatch 'Quorate:\s*Yes') { Stop-WithError "Clusterul NU are quorum. Opreste-te si investigheaza." } $haStatus = Invoke-Node -Node 'pvemini' -Command 'ha-manager status' $master = ($haStatus | Where-Object { $_ -match '^master\s+(\S+)' } | ForEach-Object { $Matches[1] }) Write-Ok "quorum OK, HA master: $master" $runningTasks = 0 foreach ($node in $NodeIp.Keys) { $t = Invoke-Node -Node $node -Command "pvesh get /nodes/$node/tasks --limit 20 --output-format json" $runningTasks += ([regex]::Matches(($t -join ''), '"status":"running"')).Count } if ($runningTasks -gt 0) { Write-Warn "$runningTasks task-uri in executie pe cluster (backup? migrare?)." Confirm-Step "Sunt task-uri active. Continui oricum?" } else { Write-Ok "niciun task in executie" } # ------------------------------------------- pas 2: inventar + salvare stare Write-Step "Inventar guest-uri" $haSids = @() foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) { if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" } } $inventory = @() foreach ($node in $NodeIp.Keys) { foreach ($g in (Get-NodeGuests -Node $node)) { $sid = "$($g.Type):$($g.Id)" $g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' }) $inventory += $g } } $running = @($inventory | Where-Object { $_.Status -eq 'running' }) if ($running.Count -eq 0) { Write-Warn "Niciun guest pornit. Trec direct la oprirea nodurilor." } else { foreach ($g in $running) { " {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host } Write-Ok "$($running.Count) guest-uri pornite" } # Salveaza starea LOCAL - cluster-startup porneste exact ce era pornit, ca sa nu # reporneasca VM-uri oprite intentionat (302, 310, 301...). # Format identic cu varianta bash, ca cele doua sa fie interschimbabile. if (-not $DryRun) { # Nu rescrie peste o stare existenta cu un inventar gol. Cazul real: scriptul # e rulat a doua oara (dupa -NoNodes, sau dupa o eroare), cand guest-urile sunt # deja oprite - inventarul e gol si am pierde exact lista de repornit. $existing = @() if (Test-Path $StateFile) { $existing = @(Get-Content $StateFile | Where-Object { $_ -notmatch '^\s*#' -and $_.Trim() }) } if ($running.Count -eq 0 -and $existing.Count -gt 0) { Write-Warn "Niciun guest pornit, dar $StateFile contine deja $($existing.Count) intrari - il PASTREZ." Write-Warn "Altfel cluster-startup n-ar mai sti ce sa porneasca." } else { $lines = @("# stare cluster salvata la $(Get-Date -Format 'yyyy-MM-dd HH:mm:ss')") foreach ($g in $running) { $lines += "$($g.Type) $($g.Node) $($g.Id) $($g.Status) $($g.Ha)" } Set-Content -Path $StateFile -Value $lines -Encoding utf8 Write-Ok "stare salvata in $StateFile" } } Confirm-Step "Se opresc $($running.Count) guest-uri, apoi nodurile: $($NodeShutdownOrder -join ', ')." # -------------------------------------------- pas 3: shutdown_policy=freeze Write-Step "Shutdown policy -> freeze" $dcCfg = Invoke-Node -Node 'pvemini' -Command 'cat /etc/pve/datacenter.cfg 2>/dev/null || true' $currentPolicy = $null if (($dcCfg -join "`n") -match 'shutdown_policy=(\w+)') { $currentPolicy = $Matches[1] } if ($currentPolicy -eq 'freeze') { Write-Ok "deja pe freeze" } else { if ($null -ne $currentPolicy -or ($dcCfg | Measure-Object).Count -gt 0) { # sed sterge DOAR linia 'ha:', nu rescrie fisierul: din 2026-08-29 acolo # sta si 'migration: network=10.10.10.0/24,type=insecure', fara de care # replicarea se intoarce tacut pe reteaua de productie. Invoke-NodeChange -Node 'pvemini' -Command 'cp /etc/pve/datacenter.cfg /root/datacenter.cfg.backup-mentenanta' | Out-Null Invoke-NodeChange -Node 'pvemini' -Command "sed -i '/^ha:/d' /etc/pve/datacenter.cfg; printf 'ha: shutdown_policy=freeze\n' >> /etc/pve/datacenter.cfg" | Out-Null } else { Invoke-NodeChange -Node 'pvemini' -Command "printf 'ha: shutdown_policy=freeze\n' > /etc/pve/datacenter.cfg" | Out-Null } $was = if ($currentPolicy) { $currentPolicy } else { 'nesetat' } Write-Ok "setat freeze (era: $was)" } # ---------------------------------------- pas 4: dezactivare cron-uri alerta Write-Step "Dezactivare cron-uri de alerta" $sedCron = "crontab -l | sed -E '/^[^#]/ s%^(.*($CronPattern)\.sh.*)`$%#MENTENANTA \1%' | crontab -" foreach ($node in $NodeIp.Keys) { # Backup DOAR daca crontab-ul curent nu e deja comentat. Altfel o a doua rulare # a scriptului (dupa o eroare, de exemplu) ar salva peste backup versiunea deja # comentata, iar cluster-startup ar "restaura" monitorizarea oprita - tacut. Invoke-NodeChange -Node $node -Command "if crontab -l 2>/dev/null | grep -q '^#MENTENANTA'; then echo 'backup pastrat'; else crontab -l > $CronBackup 2>/dev/null || true; echo 'backup nou'; fi" | Out-Null Invoke-NodeChange -Node $node -Command $sedCron | Out-Null if (-not $DryRun) { $n = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true" Write-Ok "$node - $n linii dezactivate, backup in $CronBackup" } } Write-Warn "Monitorizarea e acum OPRITA pe tot clusterul. cluster-startup o restaureaza." # --------------------------------------------- pas 5: shutdown curat Oracle Write-Step "Oracle - shutdown immediate" $oracleGuest = $running | Where-Object { $_.Type -eq 'ct' -and $_.Id -eq $OracleCt } | Select-Object -First 1 if (-not $oracleGuest) { Write-Warn "CT $OracleCt nu ruleaza - sar peste shutdown-ul bazei." } else { Write-Log "CT $OracleCt pe $($oracleGuest.Node) - opresc instanta..." if ($DryRun) { Write-Host " [dry-run] shutdown immediate in $OracleDocker" -ForegroundColor DarkGray } else { # 'bash -lc', NU 'bash -c': fara shell de login, PATH-ul Oracle nu e incarcat # si comanda esueaza cu "sqlplus: command not found" - adica baza ramane # DESCHISA, iar containerul ar fi oprit peste ea. # '2>&1' se face pe nod, nu in PowerShell: altfel stderr-ul lui ssh devine # ErrorRecord in PS 5.1 si arunca exceptie in loc sa fie tratat aici. # 'timeout 180': un 'shutdown immediate' poate dura, dar nu la infinit. Fara # el, un sqlplus blocat ar tine scriptul agatat fara niciun mesaj (patit pe # 2026-08-29, la sonda echivalenta din cluster-startup). $sqlScript = @" timeout 180 pct exec $OracleCt -- docker exec $OracleDocker bash -lc "printf 'shutdown immediate\nexit\n' | sqlplus -s / as sysdba" 2>&1 "@ $out = Invoke-NodeScript -Node $oracleGuest.Node -Script $sqlScript # Codul de iesire nu e o dovada: sqlplus intoarce 0 si cand n-a inchis nimic. # Singura confirmare reala e mesajul instantei. if (($out -join "`n") -match 'ORACLE instance shut down') { Write-Ok "instanta Oracle oprita curat" } else { Write-Warn "Baza NU a confirmat 'ORACLE instance shut down'. Ultimele linii:" $out | Select-Object -Last 8 | Write-Host Confirm-Step "Oprirea CT $OracleCt peste o baza posibil deschisa. Continui?" } } } # ------------------------------------------------ pas 6: oprire guest-uri Write-Step "Oprire guest-uri" function Stop-Guest { param([pscustomobject]$Guest) $sid = "$($Guest.Type):$($Guest.Id)" $cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' } if ($Guest.Ha -eq 'ha') { # Pentru resursele HA se foloseste ha-manager, NU pct/qm shutdown: din CLI # acestea nu actualizeaza state-ul HA, iar CRM-ul ar reporni guest-ul. Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state stopped" | Out-Null } else { Invoke-NodeChange -Node $Guest.Node -Command "$cmd shutdown $($Guest.Id) --timeout $GuestTimeout" | Out-Null } if ($DryRun) { return } $waited = 0 while ($waited -lt $GuestTimeout) { if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'stopped') { Write-Ok "$sid oprit (${waited}s)" return } Start-Sleep -Seconds 5 $waited += 5 } Write-Warn "$sid nu s-a oprit in ${GuestTimeout}s - fortez stop" Invoke-Node -Node $Guest.Node -Command "$cmd stop $($Guest.Id)" | Out-Null Start-Sleep -Seconds 5 if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -ne 'stopped') { Stop-WithError "$sid REFUZA sa se opreasca. Rezolva manual inainte de a opri nodurile." } Write-Ok "$sid oprit fortat" } # intai in ordinea definita... foreach ($id in $GuestShutdownOrder) { $g = $running | Where-Object { $_.Id -eq $id } | Select-Object -First 1 if (-not $g) { continue } Write-Log "opresc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))" Stop-Guest -Guest $g } # ...apoi orice a mai ramas pornit si nu era in lista foreach ($g in $running) { if ($GuestShutdownOrder -contains $g.Id) { continue } Write-Warn "guest neplanificat inca pornit: $($g.Type):$($g.Id) pe $($g.Node) - il opresc" Stop-Guest -Guest $g } # ------------------------------------------- pas 7: verificare obligatorie Write-Step "Verificare inainte de oprirea nodurilor" if ($DryRun) { Write-Warn "dry-run - sar peste verificare" } else { $ha = Invoke-Node -Node 'pvemini' -Command 'ha-manager status' $bad = @($ha | Where-Object { $_ -match 'started|migrate|relocate|fence' }) if ($bad.Count -gt 0) { $bad | Write-Host Stop-WithError "Exista servicii HA inca active. NU opri nodurile." } Write-Ok "toate serviciile HA sunt in state stopped" $still = 0 foreach ($node in $NodeIp.Keys) { $still += @((Get-NodeGuests -Node $node) | Where-Object { $_.Status -eq 'running' }).Count } if ($still -ne 0) { Stop-WithError "$still guest-uri inca pornite. NU opri nodurile." } Write-Ok "0 guest-uri pornite pe cluster" Write-Ok "watchdog-ul nu mai e armat - pierderea quorumului e inofensiva" } if ($NoNodes) { Write-Step "-NoNodes: nodurile raman pornite. Gata." exit 0 } Confirm-Step "Se opresc nodurile in ordinea: $($NodeShutdownOrder -join ', ') (pvemini ultimul)." # ------------------------------------------------- pas 8: oprire noduri Write-Step "Oprire noduri" foreach ($node in $NodeShutdownOrder) { Write-Log "opresc $node ($($NodeIp[$node]))..." Invoke-NodeChange -Node $node -Command 'systemctl poweroff --no-block' | Out-Null if ($DryRun) { continue } $waited = 0 $down = $false while ($waited -lt $NodeTimeout) { if (-not (Test-Alive $NodeIp[$node])) { Write-Ok "$node OPRIT (${waited}s)" $down = $true break } Start-Sleep -Seconds 10 $waited += 10 } if (-not $down) { Write-Warn "$node inca raspunde la ping dupa ${NodeTimeout}s." Write-Warn "Verifica manual - sshd poate fi deja jos desi reteaua e sus." } } Write-Step "Cluster oprit" Write-Host "" Write-Host " Stare salvata: $StateFile" Write-Host " Backup crontab: $CronBackup (pe fiecare nod)" Write-Host " Shutdown policy: freeze" Write-Host "" Write-Host " La revenirea curentului: .\cluster-startup.ps1" Write-Host ""