Files
ROMFASTSQL/proxmox/cluster/scripts/cluster-shutdown.ps1
Marius 730ce94fac feat(cluster): varianta PowerShell a scripturilor de oprire/repornire
Statia de admin e Windows si nu ruleaza .sh direct: `bash` din PATH e cel din
WSL, cu alt filesystem si alta configuratie de chei SSH. Adaug echivalentele
native PowerShell.

- cluster-shutdown.ps1 / cluster-startup.ps1, aceeasi logica si aceleasi
  garantii ca variantele bash
- folosesc clientul OpenSSH din Windows, deja prezent in System32
- compatibile PowerShell 5.1: fara &&/||, fara ternar, fara ?., fara
  -AsHashtable; ping prin System.Net.NetworkInformation.Ping, nu
  Test-Connection (WMI, lent si des blocat)
- nu redirecteaza stderr-ul lui ssh: in 5.1 asta transforma fiecare linie
  intr-un ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit
- parsarea `pct list` / `qm list` se face in PowerShell, nu prin awk remote,
  ca sa nu se incurce interpolarea $1/$2 din stringurile PowerShell
- scriu acelasi format de fisier de stare ca variantele bash, deci poti opri
  cu una si porni cu cealalta

Ambele testate cu -DryRun pe clusterul live, rezultat identic cu bash.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RRyaDj39hPQ89SZS6URRpS
2026-08-27 14:04:50 +03:00

418 lines
15 KiB
PowerShell

<#
.SYNOPSIS
Oprire controlata a intregului cluster Proxmox `romfast`.
.DESCRIPTION
Opreste guest-urile in ordinea corecta de dependente, apoi nodurile, fara ca
HA sa relocheze resursele si fara ca watchdog-ul sa reseteze hard vreun nod.
Procedura completa si explicatiile: docs/oprire-planificata-cluster.md
RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod Proxmox. Motivul: nodul care
se opreste ultimul nu poate raporta rezultatul propriei opriri.
Necesita clientul OpenSSH din Windows (C:\Windows\System32\OpenSSH\ssh.exe)
si o cheie SSH acceptata de root pe toate cele trei noduri.
.PARAMETER Yes
Nu cere confirmari.
.PARAMETER DryRun
Arata ce ar face, fara sa execute nimic care modifica starea.
.PARAMETER NoNodes
Opreste doar guest-urile, lasa nodurile pornite.
.EXAMPLE
.\cluster-shutdown.ps1 -DryRun
.EXAMPLE
.\cluster-shutdown.ps1
#>
[CmdletBinding()]
param(
[switch]$Yes,
[switch]$DryRun,
[switch]$NoNodes
)
$ErrorActionPreference = 'Stop'
# ---------------------------------------------------------------- configurare
$NodeIp = [ordered]@{
pve1 = '10.0.20.200'
pvemini = '10.0.20.201'
pveelite = '10.0.20.202'
}
# Ordinea de OPRIRE a nodurilor. pvemini ultimul - e nodul principal si serverul NFS.
$NodeShutdownOrder = @('pveelite', 'pve1', 'pvemini')
# Ordinea de OPRIRE a guest-urilor: consumatorii intai, baza de date ULTIMA.
# VM 201 (roacentral) si CT 104 (flowise) folosesc Oracle din CT 108.
# Orice guest pornit care nu apare aici e oprit la final, intr-o maturare.
$GuestShutdownOrder = @(
303, # Win11-Adina - desktop, fara dependente
302, # oracle-test-302 - VM de test
310, # Win11-Template - template
201, # roacentral - IIS reverse proxy, consumator Oracle
101, # minecraft
110, # moltbot
104, # flowise - consumator Oracle
106, # gitea
103, # dokploy
102, # docker.romfast.ro
100, # portainer
171, # claude-agent
109, # oracle-dr-windows
301, # docker-portainer-template
108 # central-oracle - ULTIMUL
)
$OracleCt = 108
$OracleDocker = 'oracle-xe'
$GuestTimeout = 300 # secunde asteptate pentru shutdown-ul unui guest
$NodeTimeout = 600 # secunde asteptate pana un nod nu mai raspunde la ping
$CronBackup = '/root/crontab.backup-mentenanta.txt'
# Cron-uri de dezactivat cat timp clusterul e jos (altfel mail storm, iar
# vm109-watchdog.sh chiar porneste VM 109).
$CronPattern = 'oom-alert|pvemini-down-alert|pveelite-down-alert|vm109-watchdog'
$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path
$StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' }
# ------------------------------------------------------------------- utilitare
function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green }
function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green }
function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow }
function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray }
function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 }
function Test-Alive {
param([string]$Address, [int]$TimeoutMs = 2000)
try {
$p = New-Object System.Net.NetworkInformation.Ping
return ($p.Send($Address, $TimeoutMs).Status -eq 'Success')
} catch {
return $false
}
}
# Executa o comanda pe un nod. Intoarce liniile de output.
# NU redirectam stderr: in PS 5.1 asta transforma fiecare linie intr-un
# ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit.
function Invoke-Node {
param([string]$Node, [string]$Command)
$out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command
return $out
}
# Ca Invoke-Node, dar respecta -DryRun (pentru comenzi care modifica starea).
function Invoke-NodeChange {
param([string]$Node, [string]$Command)
if ($DryRun) {
Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray
return @()
}
return Invoke-Node -Node $Node -Command $Command
}
function Confirm-Step {
param([string]$Message)
if ($Yes -or $DryRun) { return }
Write-Host ""
Write-Host $Message -ForegroundColor Yellow
$reply = Read-Host "Scrie DA ca sa continui"
if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." }
}
# Parseaza `pct list` / `qm list` in obiecte.
function Get-NodeGuests {
param([string]$Node)
$result = @()
$ctLines = Invoke-Node -Node $Node -Command 'pct list'
foreach ($line in ($ctLines | Select-Object -Skip 1)) {
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
if ($f.Count -ge 2) {
$result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] }
}
}
$vmLines = Invoke-Node -Node $Node -Command 'qm list'
foreach ($line in ($vmLines | Select-Object -Skip 1)) {
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
if ($f.Count -ge 3) {
$result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] }
}
}
return $result
}
function Get-GuestStatus {
param([string]$Node, [string]$Type, [int]$Id)
$cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" }
$out = Invoke-Node -Node $Node -Command $cmd
if ($out -match 'status:\s*(\w+)') { return $Matches[1] }
return 'unknown'
}
# ------------------------------------------------------------ pas 1: preflight
Write-Step "Preflight"
foreach ($node in $NodeIp.Keys) {
if (-not (Test-Alive $NodeIp[$node])) {
Stop-WithError "$node ($($NodeIp[$node])) nu raspunde la ping."
}
$h = Invoke-Node -Node $node -Command 'hostname'
if ($LASTEXITCODE -ne 0 -or -not $h) { Stop-WithError "$node nu accepta SSH cu cheie." }
Write-Ok "$node ($($NodeIp[$node])) - reachable"
}
$quorum = Invoke-Node -Node 'pvemini' -Command 'pvecm status'
if (($quorum -join "`n") -notmatch 'Quorate:\s*Yes') {
Stop-WithError "Clusterul NU are quorum. Opreste-te si investigheaza."
}
$haStatus = Invoke-Node -Node 'pvemini' -Command 'ha-manager status'
$master = ($haStatus | Where-Object { $_ -match '^master\s+(\S+)' } | ForEach-Object { $Matches[1] })
Write-Ok "quorum OK, HA master: $master"
$runningTasks = 0
foreach ($node in $NodeIp.Keys) {
$t = Invoke-Node -Node $node -Command "pvesh get /nodes/$node/tasks --limit 20 --output-format json"
$runningTasks += ([regex]::Matches(($t -join ''), '"status":"running"')).Count
}
if ($runningTasks -gt 0) {
Write-Warn "$runningTasks task-uri in executie pe cluster (backup? migrare?)."
Confirm-Step "Sunt task-uri active. Continui oricum?"
} else {
Write-Ok "niciun task in executie"
}
# ------------------------------------------- pas 2: inventar + salvare stare
Write-Step "Inventar guest-uri"
$haSids = @()
foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) {
if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" }
}
$inventory = @()
foreach ($node in $NodeIp.Keys) {
foreach ($g in (Get-NodeGuests -Node $node)) {
$sid = "$($g.Type):$($g.Id)"
$g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' })
$inventory += $g
}
}
$running = @($inventory | Where-Object { $_.Status -eq 'running' })
if ($running.Count -eq 0) {
Write-Warn "Niciun guest pornit. Trec direct la oprirea nodurilor."
} else {
foreach ($g in $running) {
" {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host
}
Write-Ok "$($running.Count) guest-uri pornite"
}
# Salveaza starea LOCAL - cluster-startup porneste exact ce era pornit, ca sa nu
# reporneasca VM-uri oprite intentionat (302, 310, 301...).
# Format identic cu varianta bash, ca cele doua sa fie interschimbabile.
if (-not $DryRun) {
$lines = @("# stare cluster salvata la $(Get-Date -Format 'yyyy-MM-dd HH:mm:ss')")
foreach ($g in $running) { $lines += "$($g.Type) $($g.Node) $($g.Id) $($g.Status) $($g.Ha)" }
Set-Content -Path $StateFile -Value $lines -Encoding utf8
Write-Ok "stare salvata in $StateFile"
}
Confirm-Step "Se opresc $($running.Count) guest-uri, apoi nodurile: $($NodeShutdownOrder -join ', ')."
# -------------------------------------------- pas 3: shutdown_policy=freeze
Write-Step "Shutdown policy -> freeze"
$dcCfg = Invoke-Node -Node 'pvemini' -Command 'cat /etc/pve/datacenter.cfg 2>/dev/null || true'
$currentPolicy = $null
if (($dcCfg -join "`n") -match 'shutdown_policy=(\w+)') { $currentPolicy = $Matches[1] }
if ($currentPolicy -eq 'freeze') {
Write-Ok "deja pe freeze"
} else {
if ($null -ne $currentPolicy -or ($dcCfg | Measure-Object).Count -gt 0) {
Invoke-NodeChange -Node 'pvemini' -Command 'cp /etc/pve/datacenter.cfg /root/datacenter.cfg.backup-mentenanta' | Out-Null
Invoke-NodeChange -Node 'pvemini' -Command "sed -i '/^ha:/d' /etc/pve/datacenter.cfg; printf 'ha: shutdown_policy=freeze\n' >> /etc/pve/datacenter.cfg" | Out-Null
} else {
Invoke-NodeChange -Node 'pvemini' -Command "printf 'ha: shutdown_policy=freeze\n' > /etc/pve/datacenter.cfg" | Out-Null
}
$was = if ($currentPolicy) { $currentPolicy } else { 'nesetat' }
Write-Ok "setat freeze (era: $was)"
}
# ---------------------------------------- pas 4: dezactivare cron-uri alerta
Write-Step "Dezactivare cron-uri de alerta"
$sedCron = "crontab -l | sed -E '/^[^#]/ s%^(.*($CronPattern)\.sh.*)`$%#MENTENANTA \1%' | crontab -"
foreach ($node in $NodeIp.Keys) {
Invoke-NodeChange -Node $node -Command "crontab -l > $CronBackup 2>/dev/null || true" | Out-Null
Invoke-NodeChange -Node $node -Command $sedCron | Out-Null
if (-not $DryRun) {
$n = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true"
Write-Ok "$node - $n linii dezactivate, backup in $CronBackup"
}
}
Write-Warn "Monitorizarea e acum OPRITA pe tot clusterul. cluster-startup o restaureaza."
# --------------------------------------------- pas 5: shutdown curat Oracle
Write-Step "Oracle - shutdown immediate"
$oracleGuest = $running | Where-Object { $_.Type -eq 'ct' -and $_.Id -eq $OracleCt } | Select-Object -First 1
if (-not $oracleGuest) {
Write-Warn "CT $OracleCt nu ruleaza - sar peste shutdown-ul bazei."
} else {
Write-Log "CT $OracleCt pe $($oracleGuest.Node) - opresc instanta..."
if ($DryRun) {
Write-Host " [dry-run] shutdown immediate in $OracleDocker" -ForegroundColor DarkGray
} else {
$sqlCmd = 'pct exec ' + $OracleCt + ' -- docker exec ' + $OracleDocker + ' bash -c "printf ''shutdown immediate\nexit\n'' | sqlplus -s / as sysdba"'
$out = Invoke-Node -Node $oracleGuest.Node -Command $sqlCmd
if ($LASTEXITCODE -ne 0) {
Write-Warn "Shutdown-ul bazei a raportat eroare. Verifica manual inainte de a continua!"
$out | Select-Object -Last 5 | Write-Host
} else {
Write-Ok "instanta Oracle oprita"
}
}
}
# ------------------------------------------------ pas 6: oprire guest-uri
Write-Step "Oprire guest-uri"
function Stop-Guest {
param([pscustomobject]$Guest)
$sid = "$($Guest.Type):$($Guest.Id)"
$cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' }
if ($Guest.Ha -eq 'ha') {
# Pentru resursele HA se foloseste ha-manager, NU pct/qm shutdown: din CLI
# acestea nu actualizeaza state-ul HA, iar CRM-ul ar reporni guest-ul.
Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state stopped" | Out-Null
} else {
Invoke-NodeChange -Node $Guest.Node -Command "$cmd shutdown $($Guest.Id) --timeout $GuestTimeout" | Out-Null
}
if ($DryRun) { return }
$waited = 0
while ($waited -lt $GuestTimeout) {
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'stopped') {
Write-Ok "$sid oprit (${waited}s)"
return
}
Start-Sleep -Seconds 5
$waited += 5
}
Write-Warn "$sid nu s-a oprit in ${GuestTimeout}s - fortez stop"
Invoke-Node -Node $Guest.Node -Command "$cmd stop $($Guest.Id)" | Out-Null
Start-Sleep -Seconds 5
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -ne 'stopped') {
Stop-WithError "$sid REFUZA sa se opreasca. Rezolva manual inainte de a opri nodurile."
}
Write-Ok "$sid oprit fortat"
}
# intai in ordinea definita...
foreach ($id in $GuestShutdownOrder) {
$g = $running | Where-Object { $_.Id -eq $id } | Select-Object -First 1
if (-not $g) { continue }
Write-Log "opresc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))"
Stop-Guest -Guest $g
}
# ...apoi orice a mai ramas pornit si nu era in lista
foreach ($g in $running) {
if ($GuestShutdownOrder -contains $g.Id) { continue }
Write-Warn "guest neplanificat inca pornit: $($g.Type):$($g.Id) pe $($g.Node) - il opresc"
Stop-Guest -Guest $g
}
# ------------------------------------------- pas 7: verificare obligatorie
Write-Step "Verificare inainte de oprirea nodurilor"
if ($DryRun) {
Write-Warn "dry-run - sar peste verificare"
} else {
$ha = Invoke-Node -Node 'pvemini' -Command 'ha-manager status'
$bad = @($ha | Where-Object { $_ -match 'started|migrate|relocate|fence' })
if ($bad.Count -gt 0) {
$bad | Write-Host
Stop-WithError "Exista servicii HA inca active. NU opri nodurile."
}
Write-Ok "toate serviciile HA sunt in state stopped"
$still = 0
foreach ($node in $NodeIp.Keys) {
$still += @((Get-NodeGuests -Node $node) | Where-Object { $_.Status -eq 'running' }).Count
}
if ($still -ne 0) { Stop-WithError "$still guest-uri inca pornite. NU opri nodurile." }
Write-Ok "0 guest-uri pornite pe cluster"
Write-Ok "watchdog-ul nu mai e armat - pierderea quorumului e inofensiva"
}
if ($NoNodes) {
Write-Step "-NoNodes: nodurile raman pornite. Gata."
exit 0
}
Confirm-Step "Se opresc nodurile in ordinea: $($NodeShutdownOrder -join ', ') (pvemini ultimul)."
# ------------------------------------------------- pas 8: oprire noduri
Write-Step "Oprire noduri"
foreach ($node in $NodeShutdownOrder) {
Write-Log "opresc $node ($($NodeIp[$node]))..."
Invoke-NodeChange -Node $node -Command 'systemctl poweroff --no-block' | Out-Null
if ($DryRun) { continue }
$waited = 0
$down = $false
while ($waited -lt $NodeTimeout) {
if (-not (Test-Alive $NodeIp[$node])) {
Write-Ok "$node OPRIT (${waited}s)"
$down = $true
break
}
Start-Sleep -Seconds 10
$waited += 10
}
if (-not $down) {
Write-Warn "$node inca raspunde la ping dupa ${NodeTimeout}s."
Write-Warn "Verifica manual - sshd poate fi deja jos desi reteaua e sus."
}
}
Write-Step "Cluster oprit"
Write-Host ""
Write-Host " Stare salvata: $StateFile"
Write-Host " Backup crontab: $CronBackup (pe fiecare nod)"
Write-Host " Shutdown policy: freeze"
Write-Host ""
Write-Host " La revenirea curentului: .\cluster-startup.ps1"
Write-Host ""