feat(cluster): varianta PowerShell a scripturilor de oprire/repornire
Statia de admin e Windows si nu ruleaza .sh direct: `bash` din PATH e cel din WSL, cu alt filesystem si alta configuratie de chei SSH. Adaug echivalentele native PowerShell. - cluster-shutdown.ps1 / cluster-startup.ps1, aceeasi logica si aceleasi garantii ca variantele bash - folosesc clientul OpenSSH din Windows, deja prezent in System32 - compatibile PowerShell 5.1: fara &&/||, fara ternar, fara ?., fara -AsHashtable; ping prin System.Net.NetworkInformation.Ping, nu Test-Connection (WMI, lent si des blocat) - nu redirecteaza stderr-ul lui ssh: in 5.1 asta transforma fiecare linie intr-un ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit - parsarea `pct list` / `qm list` se face in PowerShell, nu prin awk remote, ca sa nu se incurce interpolarea $1/$2 din stringurile PowerShell - scriu acelasi format de fisier de stare ca variantele bash, deci poti opri cu una si porni cu cealalta Ambele testate cu -DryRun pe clusterul live, rezultat identic cu bash. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RRyaDj39hPQ89SZS6URRpS
This commit is contained in:
417
proxmox/cluster/scripts/cluster-shutdown.ps1
Normal file
417
proxmox/cluster/scripts/cluster-shutdown.ps1
Normal file
@@ -0,0 +1,417 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Oprire controlata a intregului cluster Proxmox `romfast`.
|
||||
|
||||
.DESCRIPTION
|
||||
Opreste guest-urile in ordinea corecta de dependente, apoi nodurile, fara ca
|
||||
HA sa relocheze resursele si fara ca watchdog-ul sa reseteze hard vreun nod.
|
||||
|
||||
Procedura completa si explicatiile: docs/oprire-planificata-cluster.md
|
||||
|
||||
RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod Proxmox. Motivul: nodul care
|
||||
se opreste ultimul nu poate raporta rezultatul propriei opriri.
|
||||
|
||||
Necesita clientul OpenSSH din Windows (C:\Windows\System32\OpenSSH\ssh.exe)
|
||||
si o cheie SSH acceptata de root pe toate cele trei noduri.
|
||||
|
||||
.PARAMETER Yes
|
||||
Nu cere confirmari.
|
||||
|
||||
.PARAMETER DryRun
|
||||
Arata ce ar face, fara sa execute nimic care modifica starea.
|
||||
|
||||
.PARAMETER NoNodes
|
||||
Opreste doar guest-urile, lasa nodurile pornite.
|
||||
|
||||
.EXAMPLE
|
||||
.\cluster-shutdown.ps1 -DryRun
|
||||
.EXAMPLE
|
||||
.\cluster-shutdown.ps1
|
||||
#>
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[switch]$Yes,
|
||||
[switch]$DryRun,
|
||||
[switch]$NoNodes
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
# ---------------------------------------------------------------- configurare
|
||||
|
||||
$NodeIp = [ordered]@{
|
||||
pve1 = '10.0.20.200'
|
||||
pvemini = '10.0.20.201'
|
||||
pveelite = '10.0.20.202'
|
||||
}
|
||||
|
||||
# Ordinea de OPRIRE a nodurilor. pvemini ultimul - e nodul principal si serverul NFS.
|
||||
$NodeShutdownOrder = @('pveelite', 'pve1', 'pvemini')
|
||||
|
||||
# Ordinea de OPRIRE a guest-urilor: consumatorii intai, baza de date ULTIMA.
|
||||
# VM 201 (roacentral) si CT 104 (flowise) folosesc Oracle din CT 108.
|
||||
# Orice guest pornit care nu apare aici e oprit la final, intr-o maturare.
|
||||
$GuestShutdownOrder = @(
|
||||
303, # Win11-Adina - desktop, fara dependente
|
||||
302, # oracle-test-302 - VM de test
|
||||
310, # Win11-Template - template
|
||||
201, # roacentral - IIS reverse proxy, consumator Oracle
|
||||
101, # minecraft
|
||||
110, # moltbot
|
||||
104, # flowise - consumator Oracle
|
||||
106, # gitea
|
||||
103, # dokploy
|
||||
102, # docker.romfast.ro
|
||||
100, # portainer
|
||||
171, # claude-agent
|
||||
109, # oracle-dr-windows
|
||||
301, # docker-portainer-template
|
||||
108 # central-oracle - ULTIMUL
|
||||
)
|
||||
|
||||
$OracleCt = 108
|
||||
$OracleDocker = 'oracle-xe'
|
||||
|
||||
$GuestTimeout = 300 # secunde asteptate pentru shutdown-ul unui guest
|
||||
$NodeTimeout = 600 # secunde asteptate pana un nod nu mai raspunde la ping
|
||||
$CronBackup = '/root/crontab.backup-mentenanta.txt'
|
||||
|
||||
# Cron-uri de dezactivat cat timp clusterul e jos (altfel mail storm, iar
|
||||
# vm109-watchdog.sh chiar porneste VM 109).
|
||||
$CronPattern = 'oom-alert|pvemini-down-alert|pveelite-down-alert|vm109-watchdog'
|
||||
|
||||
$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path
|
||||
$StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' }
|
||||
|
||||
# ------------------------------------------------------------------- utilitare
|
||||
|
||||
function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green }
|
||||
function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green }
|
||||
function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow }
|
||||
function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray }
|
||||
function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 }
|
||||
|
||||
function Test-Alive {
|
||||
param([string]$Address, [int]$TimeoutMs = 2000)
|
||||
try {
|
||||
$p = New-Object System.Net.NetworkInformation.Ping
|
||||
return ($p.Send($Address, $TimeoutMs).Status -eq 'Success')
|
||||
} catch {
|
||||
return $false
|
||||
}
|
||||
}
|
||||
|
||||
# Executa o comanda pe un nod. Intoarce liniile de output.
|
||||
# NU redirectam stderr: in PS 5.1 asta transforma fiecare linie intr-un
|
||||
# ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit.
|
||||
function Invoke-Node {
|
||||
param([string]$Node, [string]$Command)
|
||||
$out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command
|
||||
return $out
|
||||
}
|
||||
|
||||
# Ca Invoke-Node, dar respecta -DryRun (pentru comenzi care modifica starea).
|
||||
function Invoke-NodeChange {
|
||||
param([string]$Node, [string]$Command)
|
||||
if ($DryRun) {
|
||||
Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray
|
||||
return @()
|
||||
}
|
||||
return Invoke-Node -Node $Node -Command $Command
|
||||
}
|
||||
|
||||
function Confirm-Step {
|
||||
param([string]$Message)
|
||||
if ($Yes -or $DryRun) { return }
|
||||
Write-Host ""
|
||||
Write-Host $Message -ForegroundColor Yellow
|
||||
$reply = Read-Host "Scrie DA ca sa continui"
|
||||
if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." }
|
||||
}
|
||||
|
||||
# Parseaza `pct list` / `qm list` in obiecte.
|
||||
function Get-NodeGuests {
|
||||
param([string]$Node)
|
||||
$result = @()
|
||||
|
||||
$ctLines = Invoke-Node -Node $Node -Command 'pct list'
|
||||
foreach ($line in ($ctLines | Select-Object -Skip 1)) {
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 2) {
|
||||
$result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] }
|
||||
}
|
||||
}
|
||||
|
||||
$vmLines = Invoke-Node -Node $Node -Command 'qm list'
|
||||
foreach ($line in ($vmLines | Select-Object -Skip 1)) {
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 3) {
|
||||
$result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] }
|
||||
}
|
||||
}
|
||||
return $result
|
||||
}
|
||||
|
||||
function Get-GuestStatus {
|
||||
param([string]$Node, [string]$Type, [int]$Id)
|
||||
$cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" }
|
||||
$out = Invoke-Node -Node $Node -Command $cmd
|
||||
if ($out -match 'status:\s*(\w+)') { return $Matches[1] }
|
||||
return 'unknown'
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------ pas 1: preflight
|
||||
|
||||
Write-Step "Preflight"
|
||||
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
if (-not (Test-Alive $NodeIp[$node])) {
|
||||
Stop-WithError "$node ($($NodeIp[$node])) nu raspunde la ping."
|
||||
}
|
||||
$h = Invoke-Node -Node $node -Command 'hostname'
|
||||
if ($LASTEXITCODE -ne 0 -or -not $h) { Stop-WithError "$node nu accepta SSH cu cheie." }
|
||||
Write-Ok "$node ($($NodeIp[$node])) - reachable"
|
||||
}
|
||||
|
||||
$quorum = Invoke-Node -Node 'pvemini' -Command 'pvecm status'
|
||||
if (($quorum -join "`n") -notmatch 'Quorate:\s*Yes') {
|
||||
Stop-WithError "Clusterul NU are quorum. Opreste-te si investigheaza."
|
||||
}
|
||||
$haStatus = Invoke-Node -Node 'pvemini' -Command 'ha-manager status'
|
||||
$master = ($haStatus | Where-Object { $_ -match '^master\s+(\S+)' } | ForEach-Object { $Matches[1] })
|
||||
Write-Ok "quorum OK, HA master: $master"
|
||||
|
||||
$runningTasks = 0
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
$t = Invoke-Node -Node $node -Command "pvesh get /nodes/$node/tasks --limit 20 --output-format json"
|
||||
$runningTasks += ([regex]::Matches(($t -join ''), '"status":"running"')).Count
|
||||
}
|
||||
if ($runningTasks -gt 0) {
|
||||
Write-Warn "$runningTasks task-uri in executie pe cluster (backup? migrare?)."
|
||||
Confirm-Step "Sunt task-uri active. Continui oricum?"
|
||||
} else {
|
||||
Write-Ok "niciun task in executie"
|
||||
}
|
||||
|
||||
# ------------------------------------------- pas 2: inventar + salvare stare
|
||||
|
||||
Write-Step "Inventar guest-uri"
|
||||
|
||||
$haSids = @()
|
||||
foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) {
|
||||
if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" }
|
||||
}
|
||||
|
||||
$inventory = @()
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
foreach ($g in (Get-NodeGuests -Node $node)) {
|
||||
$sid = "$($g.Type):$($g.Id)"
|
||||
$g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' })
|
||||
$inventory += $g
|
||||
}
|
||||
}
|
||||
|
||||
$running = @($inventory | Where-Object { $_.Status -eq 'running' })
|
||||
|
||||
if ($running.Count -eq 0) {
|
||||
Write-Warn "Niciun guest pornit. Trec direct la oprirea nodurilor."
|
||||
} else {
|
||||
foreach ($g in $running) {
|
||||
" {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host
|
||||
}
|
||||
Write-Ok "$($running.Count) guest-uri pornite"
|
||||
}
|
||||
|
||||
# Salveaza starea LOCAL - cluster-startup porneste exact ce era pornit, ca sa nu
|
||||
# reporneasca VM-uri oprite intentionat (302, 310, 301...).
|
||||
# Format identic cu varianta bash, ca cele doua sa fie interschimbabile.
|
||||
if (-not $DryRun) {
|
||||
$lines = @("# stare cluster salvata la $(Get-Date -Format 'yyyy-MM-dd HH:mm:ss')")
|
||||
foreach ($g in $running) { $lines += "$($g.Type) $($g.Node) $($g.Id) $($g.Status) $($g.Ha)" }
|
||||
Set-Content -Path $StateFile -Value $lines -Encoding utf8
|
||||
Write-Ok "stare salvata in $StateFile"
|
||||
}
|
||||
|
||||
Confirm-Step "Se opresc $($running.Count) guest-uri, apoi nodurile: $($NodeShutdownOrder -join ', ')."
|
||||
|
||||
# -------------------------------------------- pas 3: shutdown_policy=freeze
|
||||
|
||||
Write-Step "Shutdown policy -> freeze"
|
||||
|
||||
$dcCfg = Invoke-Node -Node 'pvemini' -Command 'cat /etc/pve/datacenter.cfg 2>/dev/null || true'
|
||||
$currentPolicy = $null
|
||||
if (($dcCfg -join "`n") -match 'shutdown_policy=(\w+)') { $currentPolicy = $Matches[1] }
|
||||
|
||||
if ($currentPolicy -eq 'freeze') {
|
||||
Write-Ok "deja pe freeze"
|
||||
} else {
|
||||
if ($null -ne $currentPolicy -or ($dcCfg | Measure-Object).Count -gt 0) {
|
||||
Invoke-NodeChange -Node 'pvemini' -Command 'cp /etc/pve/datacenter.cfg /root/datacenter.cfg.backup-mentenanta' | Out-Null
|
||||
Invoke-NodeChange -Node 'pvemini' -Command "sed -i '/^ha:/d' /etc/pve/datacenter.cfg; printf 'ha: shutdown_policy=freeze\n' >> /etc/pve/datacenter.cfg" | Out-Null
|
||||
} else {
|
||||
Invoke-NodeChange -Node 'pvemini' -Command "printf 'ha: shutdown_policy=freeze\n' > /etc/pve/datacenter.cfg" | Out-Null
|
||||
}
|
||||
$was = if ($currentPolicy) { $currentPolicy } else { 'nesetat' }
|
||||
Write-Ok "setat freeze (era: $was)"
|
||||
}
|
||||
|
||||
# ---------------------------------------- pas 4: dezactivare cron-uri alerta
|
||||
|
||||
Write-Step "Dezactivare cron-uri de alerta"
|
||||
|
||||
$sedCron = "crontab -l | sed -E '/^[^#]/ s%^(.*($CronPattern)\.sh.*)`$%#MENTENANTA \1%' | crontab -"
|
||||
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
Invoke-NodeChange -Node $node -Command "crontab -l > $CronBackup 2>/dev/null || true" | Out-Null
|
||||
Invoke-NodeChange -Node $node -Command $sedCron | Out-Null
|
||||
if (-not $DryRun) {
|
||||
$n = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true"
|
||||
Write-Ok "$node - $n linii dezactivate, backup in $CronBackup"
|
||||
}
|
||||
}
|
||||
|
||||
Write-Warn "Monitorizarea e acum OPRITA pe tot clusterul. cluster-startup o restaureaza."
|
||||
|
||||
# --------------------------------------------- pas 5: shutdown curat Oracle
|
||||
|
||||
Write-Step "Oracle - shutdown immediate"
|
||||
|
||||
$oracleGuest = $running | Where-Object { $_.Type -eq 'ct' -and $_.Id -eq $OracleCt } | Select-Object -First 1
|
||||
|
||||
if (-not $oracleGuest) {
|
||||
Write-Warn "CT $OracleCt nu ruleaza - sar peste shutdown-ul bazei."
|
||||
} else {
|
||||
Write-Log "CT $OracleCt pe $($oracleGuest.Node) - opresc instanta..."
|
||||
if ($DryRun) {
|
||||
Write-Host " [dry-run] shutdown immediate in $OracleDocker" -ForegroundColor DarkGray
|
||||
} else {
|
||||
$sqlCmd = 'pct exec ' + $OracleCt + ' -- docker exec ' + $OracleDocker + ' bash -c "printf ''shutdown immediate\nexit\n'' | sqlplus -s / as sysdba"'
|
||||
$out = Invoke-Node -Node $oracleGuest.Node -Command $sqlCmd
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warn "Shutdown-ul bazei a raportat eroare. Verifica manual inainte de a continua!"
|
||||
$out | Select-Object -Last 5 | Write-Host
|
||||
} else {
|
||||
Write-Ok "instanta Oracle oprita"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# ------------------------------------------------ pas 6: oprire guest-uri
|
||||
|
||||
Write-Step "Oprire guest-uri"
|
||||
|
||||
function Stop-Guest {
|
||||
param([pscustomobject]$Guest)
|
||||
|
||||
$sid = "$($Guest.Type):$($Guest.Id)"
|
||||
$cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' }
|
||||
|
||||
if ($Guest.Ha -eq 'ha') {
|
||||
# Pentru resursele HA se foloseste ha-manager, NU pct/qm shutdown: din CLI
|
||||
# acestea nu actualizeaza state-ul HA, iar CRM-ul ar reporni guest-ul.
|
||||
Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state stopped" | Out-Null
|
||||
} else {
|
||||
Invoke-NodeChange -Node $Guest.Node -Command "$cmd shutdown $($Guest.Id) --timeout $GuestTimeout" | Out-Null
|
||||
}
|
||||
|
||||
if ($DryRun) { return }
|
||||
|
||||
$waited = 0
|
||||
while ($waited -lt $GuestTimeout) {
|
||||
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'stopped') {
|
||||
Write-Ok "$sid oprit (${waited}s)"
|
||||
return
|
||||
}
|
||||
Start-Sleep -Seconds 5
|
||||
$waited += 5
|
||||
}
|
||||
|
||||
Write-Warn "$sid nu s-a oprit in ${GuestTimeout}s - fortez stop"
|
||||
Invoke-Node -Node $Guest.Node -Command "$cmd stop $($Guest.Id)" | Out-Null
|
||||
Start-Sleep -Seconds 5
|
||||
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -ne 'stopped') {
|
||||
Stop-WithError "$sid REFUZA sa se opreasca. Rezolva manual inainte de a opri nodurile."
|
||||
}
|
||||
Write-Ok "$sid oprit fortat"
|
||||
}
|
||||
|
||||
# intai in ordinea definita...
|
||||
foreach ($id in $GuestShutdownOrder) {
|
||||
$g = $running | Where-Object { $_.Id -eq $id } | Select-Object -First 1
|
||||
if (-not $g) { continue }
|
||||
Write-Log "opresc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))"
|
||||
Stop-Guest -Guest $g
|
||||
}
|
||||
|
||||
# ...apoi orice a mai ramas pornit si nu era in lista
|
||||
foreach ($g in $running) {
|
||||
if ($GuestShutdownOrder -contains $g.Id) { continue }
|
||||
Write-Warn "guest neplanificat inca pornit: $($g.Type):$($g.Id) pe $($g.Node) - il opresc"
|
||||
Stop-Guest -Guest $g
|
||||
}
|
||||
|
||||
# ------------------------------------------- pas 7: verificare obligatorie
|
||||
|
||||
Write-Step "Verificare inainte de oprirea nodurilor"
|
||||
|
||||
if ($DryRun) {
|
||||
Write-Warn "dry-run - sar peste verificare"
|
||||
} else {
|
||||
$ha = Invoke-Node -Node 'pvemini' -Command 'ha-manager status'
|
||||
$bad = @($ha | Where-Object { $_ -match 'started|migrate|relocate|fence' })
|
||||
if ($bad.Count -gt 0) {
|
||||
$bad | Write-Host
|
||||
Stop-WithError "Exista servicii HA inca active. NU opri nodurile."
|
||||
}
|
||||
Write-Ok "toate serviciile HA sunt in state stopped"
|
||||
|
||||
$still = 0
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
$still += @((Get-NodeGuests -Node $node) | Where-Object { $_.Status -eq 'running' }).Count
|
||||
}
|
||||
if ($still -ne 0) { Stop-WithError "$still guest-uri inca pornite. NU opri nodurile." }
|
||||
Write-Ok "0 guest-uri pornite pe cluster"
|
||||
Write-Ok "watchdog-ul nu mai e armat - pierderea quorumului e inofensiva"
|
||||
}
|
||||
|
||||
if ($NoNodes) {
|
||||
Write-Step "-NoNodes: nodurile raman pornite. Gata."
|
||||
exit 0
|
||||
}
|
||||
|
||||
Confirm-Step "Se opresc nodurile in ordinea: $($NodeShutdownOrder -join ', ') (pvemini ultimul)."
|
||||
|
||||
# ------------------------------------------------- pas 8: oprire noduri
|
||||
|
||||
Write-Step "Oprire noduri"
|
||||
|
||||
foreach ($node in $NodeShutdownOrder) {
|
||||
Write-Log "opresc $node ($($NodeIp[$node]))..."
|
||||
Invoke-NodeChange -Node $node -Command 'systemctl poweroff --no-block' | Out-Null
|
||||
if ($DryRun) { continue }
|
||||
|
||||
$waited = 0
|
||||
$down = $false
|
||||
while ($waited -lt $NodeTimeout) {
|
||||
if (-not (Test-Alive $NodeIp[$node])) {
|
||||
Write-Ok "$node OPRIT (${waited}s)"
|
||||
$down = $true
|
||||
break
|
||||
}
|
||||
Start-Sleep -Seconds 10
|
||||
$waited += 10
|
||||
}
|
||||
if (-not $down) {
|
||||
Write-Warn "$node inca raspunde la ping dupa ${NodeTimeout}s."
|
||||
Write-Warn "Verifica manual - sshd poate fi deja jos desi reteaua e sus."
|
||||
}
|
||||
}
|
||||
|
||||
Write-Step "Cluster oprit"
|
||||
Write-Host ""
|
||||
Write-Host " Stare salvata: $StateFile"
|
||||
Write-Host " Backup crontab: $CronBackup (pe fiecare nod)"
|
||||
Write-Host " Shutdown policy: freeze"
|
||||
Write-Host ""
|
||||
Write-Host " La revenirea curentului: .\cluster-startup.ps1"
|
||||
Write-Host ""
|
||||
Reference in New Issue
Block a user