feat(cluster): varianta PowerShell a scripturilor de oprire/repornire
Statia de admin e Windows si nu ruleaza .sh direct: `bash` din PATH e cel din WSL, cu alt filesystem si alta configuratie de chei SSH. Adaug echivalentele native PowerShell. - cluster-shutdown.ps1 / cluster-startup.ps1, aceeasi logica si aceleasi garantii ca variantele bash - folosesc clientul OpenSSH din Windows, deja prezent in System32 - compatibile PowerShell 5.1: fara &&/||, fara ternar, fara ?., fara -AsHashtable; ping prin System.Net.NetworkInformation.Ping, nu Test-Connection (WMI, lent si des blocat) - nu redirecteaza stderr-ul lui ssh: in 5.1 asta transforma fiecare linie intr-un ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit - parsarea `pct list` / `qm list` se face in PowerShell, nu prin awk remote, ca sa nu se incurce interpolarea $1/$2 din stringurile PowerShell - scriu acelasi format de fisier de stare ca variantele bash, deci poti opri cu una si porni cu cealalta Ambele testate cu -DryRun pe clusterul live, rezultat identic cu bash. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RRyaDj39hPQ89SZS6URRpS
This commit is contained in:
417
proxmox/cluster/scripts/cluster-shutdown.ps1
Normal file
417
proxmox/cluster/scripts/cluster-shutdown.ps1
Normal file
@@ -0,0 +1,417 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Oprire controlata a intregului cluster Proxmox `romfast`.
|
||||
|
||||
.DESCRIPTION
|
||||
Opreste guest-urile in ordinea corecta de dependente, apoi nodurile, fara ca
|
||||
HA sa relocheze resursele si fara ca watchdog-ul sa reseteze hard vreun nod.
|
||||
|
||||
Procedura completa si explicatiile: docs/oprire-planificata-cluster.md
|
||||
|
||||
RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod Proxmox. Motivul: nodul care
|
||||
se opreste ultimul nu poate raporta rezultatul propriei opriri.
|
||||
|
||||
Necesita clientul OpenSSH din Windows (C:\Windows\System32\OpenSSH\ssh.exe)
|
||||
si o cheie SSH acceptata de root pe toate cele trei noduri.
|
||||
|
||||
.PARAMETER Yes
|
||||
Nu cere confirmari.
|
||||
|
||||
.PARAMETER DryRun
|
||||
Arata ce ar face, fara sa execute nimic care modifica starea.
|
||||
|
||||
.PARAMETER NoNodes
|
||||
Opreste doar guest-urile, lasa nodurile pornite.
|
||||
|
||||
.EXAMPLE
|
||||
.\cluster-shutdown.ps1 -DryRun
|
||||
.EXAMPLE
|
||||
.\cluster-shutdown.ps1
|
||||
#>
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[switch]$Yes,
|
||||
[switch]$DryRun,
|
||||
[switch]$NoNodes
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
# ---------------------------------------------------------------- configurare
|
||||
|
||||
$NodeIp = [ordered]@{
|
||||
pve1 = '10.0.20.200'
|
||||
pvemini = '10.0.20.201'
|
||||
pveelite = '10.0.20.202'
|
||||
}
|
||||
|
||||
# Ordinea de OPRIRE a nodurilor. pvemini ultimul - e nodul principal si serverul NFS.
|
||||
$NodeShutdownOrder = @('pveelite', 'pve1', 'pvemini')
|
||||
|
||||
# Ordinea de OPRIRE a guest-urilor: consumatorii intai, baza de date ULTIMA.
|
||||
# VM 201 (roacentral) si CT 104 (flowise) folosesc Oracle din CT 108.
|
||||
# Orice guest pornit care nu apare aici e oprit la final, intr-o maturare.
|
||||
$GuestShutdownOrder = @(
|
||||
303, # Win11-Adina - desktop, fara dependente
|
||||
302, # oracle-test-302 - VM de test
|
||||
310, # Win11-Template - template
|
||||
201, # roacentral - IIS reverse proxy, consumator Oracle
|
||||
101, # minecraft
|
||||
110, # moltbot
|
||||
104, # flowise - consumator Oracle
|
||||
106, # gitea
|
||||
103, # dokploy
|
||||
102, # docker.romfast.ro
|
||||
100, # portainer
|
||||
171, # claude-agent
|
||||
109, # oracle-dr-windows
|
||||
301, # docker-portainer-template
|
||||
108 # central-oracle - ULTIMUL
|
||||
)
|
||||
|
||||
$OracleCt = 108
|
||||
$OracleDocker = 'oracle-xe'
|
||||
|
||||
$GuestTimeout = 300 # secunde asteptate pentru shutdown-ul unui guest
|
||||
$NodeTimeout = 600 # secunde asteptate pana un nod nu mai raspunde la ping
|
||||
$CronBackup = '/root/crontab.backup-mentenanta.txt'
|
||||
|
||||
# Cron-uri de dezactivat cat timp clusterul e jos (altfel mail storm, iar
|
||||
# vm109-watchdog.sh chiar porneste VM 109).
|
||||
$CronPattern = 'oom-alert|pvemini-down-alert|pveelite-down-alert|vm109-watchdog'
|
||||
|
||||
$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path
|
||||
$StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' }
|
||||
|
||||
# ------------------------------------------------------------------- utilitare
|
||||
|
||||
function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green }
|
||||
function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green }
|
||||
function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow }
|
||||
function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray }
|
||||
function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 }
|
||||
|
||||
function Test-Alive {
|
||||
param([string]$Address, [int]$TimeoutMs = 2000)
|
||||
try {
|
||||
$p = New-Object System.Net.NetworkInformation.Ping
|
||||
return ($p.Send($Address, $TimeoutMs).Status -eq 'Success')
|
||||
} catch {
|
||||
return $false
|
||||
}
|
||||
}
|
||||
|
||||
# Executa o comanda pe un nod. Intoarce liniile de output.
|
||||
# NU redirectam stderr: in PS 5.1 asta transforma fiecare linie intr-un
|
||||
# ErrorRecord si strica $LASTEXITCODE chiar cand comanda a reusit.
|
||||
function Invoke-Node {
|
||||
param([string]$Node, [string]$Command)
|
||||
$out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command
|
||||
return $out
|
||||
}
|
||||
|
||||
# Ca Invoke-Node, dar respecta -DryRun (pentru comenzi care modifica starea).
|
||||
function Invoke-NodeChange {
|
||||
param([string]$Node, [string]$Command)
|
||||
if ($DryRun) {
|
||||
Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray
|
||||
return @()
|
||||
}
|
||||
return Invoke-Node -Node $Node -Command $Command
|
||||
}
|
||||
|
||||
function Confirm-Step {
|
||||
param([string]$Message)
|
||||
if ($Yes -or $DryRun) { return }
|
||||
Write-Host ""
|
||||
Write-Host $Message -ForegroundColor Yellow
|
||||
$reply = Read-Host "Scrie DA ca sa continui"
|
||||
if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." }
|
||||
}
|
||||
|
||||
# Parseaza `pct list` / `qm list` in obiecte.
|
||||
function Get-NodeGuests {
|
||||
param([string]$Node)
|
||||
$result = @()
|
||||
|
||||
$ctLines = Invoke-Node -Node $Node -Command 'pct list'
|
||||
foreach ($line in ($ctLines | Select-Object -Skip 1)) {
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 2) {
|
||||
$result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] }
|
||||
}
|
||||
}
|
||||
|
||||
$vmLines = Invoke-Node -Node $Node -Command 'qm list'
|
||||
foreach ($line in ($vmLines | Select-Object -Skip 1)) {
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 3) {
|
||||
$result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] }
|
||||
}
|
||||
}
|
||||
return $result
|
||||
}
|
||||
|
||||
function Get-GuestStatus {
|
||||
param([string]$Node, [string]$Type, [int]$Id)
|
||||
$cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" }
|
||||
$out = Invoke-Node -Node $Node -Command $cmd
|
||||
if ($out -match 'status:\s*(\w+)') { return $Matches[1] }
|
||||
return 'unknown'
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------ pas 1: preflight
|
||||
|
||||
Write-Step "Preflight"
|
||||
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
if (-not (Test-Alive $NodeIp[$node])) {
|
||||
Stop-WithError "$node ($($NodeIp[$node])) nu raspunde la ping."
|
||||
}
|
||||
$h = Invoke-Node -Node $node -Command 'hostname'
|
||||
if ($LASTEXITCODE -ne 0 -or -not $h) { Stop-WithError "$node nu accepta SSH cu cheie." }
|
||||
Write-Ok "$node ($($NodeIp[$node])) - reachable"
|
||||
}
|
||||
|
||||
$quorum = Invoke-Node -Node 'pvemini' -Command 'pvecm status'
|
||||
if (($quorum -join "`n") -notmatch 'Quorate:\s*Yes') {
|
||||
Stop-WithError "Clusterul NU are quorum. Opreste-te si investigheaza."
|
||||
}
|
||||
$haStatus = Invoke-Node -Node 'pvemini' -Command 'ha-manager status'
|
||||
$master = ($haStatus | Where-Object { $_ -match '^master\s+(\S+)' } | ForEach-Object { $Matches[1] })
|
||||
Write-Ok "quorum OK, HA master: $master"
|
||||
|
||||
$runningTasks = 0
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
$t = Invoke-Node -Node $node -Command "pvesh get /nodes/$node/tasks --limit 20 --output-format json"
|
||||
$runningTasks += ([regex]::Matches(($t -join ''), '"status":"running"')).Count
|
||||
}
|
||||
if ($runningTasks -gt 0) {
|
||||
Write-Warn "$runningTasks task-uri in executie pe cluster (backup? migrare?)."
|
||||
Confirm-Step "Sunt task-uri active. Continui oricum?"
|
||||
} else {
|
||||
Write-Ok "niciun task in executie"
|
||||
}
|
||||
|
||||
# ------------------------------------------- pas 2: inventar + salvare stare
|
||||
|
||||
Write-Step "Inventar guest-uri"
|
||||
|
||||
$haSids = @()
|
||||
foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) {
|
||||
if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" }
|
||||
}
|
||||
|
||||
$inventory = @()
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
foreach ($g in (Get-NodeGuests -Node $node)) {
|
||||
$sid = "$($g.Type):$($g.Id)"
|
||||
$g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' })
|
||||
$inventory += $g
|
||||
}
|
||||
}
|
||||
|
||||
$running = @($inventory | Where-Object { $_.Status -eq 'running' })
|
||||
|
||||
if ($running.Count -eq 0) {
|
||||
Write-Warn "Niciun guest pornit. Trec direct la oprirea nodurilor."
|
||||
} else {
|
||||
foreach ($g in $running) {
|
||||
" {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host
|
||||
}
|
||||
Write-Ok "$($running.Count) guest-uri pornite"
|
||||
}
|
||||
|
||||
# Salveaza starea LOCAL - cluster-startup porneste exact ce era pornit, ca sa nu
|
||||
# reporneasca VM-uri oprite intentionat (302, 310, 301...).
|
||||
# Format identic cu varianta bash, ca cele doua sa fie interschimbabile.
|
||||
if (-not $DryRun) {
|
||||
$lines = @("# stare cluster salvata la $(Get-Date -Format 'yyyy-MM-dd HH:mm:ss')")
|
||||
foreach ($g in $running) { $lines += "$($g.Type) $($g.Node) $($g.Id) $($g.Status) $($g.Ha)" }
|
||||
Set-Content -Path $StateFile -Value $lines -Encoding utf8
|
||||
Write-Ok "stare salvata in $StateFile"
|
||||
}
|
||||
|
||||
Confirm-Step "Se opresc $($running.Count) guest-uri, apoi nodurile: $($NodeShutdownOrder -join ', ')."
|
||||
|
||||
# -------------------------------------------- pas 3: shutdown_policy=freeze
|
||||
|
||||
Write-Step "Shutdown policy -> freeze"
|
||||
|
||||
$dcCfg = Invoke-Node -Node 'pvemini' -Command 'cat /etc/pve/datacenter.cfg 2>/dev/null || true'
|
||||
$currentPolicy = $null
|
||||
if (($dcCfg -join "`n") -match 'shutdown_policy=(\w+)') { $currentPolicy = $Matches[1] }
|
||||
|
||||
if ($currentPolicy -eq 'freeze') {
|
||||
Write-Ok "deja pe freeze"
|
||||
} else {
|
||||
if ($null -ne $currentPolicy -or ($dcCfg | Measure-Object).Count -gt 0) {
|
||||
Invoke-NodeChange -Node 'pvemini' -Command 'cp /etc/pve/datacenter.cfg /root/datacenter.cfg.backup-mentenanta' | Out-Null
|
||||
Invoke-NodeChange -Node 'pvemini' -Command "sed -i '/^ha:/d' /etc/pve/datacenter.cfg; printf 'ha: shutdown_policy=freeze\n' >> /etc/pve/datacenter.cfg" | Out-Null
|
||||
} else {
|
||||
Invoke-NodeChange -Node 'pvemini' -Command "printf 'ha: shutdown_policy=freeze\n' > /etc/pve/datacenter.cfg" | Out-Null
|
||||
}
|
||||
$was = if ($currentPolicy) { $currentPolicy } else { 'nesetat' }
|
||||
Write-Ok "setat freeze (era: $was)"
|
||||
}
|
||||
|
||||
# ---------------------------------------- pas 4: dezactivare cron-uri alerta
|
||||
|
||||
Write-Step "Dezactivare cron-uri de alerta"
|
||||
|
||||
$sedCron = "crontab -l | sed -E '/^[^#]/ s%^(.*($CronPattern)\.sh.*)`$%#MENTENANTA \1%' | crontab -"
|
||||
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
Invoke-NodeChange -Node $node -Command "crontab -l > $CronBackup 2>/dev/null || true" | Out-Null
|
||||
Invoke-NodeChange -Node $node -Command $sedCron | Out-Null
|
||||
if (-not $DryRun) {
|
||||
$n = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true"
|
||||
Write-Ok "$node - $n linii dezactivate, backup in $CronBackup"
|
||||
}
|
||||
}
|
||||
|
||||
Write-Warn "Monitorizarea e acum OPRITA pe tot clusterul. cluster-startup o restaureaza."
|
||||
|
||||
# --------------------------------------------- pas 5: shutdown curat Oracle
|
||||
|
||||
Write-Step "Oracle - shutdown immediate"
|
||||
|
||||
$oracleGuest = $running | Where-Object { $_.Type -eq 'ct' -and $_.Id -eq $OracleCt } | Select-Object -First 1
|
||||
|
||||
if (-not $oracleGuest) {
|
||||
Write-Warn "CT $OracleCt nu ruleaza - sar peste shutdown-ul bazei."
|
||||
} else {
|
||||
Write-Log "CT $OracleCt pe $($oracleGuest.Node) - opresc instanta..."
|
||||
if ($DryRun) {
|
||||
Write-Host " [dry-run] shutdown immediate in $OracleDocker" -ForegroundColor DarkGray
|
||||
} else {
|
||||
$sqlCmd = 'pct exec ' + $OracleCt + ' -- docker exec ' + $OracleDocker + ' bash -c "printf ''shutdown immediate\nexit\n'' | sqlplus -s / as sysdba"'
|
||||
$out = Invoke-Node -Node $oracleGuest.Node -Command $sqlCmd
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warn "Shutdown-ul bazei a raportat eroare. Verifica manual inainte de a continua!"
|
||||
$out | Select-Object -Last 5 | Write-Host
|
||||
} else {
|
||||
Write-Ok "instanta Oracle oprita"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# ------------------------------------------------ pas 6: oprire guest-uri
|
||||
|
||||
Write-Step "Oprire guest-uri"
|
||||
|
||||
function Stop-Guest {
|
||||
param([pscustomobject]$Guest)
|
||||
|
||||
$sid = "$($Guest.Type):$($Guest.Id)"
|
||||
$cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' }
|
||||
|
||||
if ($Guest.Ha -eq 'ha') {
|
||||
# Pentru resursele HA se foloseste ha-manager, NU pct/qm shutdown: din CLI
|
||||
# acestea nu actualizeaza state-ul HA, iar CRM-ul ar reporni guest-ul.
|
||||
Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state stopped" | Out-Null
|
||||
} else {
|
||||
Invoke-NodeChange -Node $Guest.Node -Command "$cmd shutdown $($Guest.Id) --timeout $GuestTimeout" | Out-Null
|
||||
}
|
||||
|
||||
if ($DryRun) { return }
|
||||
|
||||
$waited = 0
|
||||
while ($waited -lt $GuestTimeout) {
|
||||
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'stopped') {
|
||||
Write-Ok "$sid oprit (${waited}s)"
|
||||
return
|
||||
}
|
||||
Start-Sleep -Seconds 5
|
||||
$waited += 5
|
||||
}
|
||||
|
||||
Write-Warn "$sid nu s-a oprit in ${GuestTimeout}s - fortez stop"
|
||||
Invoke-Node -Node $Guest.Node -Command "$cmd stop $($Guest.Id)" | Out-Null
|
||||
Start-Sleep -Seconds 5
|
||||
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -ne 'stopped') {
|
||||
Stop-WithError "$sid REFUZA sa se opreasca. Rezolva manual inainte de a opri nodurile."
|
||||
}
|
||||
Write-Ok "$sid oprit fortat"
|
||||
}
|
||||
|
||||
# intai in ordinea definita...
|
||||
foreach ($id in $GuestShutdownOrder) {
|
||||
$g = $running | Where-Object { $_.Id -eq $id } | Select-Object -First 1
|
||||
if (-not $g) { continue }
|
||||
Write-Log "opresc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))"
|
||||
Stop-Guest -Guest $g
|
||||
}
|
||||
|
||||
# ...apoi orice a mai ramas pornit si nu era in lista
|
||||
foreach ($g in $running) {
|
||||
if ($GuestShutdownOrder -contains $g.Id) { continue }
|
||||
Write-Warn "guest neplanificat inca pornit: $($g.Type):$($g.Id) pe $($g.Node) - il opresc"
|
||||
Stop-Guest -Guest $g
|
||||
}
|
||||
|
||||
# ------------------------------------------- pas 7: verificare obligatorie
|
||||
|
||||
Write-Step "Verificare inainte de oprirea nodurilor"
|
||||
|
||||
if ($DryRun) {
|
||||
Write-Warn "dry-run - sar peste verificare"
|
||||
} else {
|
||||
$ha = Invoke-Node -Node 'pvemini' -Command 'ha-manager status'
|
||||
$bad = @($ha | Where-Object { $_ -match 'started|migrate|relocate|fence' })
|
||||
if ($bad.Count -gt 0) {
|
||||
$bad | Write-Host
|
||||
Stop-WithError "Exista servicii HA inca active. NU opri nodurile."
|
||||
}
|
||||
Write-Ok "toate serviciile HA sunt in state stopped"
|
||||
|
||||
$still = 0
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
$still += @((Get-NodeGuests -Node $node) | Where-Object { $_.Status -eq 'running' }).Count
|
||||
}
|
||||
if ($still -ne 0) { Stop-WithError "$still guest-uri inca pornite. NU opri nodurile." }
|
||||
Write-Ok "0 guest-uri pornite pe cluster"
|
||||
Write-Ok "watchdog-ul nu mai e armat - pierderea quorumului e inofensiva"
|
||||
}
|
||||
|
||||
if ($NoNodes) {
|
||||
Write-Step "-NoNodes: nodurile raman pornite. Gata."
|
||||
exit 0
|
||||
}
|
||||
|
||||
Confirm-Step "Se opresc nodurile in ordinea: $($NodeShutdownOrder -join ', ') (pvemini ultimul)."
|
||||
|
||||
# ------------------------------------------------- pas 8: oprire noduri
|
||||
|
||||
Write-Step "Oprire noduri"
|
||||
|
||||
foreach ($node in $NodeShutdownOrder) {
|
||||
Write-Log "opresc $node ($($NodeIp[$node]))..."
|
||||
Invoke-NodeChange -Node $node -Command 'systemctl poweroff --no-block' | Out-Null
|
||||
if ($DryRun) { continue }
|
||||
|
||||
$waited = 0
|
||||
$down = $false
|
||||
while ($waited -lt $NodeTimeout) {
|
||||
if (-not (Test-Alive $NodeIp[$node])) {
|
||||
Write-Ok "$node OPRIT (${waited}s)"
|
||||
$down = $true
|
||||
break
|
||||
}
|
||||
Start-Sleep -Seconds 10
|
||||
$waited += 10
|
||||
}
|
||||
if (-not $down) {
|
||||
Write-Warn "$node inca raspunde la ping dupa ${NodeTimeout}s."
|
||||
Write-Warn "Verifica manual - sshd poate fi deja jos desi reteaua e sus."
|
||||
}
|
||||
}
|
||||
|
||||
Write-Step "Cluster oprit"
|
||||
Write-Host ""
|
||||
Write-Host " Stare salvata: $StateFile"
|
||||
Write-Host " Backup crontab: $CronBackup (pe fiecare nod)"
|
||||
Write-Host " Shutdown policy: freeze"
|
||||
Write-Host ""
|
||||
Write-Host " La revenirea curentului: .\cluster-startup.ps1"
|
||||
Write-Host ""
|
||||
401
proxmox/cluster/scripts/cluster-startup.ps1
Normal file
401
proxmox/cluster/scripts/cluster-startup.ps1
Normal file
@@ -0,0 +1,401 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Repornirea clusterului Proxmox `romfast` dupa o oprire planificata.
|
||||
|
||||
.DESCRIPTION
|
||||
Asteapta nodurile, asteapta quorumul, porneste guest-urile in ordinea corecta
|
||||
de dependente (Oracle primul), restaureaza cron-urile de monitorizare.
|
||||
|
||||
Porneste DOAR guest-urile care rulau inainte de oprire, citite din fisierul de
|
||||
stare scris de cluster-shutdown. Fara fisier de stare, foloseste lista
|
||||
implicita, minus guest-urile din NeverAutostart.
|
||||
|
||||
RULEAZA DE PE STATIA DE ADMIN, nu de pe un nod.
|
||||
|
||||
NODURILE TREBUIE PORNITE FIZIC INTAI. Scriptul le asteapta.
|
||||
Porneste-le pe TOATE TREI aproximativ simultan: un singur nod pornit nu are
|
||||
quorum, /etc/pve ramane read-only si nu se poate porni niciun guest.
|
||||
|
||||
.PARAMETER Yes
|
||||
Nu cere confirmari.
|
||||
|
||||
.PARAMETER DryRun
|
||||
Arata ce ar face, fara sa execute nimic care modifica starea.
|
||||
|
||||
.PARAMETER All
|
||||
Ignora fisierul de stare, foloseste lista implicita.
|
||||
|
||||
.EXAMPLE
|
||||
.\cluster-startup.ps1 -DryRun
|
||||
.EXAMPLE
|
||||
.\cluster-startup.ps1
|
||||
#>
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[switch]$Yes,
|
||||
[switch]$DryRun,
|
||||
[switch]$All
|
||||
)
|
||||
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
# ---------------------------------------------------------------- configurare
|
||||
|
||||
$NodeIp = [ordered]@{
|
||||
pve1 = '10.0.20.200'
|
||||
pvemini = '10.0.20.201'
|
||||
pveelite = '10.0.20.202'
|
||||
}
|
||||
|
||||
# Ordinea de PORNIRE: baza de date intai, consumatorii dupa.
|
||||
# Inversul ordinii de oprire din cluster-shutdown.
|
||||
$GuestStartOrder = @(
|
||||
108, # central-oracle - PRIMUL, restul depind de el
|
||||
201, # roacentral - IIS reverse proxy, consumator Oracle
|
||||
104, # flowise - consumator Oracle
|
||||
100, # portainer
|
||||
102, # docker.romfast.ro
|
||||
103, # dokploy
|
||||
106, # gitea
|
||||
171, # claude-agent
|
||||
101, # minecraft
|
||||
110, # moltbot
|
||||
303, # Win11-Adina
|
||||
302, # oracle-test-302
|
||||
109, # oracle-dr-windows
|
||||
301, # docker-portainer-template
|
||||
310 # Win11-Template
|
||||
)
|
||||
|
||||
# Guest-uri care NU se pornesc automat cand LIPSESTE fisierul de stare.
|
||||
# Daca fisierul de stare exista, el decide - daca erau pornite, se repornesc.
|
||||
# 109 - VM DR. Pornirea lui nedorita a declansat bucla OOM din incidentul
|
||||
# 2026-04-20; in mod normal sta oprit si e pornit doar de testul DR.
|
||||
# 301, 310 - template-uri, nu servicii.
|
||||
$NeverAutostart = @(109, 301, 310)
|
||||
|
||||
$OracleCt = 108
|
||||
$OracleDocker = 'oracle-xe'
|
||||
$OracleWait = 600 # secunde asteptate pana baza raspunde la interogari
|
||||
|
||||
$NodeWait = 1800 # secunde asteptate pana apar toate nodurile
|
||||
$QuorumWait = 300
|
||||
$GuestWait = 300
|
||||
$CronBackup = '/root/crontab.backup-mentenanta.txt'
|
||||
|
||||
$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path
|
||||
$StateFile = if ($env:CLUSTER_STATE_FILE) { $env:CLUSTER_STATE_FILE } else { Join-Path $ScriptDir '.cluster-state.txt' }
|
||||
|
||||
# ------------------------------------------------------------------- utilitare
|
||||
|
||||
function Write-Step { param([string]$Text) Write-Host ""; Write-Host "== $Text" -ForegroundColor Green }
|
||||
function Write-Ok { param([string]$Text) Write-Host " [ok] $Text" -ForegroundColor Green }
|
||||
function Write-Warn { param([string]$Text) Write-Host " [!] $Text" -ForegroundColor Yellow }
|
||||
function Write-Log { param([string]$Text) Write-Host "[$(Get-Date -Format HH:mm:ss)] $Text" -ForegroundColor DarkGray }
|
||||
function Stop-WithError { param([string]$Text) Write-Host " [X] $Text" -ForegroundColor Red; exit 1 }
|
||||
|
||||
function Test-Alive {
|
||||
param([string]$Address, [int]$TimeoutMs = 2000)
|
||||
try {
|
||||
$p = New-Object System.Net.NetworkInformation.Ping
|
||||
return ($p.Send($Address, $TimeoutMs).Status -eq 'Success')
|
||||
} catch {
|
||||
return $false
|
||||
}
|
||||
}
|
||||
|
||||
function Invoke-Node {
|
||||
param([string]$Node, [string]$Command)
|
||||
$out = ssh -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new -o LogLevel=ERROR "root@$($NodeIp[$Node])" $Command
|
||||
return $out
|
||||
}
|
||||
|
||||
function Invoke-NodeChange {
|
||||
param([string]$Node, [string]$Command)
|
||||
if ($DryRun) {
|
||||
Write-Host " [dry-run] ${Node}: $Command" -ForegroundColor DarkGray
|
||||
return @()
|
||||
}
|
||||
return Invoke-Node -Node $Node -Command $Command
|
||||
}
|
||||
|
||||
function Confirm-Step {
|
||||
param([string]$Message)
|
||||
if ($Yes -or $DryRun) { return }
|
||||
Write-Host ""
|
||||
Write-Host $Message -ForegroundColor Yellow
|
||||
$reply = Read-Host "Scrie DA ca sa continui"
|
||||
if ($reply -ne 'DA') { Stop-WithError "Anulat de utilizator." }
|
||||
}
|
||||
|
||||
function Get-NodeGuests {
|
||||
param([string]$Node)
|
||||
$result = @()
|
||||
|
||||
foreach ($line in ((Invoke-Node -Node $Node -Command 'pct list') | Select-Object -Skip 1)) {
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 2) {
|
||||
$result += [pscustomobject]@{ Type='ct'; Node=$Node; Id=[int]$f[0]; Status=$f[1] }
|
||||
}
|
||||
}
|
||||
foreach ($line in ((Invoke-Node -Node $Node -Command 'qm list') | Select-Object -Skip 1)) {
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 3) {
|
||||
$result += [pscustomobject]@{ Type='vm'; Node=$Node; Id=[int]$f[0]; Status=$f[2] }
|
||||
}
|
||||
}
|
||||
return $result
|
||||
}
|
||||
|
||||
function Get-GuestStatus {
|
||||
param([string]$Node, [string]$Type, [int]$Id)
|
||||
$cmd = if ($Type -eq 'ct') { "pct status $Id" } else { "qm status $Id" }
|
||||
$out = Invoke-Node -Node $Node -Command $cmd
|
||||
if ($out -match 'status:\s*(\w+)') { return $Matches[1] }
|
||||
return 'unknown'
|
||||
}
|
||||
|
||||
# ------------------------------------------------- pas 1: asteapta nodurile
|
||||
|
||||
Write-Step "Astept nodurile"
|
||||
Write-Warn "Nodurile trebuie pornite FIZIC. Porneste-le pe toate trei aproximativ simultan."
|
||||
|
||||
$waited = 0
|
||||
$missing = @()
|
||||
while ($waited -lt $NodeWait) {
|
||||
$missing = @()
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
if (-not (Test-Alive $NodeIp[$node])) { $missing += $node }
|
||||
}
|
||||
if ($missing.Count -eq 0) { break }
|
||||
Write-Host "`r astept: $($missing -join ', ') (${waited}s) " -NoNewline -ForegroundColor DarkGray
|
||||
Start-Sleep -Seconds 10
|
||||
$waited += 10
|
||||
}
|
||||
Write-Host "`r `r" -NoNewline
|
||||
if ($missing.Count -ne 0) { Stop-WithError "Nodurile $($missing -join ', ') nu au aparut in ${NodeWait}s." }
|
||||
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
$w = 0
|
||||
while ($true) {
|
||||
$h = Invoke-Node -Node $node -Command 'uptime -p'
|
||||
if ($LASTEXITCODE -eq 0 -and $h) { break }
|
||||
$w += 10
|
||||
if ($w -gt 300) { Stop-WithError "$node raspunde la ping dar nu la SSH." }
|
||||
Start-Sleep -Seconds 10
|
||||
}
|
||||
Write-Ok "$node - up, SSH OK ($h)"
|
||||
}
|
||||
|
||||
# --------------------------------------------------- pas 2: asteapta quorum
|
||||
|
||||
Write-Step "Astept quorumul"
|
||||
|
||||
$waited = 0
|
||||
while ($true) {
|
||||
$q = Invoke-Node -Node 'pvemini' -Command 'pvecm status'
|
||||
if (($q -join "`n") -match 'Quorate:\s*Yes') { break }
|
||||
if ($waited -ge $QuorumWait) { Stop-WithError "Clusterul nu a atins quorumul in ${QuorumWait}s." }
|
||||
Write-Host "`r astept quorum (${waited}s)... " -NoNewline -ForegroundColor DarkGray
|
||||
Start-Sleep -Seconds 10
|
||||
$waited += 10
|
||||
}
|
||||
Write-Host "`r `r" -NoNewline
|
||||
|
||||
$votes = 0
|
||||
if (($q -join "`n") -match 'Total votes:\s*(\d+)') { $votes = [int]$Matches[1] }
|
||||
if ($votes -ne 3) { Write-Warn "Quorum atins, dar doar $votes/3 voturi. Un nod lipseste din cluster." }
|
||||
Write-Ok "quorum OK, $votes/3 voturi"
|
||||
|
||||
Invoke-Node -Node 'pvemini' -Command 'ha-manager status' | Select-Object -First 4 | Write-Host
|
||||
|
||||
# ------------------------------------------------- pas 3: ce trebuie pornit
|
||||
|
||||
Write-Step "Ce se porneste"
|
||||
|
||||
$toStart = @()
|
||||
|
||||
if ((-not $All) -and (Test-Path $StateFile)) {
|
||||
foreach ($line in (Get-Content $StateFile)) {
|
||||
if ($line -match '^\s*#' -or $line -match '^\s*$') { continue }
|
||||
$f = ($line -split '\s+') | Where-Object { $_ -ne '' }
|
||||
if ($f.Count -ge 5) {
|
||||
$toStart += [pscustomobject]@{ Type=$f[0]; Node=$f[1]; Id=[int]$f[2]; Status=$f[3]; Ha=$f[4] }
|
||||
}
|
||||
}
|
||||
Write-Ok "citit din $StateFile ($($toStart.Count) guest-uri)"
|
||||
} else {
|
||||
if (-not $All) { Write-Warn "Nu exista $StateFile - folosesc lista implicita." }
|
||||
|
||||
$haSids = @()
|
||||
foreach ($line in (Invoke-Node -Node 'pvemini' -Command 'ha-manager config')) {
|
||||
if ($line -match '^(ct|vm):(\d+)') { $haSids += "$($Matches[1]):$($Matches[2])" }
|
||||
}
|
||||
|
||||
$allGuests = @()
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
foreach ($g in (Get-NodeGuests -Node $node)) {
|
||||
$sid = "$($g.Type):$($g.Id)"
|
||||
$g | Add-Member -NotePropertyName Ha -NotePropertyValue $(if ($haSids -contains $sid) { 'ha' } else { 'noha' })
|
||||
$allGuests += $g
|
||||
}
|
||||
}
|
||||
|
||||
# Fara fisier de stare pornim doar ce e explicit in lista de ordine SI nu e
|
||||
# in NeverAutostart. Altfel am porni DR-ul si template-urile, care erau
|
||||
# oprite intentionat.
|
||||
foreach ($id in $GuestStartOrder) {
|
||||
if ($NeverAutostart -contains $id) {
|
||||
Write-Warn "$id sarit (NeverAutostart) - porneste-l manual daca chiar il vrei"
|
||||
continue
|
||||
}
|
||||
$g = $allGuests | Where-Object { $_.Id -eq $id } | Select-Object -First 1
|
||||
if ($g) { $toStart += $g }
|
||||
}
|
||||
}
|
||||
|
||||
if ($toStart.Count -eq 0) { Stop-WithError "Nimic de pornit." }
|
||||
|
||||
foreach ($g in $toStart) {
|
||||
" {0,-3} {1,-9} {2,-4} {3}" -f $g.Type, $g.Node, $g.Id, $g.Ha | Write-Host
|
||||
}
|
||||
|
||||
Confirm-Step "Se pornesc guest-urile de mai sus, Oracle primul."
|
||||
|
||||
# ------------------------------------------------- pas 4: pornire guest-uri
|
||||
|
||||
Write-Step "Pornire guest-uri"
|
||||
|
||||
$failed = @()
|
||||
|
||||
function Start-Guest {
|
||||
param([pscustomobject]$Guest)
|
||||
|
||||
$sid = "$($Guest.Type):$($Guest.Id)"
|
||||
$cmd = if ($Guest.Type -eq 'ct') { 'pct' } else { 'qm' }
|
||||
|
||||
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'running') {
|
||||
Write-Ok "$sid deja pornit"
|
||||
return $true
|
||||
}
|
||||
|
||||
if ($Guest.Ha -eq 'ha') {
|
||||
# Pentru resursele HA, --state started repune resursa sub controlul CRM.
|
||||
Invoke-NodeChange -Node $Guest.Node -Command "ha-manager set $sid --state started" | Out-Null
|
||||
} else {
|
||||
Invoke-NodeChange -Node $Guest.Node -Command "$cmd start $($Guest.Id)" | Out-Null
|
||||
}
|
||||
|
||||
if ($DryRun) { return $true }
|
||||
|
||||
$waited = 0
|
||||
while ($waited -lt $GuestWait) {
|
||||
if ((Get-GuestStatus -Node $Guest.Node -Type $Guest.Type -Id $Guest.Id) -eq 'running') {
|
||||
Write-Ok "$sid pornit (${waited}s)"
|
||||
return $true
|
||||
}
|
||||
Start-Sleep -Seconds 5
|
||||
$waited += 5
|
||||
}
|
||||
Write-Warn "$sid nu a pornit in ${GuestWait}s - verifica manual"
|
||||
return $false
|
||||
}
|
||||
|
||||
function Wait-Oracle {
|
||||
param([string]$Node)
|
||||
Write-Log "astept ca instanta Oracle sa accepte interogari..."
|
||||
if ($DryRun) { return }
|
||||
|
||||
$sql = 'pct exec ' + $OracleCt + ' -- docker exec ' + $OracleDocker + ' bash -c "printf ''set pagesize 0 feedback off\nselect global_name from global_name;\nexit\n'' | sqlplus -s / as sysdba"'
|
||||
$waited = 0
|
||||
while ($waited -lt $OracleWait) {
|
||||
$out = Invoke-Node -Node $Node -Command $sql
|
||||
if ($LASTEXITCODE -eq 0 -and (($out -join '') -match '[A-Z]')) {
|
||||
Write-Host "`r `r" -NoNewline
|
||||
Write-Ok "Oracle deschis si interogabil (${waited}s)"
|
||||
return
|
||||
}
|
||||
Write-Host "`r Oracle inca nu raspunde (${waited}s)... " -NoNewline -ForegroundColor DarkGray
|
||||
Start-Sleep -Seconds 15
|
||||
$waited += 15
|
||||
}
|
||||
Write-Host "`r `r" -NoNewline
|
||||
Write-Warn "Oracle nu a raspuns in ${OracleWait}s. Continui, dar VERIFICA MANUAL."
|
||||
Write-Warn " ssh root@$($NodeIp[$Node]) `"pct exec $OracleCt -- docker logs --tail 50 $OracleDocker`""
|
||||
}
|
||||
|
||||
foreach ($id in $GuestStartOrder) {
|
||||
$g = $toStart | Where-Object { $_.Id -eq $id } | Select-Object -First 1
|
||||
if (-not $g) { continue }
|
||||
|
||||
Write-Log "pornesc $($g.Type):$($g.Id) pe $($g.Node) ($($g.Ha))"
|
||||
if (-not (Start-Guest -Guest $g)) { $failed += "$($g.Type):$($g.Id)" }
|
||||
|
||||
# Oracle e blocant: nimic care depinde de el nu porneste pana nu e deschis.
|
||||
if ($g.Id -eq $OracleCt) { Wait-Oracle -Node $g.Node }
|
||||
}
|
||||
|
||||
# orice a mai ramas in lista si nu era in ordinea definita
|
||||
foreach ($g in $toStart) {
|
||||
if ($GuestStartOrder -contains $g.Id) { continue }
|
||||
Write-Warn "guest neplanificat: $($g.Type):$($g.Id) pe $($g.Node) - il pornesc"
|
||||
if (-not (Start-Guest -Guest $g)) { $failed += "$($g.Type):$($g.Id)" }
|
||||
}
|
||||
|
||||
# ------------------------------------------------ pas 5: restaurare cron-uri
|
||||
|
||||
Write-Step "Restaurare cron-uri de monitorizare"
|
||||
|
||||
foreach ($node in $NodeIp.Keys) {
|
||||
$exists = Invoke-Node -Node $node -Command "test -f $CronBackup; echo `$?"
|
||||
if (($exists -join '') -notmatch '^0') {
|
||||
Write-Warn "$node - nu exista $CronBackup, sar peste"
|
||||
continue
|
||||
}
|
||||
Invoke-NodeChange -Node $node -Command "crontab $CronBackup" | Out-Null
|
||||
if (-not $DryRun) {
|
||||
$left = Invoke-Node -Node $node -Command "crontab -l | grep -c '^#MENTENANTA' || true"
|
||||
if (($left -join '').Trim() -eq '0') {
|
||||
Write-Ok "$node - cron-uri restaurate"
|
||||
} else {
|
||||
Write-Warn "$node - au ramas $left linii #MENTENANTA, verifica manual"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# ------------------------------------------------------- pas 6: verificare
|
||||
|
||||
Write-Step "Verificare finala"
|
||||
|
||||
if ($DryRun) {
|
||||
Write-Warn "dry-run - sar peste verificare"
|
||||
} else {
|
||||
Invoke-Node -Node 'pvemini' -Command 'ha-manager status' | Write-Host
|
||||
Write-Host ""
|
||||
$rep = Invoke-Node -Node 'pvemini' -Command 'pvesr status'
|
||||
$rep | Write-Host
|
||||
Write-Host ""
|
||||
|
||||
$badRep = @($rep | Select-Object -Skip 1 | Where-Object { $_ -notmatch 'OK\s*$' -and $_.Trim() -ne '' })
|
||||
if ($badRep.Count -eq 0) {
|
||||
Write-Ok "replicare: toate job-urile OK"
|
||||
} else {
|
||||
Write-Warn "job-uri de replicare cu probleme:"
|
||||
$badRep | Write-Host
|
||||
}
|
||||
}
|
||||
|
||||
Write-Step "Cluster pornit"
|
||||
|
||||
if ($failed.Count -gt 0) {
|
||||
Write-Warn "Guest-uri care NU au pornit: $($failed -join ', ')"
|
||||
Write-Warn "Verifica-le manual inainte de a considera repornirea terminata."
|
||||
} else {
|
||||
Write-Ok "toate guest-urile au pornit"
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host " De verificat manual:"
|
||||
Write-Host " - shutdown_policy e inca 'freeze' (recomandat sa ramana asa)"
|
||||
Write-Host " - daca lipsesc date recente, compara cu ultima replicare din pvesr status"
|
||||
Write-Host ""
|
||||
Reference in New Issue
Block a user