#!/bin/bash # # ups-shutdown-cluster.sh — oprire orchestrată a clusterului Proxmox `romfast` # când UPS-ul e pe baterie. # # Rulează PE pvemini (nodul cu UPS-ul pe USB), declanșat de upssched: # ONBATT + 3 min → shutdown orchestrat # LOWBATT → shutdown imediat # # TREBUIE SĂ RULEZE CA ROOT. upssched rulează ca user `nut`, deci upssched-cmd # îl invocă prin `sudo` (vezi /etc/sudoers.d/nut-shutdown). Fără asta, scriptul # nu poate face nici pct/qm/ha-manager, nici ssh (nut nu are chei), nici # shutdown — exact motivul pentru care versiunea veche nu a funcționat # niciodată, deși trimitea emailuri. # # Utilizare manuală: # ups-shutdown-cluster.sh --dry-run # arată ce ar face, nu atinge nimic # ups-shutdown-cluster.sh --force # rulează chiar dacă UPS-ul e pe rețea # # Rescris: 2026-08-27 # # NU folosim `set -e`: într-o oprire de urgență, un ssh eșuat către un nod deja # mort nu trebuie să oprească restul procedurii. set -uo pipefail # ---------------------------------------------------------------- configurare LOGFILE=/var/log/ups-shutdown.log LOCKFILE=/run/ups-shutdown.lock SECONDARY_IPS=(10.0.20.200 10.0.20.202) SECONDARY_NAMES=(pve1 pveelite) ORACLE_CT=108 ORACLE_DOCKER=oracle-xe UPS_NAME="nutdev1" UPS_USER="admin" UPS_PASS="" # din /etc/nut/ups-shutdown.conf # Bugete de timp — sunt secunde de baterie, nu le mări fără să știi autonomia. GUEST_STOP_WAIT=150 # cât așteptăm guest-urile obișnuite ORACLE_SQL_WAIT=60 # cât așteptăm `shutdown immediate` ORACLE_CT_WAIT=60 # cât așteptăm oprirea CT 108 după aceea NODE_DOWN_WAIT=90 # cât așteptăm nodurile secundare SSH_TIMEOUT=5 NOTIFY_TIMEOUT=20 # emailurile nu au voie să consume bateria TEMPLATE_DIR="/etc/pve/notification-templates/default" HOSTNAME=$(hostname) FQDN=$(hostname -f 2>/dev/null || hostname) # Credențiale în afara scriptului, dacă fișierul există. [[ -r /etc/nut/ups-shutdown.conf ]] && source /etc/nut/ups-shutdown.conf DRY_RUN=0 FORCE=0 for arg in "$@"; do case "$arg" in --dry-run) DRY_RUN=1 ;; --force) FORCE=1 ;; -h|--help) sed -n '2,25p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;; esac done # ------------------------------------------------------------------- utilitare log() { echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1" | tee -a "$LOGFILE" 2>/dev/null logger -t ups-shutdown "$1" } do_run() { # execută, sau doar loghează dacă --dry-run if (( DRY_RUN )); then log " [dry-run] $*" return 0 fi "$@" } ssh_node() { local ip=$1; shift ssh -o BatchMode=yes -o ConnectTimeout=$SSH_TIMEOUT -o StrictHostKeyChecking=no \ -o LogLevel=ERROR "root@$ip" "$@" } node_alive() { ping -c 1 -W 2 "$1" >/dev/null 2>&1; } get_ups_info() { echo "Status: $(upsc $UPS_NAME ups.status 2>/dev/null || echo UNKNOWN)" \ "| Baterie: $(upsc $UPS_NAME battery.charge 2>/dev/null || echo '?')%" \ "| Input: $(upsc $UPS_NAME input.voltage 2>/dev/null || echo '?')V" } # Notificare email via PVE::Notify. Mărginită în timp — nu consumăm baterie pe mail. send_notification() { local EVENT_TYPE="$1" EVENT_TITLE="$2" EVENT_DESC="$3" SEVERITY="$4" (( DRY_RUN )) && { log " [dry-run] email: $EVENT_TITLE"; return 0; } local UPS_STATUS BATTERY_CHARGE INPUT_VOLTAGE EVENT_DATE UPS_STATUS=$(upsc $UPS_NAME ups.status 2>/dev/null || echo UNKNOWN) BATTERY_CHARGE=$(upsc $UPS_NAME battery.charge 2>/dev/null || echo 0) INPUT_VOLTAGE=$(upsc $UPS_NAME input.voltage 2>/dev/null || echo 0) EVENT_DATE=$(date '+%Y-%m-%d %H:%M:%S') log "Trimit notificare: $EVENT_TITLE" timeout $NOTIFY_TIMEOUT /usr/bin/perl -I/usr/share/perl5 <>"$LOGFILE" 2>&1 use strict; use warnings; use PVE::Notify; my \$template_data = { 'hostname' => '$FQDN', 'event_date' => '$EVENT_DATE', 'event_type' => '$EVENT_TYPE', 'event_title' => '$EVENT_TITLE', 'event_description' => '$EVENT_DESC', 'event_class' => 'shutdown', 'alert_type' => 'danger', 'ups_status' => '$UPS_STATUS', 'battery_charge' => '$BATTERY_CHARGE', 'input_voltage' => '$INPUT_VOLTAGE', 'action_taken' => 'Shutdown in curs', 'next_steps' => '' }; my \$fields = { 'hostname' => '$HOSTNAME', 'type' => 'ups-power-event' }; eval { PVE::Notify::notify('$SEVERITY', 'ups-power-event', \$template_data, \$fields); print "Notification sent\\n"; }; if (\$@) { print STDERR "Notification failed: \$@\\n"; } EOFPERL return 0 } # ------------------------------------------------------------------- preflight # Log-ul trebuie să existe și să fie scriabil. [[ -f "$LOGFILE" ]] || touch "$LOGFILE" 2>/dev/null chmod 664 "$LOGFILE" 2>/dev/null if [[ "$(id -u)" != "0" ]]; then log "FATAL: scriptul rulează ca $(id -un), nu ca root." log " upssched rulează ca 'nut'; upssched-cmd trebuie să-l invoce prin sudo." log " Vezi /etc/sudoers.d/nut-shutdown." exit 1 fi # O singură instanță: ONBATT și LOWBATT pot declanșa amândouă scriptul. exec 9>"$LOCKFILE" if ! flock -n 9; then log "O instanță rulează deja (lock $LOCKFILE). Ies." exit 0 fi log "========================================" log "UPS SHUTDOWN ORCHESTRAT — START$( ((DRY_RUN)) && echo ' (DRY-RUN)' )" log "$(get_ups_info)" log "========================================" UPS_STATUS=$(upsc $UPS_NAME ups.status 2>/dev/null) if [[ ! $UPS_STATUS =~ (OB|LB) ]]; then if (( FORCE || DRY_RUN )); then log "UPS status '$UPS_STATUS' nu e critic, dar continui (--force/--dry-run)." else log "UPS status '$UPS_STATUS' nu e critic. Abandonez." exit 0 fi fi # ---------------------------------------------- pas 1: quorum + freeze HAVE_QUORUM=0 if pvecm status 2>/dev/null | grep -q 'Quorate: *Yes'; then HAVE_QUORUM=1 log "Pas 1: quorum OK." else log "Pas 1: ATENȚIE — clusterul NU are quorum." log " Nodurile secundare sunt probabil deja moarte (nu sunt pe UPS?)." log " /etc/pve e read-only: nu pot seta freeze și nu pot folosi ha-manager." log " Trec pe oprire directă a guest-urilor, local." fi # `freeze` împiedică HA să relocheze resursele când nodurile se opresc. Fără el, # politica implicită `conditional` transformă fiecare poweroff într-un failover # către un nod care oricum se stinge în secunda următoare — exact incidentul # 2026-01-11 (VM 201 migrat în timpul unei pene de curent). if (( HAVE_QUORUM )); then CURRENT_POLICY=$(grep -oP 'shutdown_policy=\K\w+' /etc/pve/datacenter.cfg 2>/dev/null) if [[ "$CURRENT_POLICY" == "freeze" ]]; then log " shutdown_policy deja pe freeze." else log " setez shutdown_policy=freeze (era: ${CURRENT_POLICY:-nesetat})" if (( ! DRY_RUN )); then if [[ -f /etc/pve/datacenter.cfg ]]; then sed -i '/^ha:/d' /etc/pve/datacenter.cfg printf 'ha: shutdown_policy=freeze\n' >> /etc/pve/datacenter.cfg else printf 'ha: shutdown_policy=freeze\n' > /etc/pve/datacenter.cfg fi fi fi fi send_notification "SHUTDOWN_START" "Shutdown cluster PORNIT" \ "UPS pe baterie. Se inițiază oprirea ordonată a clusterului Proxmox." "error" # ------------------------------------------------- pas 2: inventar guest-uri log "Pas 2: inventar guest-uri..." declare -A HA_SID=() if (( HAVE_QUORUM )); then while read -r sid; do [[ -n "$sid" ]] && HA_SID["$sid"]=1 done < <(ha-manager config 2>/dev/null | grep -oE '^(ct|vm):[0-9]+') fi # Linii: " " GUESTS=() collect_from() { local ip=$1 name=$2 line id st if [[ "$ip" == "local" ]]; then while read -r id st; do [[ "$st" == running ]] && GUESTS+=("ct local $name $id"); done \ < <(pct list 2>/dev/null | awk 'NR>1 {print $1, $2}') while read -r id st; do [[ "$st" == running ]] && GUESTS+=("vm local $name $id"); done \ < <(qm list 2>/dev/null | awk 'NR>1 {print $1, $3}') else node_alive "$ip" || { log " $name ($ip) nu răspunde — îl sar (probabil deja oprit)."; return; } while read -r id st; do [[ "$st" == running ]] && GUESTS+=("ct $ip $name $id"); done \ < <(ssh_node "$ip" "pct list 2>/dev/null | awk 'NR>1 {print \$1, \$2}'") while read -r id st; do [[ "$st" == running ]] && GUESTS+=("vm $ip $name $id"); done \ < <(ssh_node "$ip" "qm list 2>/dev/null | awk 'NR>1 {print \$1, \$3}'") fi } collect_from local "$HOSTNAME" for i in "${!SECONDARY_IPS[@]}"; do collect_from "${SECONDARY_IPS[$i]}" "${SECONDARY_NAMES[$i]}" done log " ${#GUESTS[@]} guest-uri pornite: $(printf '%s ' "${GUESTS[@]##* }")" # ------------------------------------------------- funcții de oprire guest guest_is_stopped() { local type=$1 ip=$2 id=$3 cmd out [[ "$type" == ct ]] && cmd=pct || cmd=qm if [[ "$ip" == "local" ]]; then out=$($cmd status "$id" 2>/dev/null) else out=$(ssh_node "$ip" "$cmd status $id 2>/dev/null") fi [[ "$out" == *stopped* ]] } # Trimite comanda de oprire, fără să aștepte. issue_stop() { local type=$1 ip=$2 id=$3 sid="$1:$3" cmd [[ "$type" == ct ]] && cmd=pct || cmd=qm # Pentru resursele HA folosim ha-manager: din CLI, pct/qm shutdown NU # actualizează state-ul HA, iar CRM-ul poate reporni guest-ul în mijlocul # opririi. Fără quorum ha-manager nu merge, deci cădem pe oprire directă. if (( HAVE_QUORUM )) && [[ -n "${HA_SID[$sid]:-}" ]]; then do_run ha-manager set "$sid" --state stopped >/dev/null 2>&1 & elif [[ "$ip" == "local" ]]; then do_run $cmd shutdown "$id" --timeout $GUEST_STOP_WAIT >/dev/null 2>&1 & else do_run ssh_node "$ip" "$cmd shutdown $id --timeout $GUEST_STOP_WAIT" >/dev/null 2>&1 & fi } force_stop() { local type=$1 ip=$2 id=$3 cmd [[ "$type" == ct ]] && cmd=pct || cmd=qm log " FORȚEZ oprirea $type:$id" if [[ "$ip" == "local" ]]; then do_run $cmd stop "$id" >/dev/null 2>&1 else do_run ssh_node "$ip" "$cmd stop $id" >/dev/null 2>&1 fi } # Așteaptă ca o listă de guest-uri să se oprească; forțează ce rămâne. wait_for_stopped() { local -n _list=$1 local budget=$2 waited=0 remaining (( DRY_RUN )) && { log " [dry-run] aș aștepta oprirea a ${#_list[@]} guest-uri"; return 0; } while (( waited < budget )); do remaining=() for g in "${_list[@]}"; do read -r type ip name id <<<"$g" guest_is_stopped "$type" "$ip" "$id" || remaining+=("$g") done (( ${#remaining[@]} == 0 )) && { log " toate oprite (${waited}s)"; return 0; } sleep 5; waited=$(( waited + 5 )) done log " ${#remaining[@]} guest-uri nu s-au oprit în ${budget}s" for g in "${remaining[@]}"; do read -r type ip name id <<<"$g" force_stop "$type" "$ip" "$id" done return 0 } # ------------------------------- pas 3: oprire consumatori (tot, mai puțin Oracle) log "Pas 3: opresc consumatorii (Oracle rămâne ultimul)..." CONSUMERS=() ORACLE_ENTRY="" for g in "${GUESTS[@]}"; do read -r type ip name id <<<"$g" if [[ "$type" == ct && "$id" == "$ORACLE_CT" ]]; then ORACLE_ENTRY="$g" else CONSUMERS+=("$g") fi done # În paralel: nu depind unul de altul, doar de Oracle. Bateria nu așteaptă. for g in "${CONSUMERS[@]}"; do read -r type ip name id <<<"$g" log " opresc $type:$id pe $name" issue_stop "$type" "$ip" "$id" done wait if (( ${#CONSUMERS[@]} > 0 )); then wait_for_stopped CONSUMERS $GUEST_STOP_WAIT fi # ----------------------------------------------- pas 4: Oracle, curat, la final if [[ -z "$ORACLE_ENTRY" ]]; then log "Pas 4: CT $ORACLE_CT nu rulează — sar peste." else read -r otype oip oname oid <<<"$ORACLE_ENTRY" log "Pas 4: opresc instanța Oracle din CT $oid ($oname)..." if (( DRY_RUN )); then log " [dry-run] shutdown immediate în $ORACLE_DOCKER" else SQL="pct exec $oid -- docker exec $ORACLE_DOCKER bash -c \"printf 'shutdown immediate\nexit\n' | sqlplus -s / as sysdba\"" if [[ "$oip" == "local" ]]; then timeout $ORACLE_SQL_WAIT bash -c "$SQL" >>"$LOGFILE" 2>&1 else timeout $ORACLE_SQL_WAIT ssh_node "$oip" "$SQL" >>"$LOGFILE" 2>&1 fi if (( $? == 0 )); then log " instanța Oracle oprită curat" else log " ATENȚIE: shutdown immediate a eșuat sau a depășit ${ORACLE_SQL_WAIT}s" log " Baza va porni cu instance recovery. Continui — bateria nu așteaptă." fi fi log " opresc containerul CT $oid" issue_stop "$otype" "$oip" "$oid" wait ORACLE_LIST=("$ORACLE_ENTRY") wait_for_stopped ORACLE_LIST $ORACLE_CT_WAIT fi # ------------------------------------------------------- pas 5: verificare if (( HAVE_QUORUM && ! DRY_RUN )); then ACTIVE=$(ha-manager status 2>/dev/null | grep -cE 'started|migrate|relocate|fence') if (( ACTIVE > 0 )); then log "Pas 5: ATENȚIE — $ACTIVE servicii HA încă active. Continui oricum (baterie)." ha-manager status 2>/dev/null | grep -E 'started|migrate|relocate|fence' >>"$LOGFILE" else log "Pas 5: toate serviciile HA sunt oprite. Watchdog-ul nu mai e armat." fi fi # --------------------------------------------- pas 6: oprire noduri secundare log "Pas 6: opresc nodurile secundare..." DOWN_TARGETS=() for i in "${!SECONDARY_IPS[@]}"; do ip="${SECONDARY_IPS[$i]}"; name="${SECONDARY_NAMES[$i]}" if ! node_alive "$ip"; then log " $name ($ip) nu răspunde — deja oprit." continue fi log " trimit poweroff către $name ($ip)" send_notification "SHUTDOWN_NODE" "Shutdown $name trimis" \ "Comanda de oprire a fost trimisă către nodul $name ($ip)." "error" do_run ssh_node "$ip" "systemctl poweroff --no-block" >>"$LOGFILE" 2>&1 DOWN_TARGETS+=("$ip:$name") done if (( ${#DOWN_TARGETS[@]} > 0 )) && (( ! DRY_RUN )); then log " aștept oprirea lor (max ${NODE_DOWN_WAIT}s)..." waited=0 while (( waited < NODE_DOWN_WAIT )); do still=() for t in "${DOWN_TARGETS[@]}"; do node_alive "${t%%:*}" && still+=("${t##*:}") done (( ${#still[@]} == 0 )) && { log " nodurile secundare sunt jos (${waited}s)"; break; } sleep 10; waited=$(( waited + 10 )) done (( ${#still[@]} > 0 )) && log " ${still[*]} încă răspund după ${NODE_DOWN_WAIT}s — continui oricum." fi # ------------------------------------------------ pas 7: oprire UPS + local send_notification "SHUTDOWN_PRIMARY" "Shutdown $HOSTNAME (ULTIMUL NOD)" \ "Se oprește nodul primary $HOSTNAME. UPS-ul se va opri după." "error" log "Pas 7: comand oprirea UPS-ului..." UPS_CMDS=$(upscmd -l $UPS_NAME 2>/dev/null) if grep -q 'shutdown.stayoff' <<<"$UPS_CMDS"; then log " shutdown.stayoff (oprire completă)" do_run upscmd -u "$UPS_USER" -p "$UPS_PASS" $UPS_NAME shutdown.stayoff >>"$LOGFILE" 2>&1 elif grep -q 'shutdown.return' <<<"$UPS_CMDS"; then log " shutdown.return (repornire la revenirea curentului)" do_run upscmd -u "$UPS_USER" -p "$UPS_PASS" $UPS_NAME shutdown.return >>"$LOGFILE" 2>&1 else log " ATENȚIE: nicio comandă de oprire disponibilă pe UPS" fi log "========================================" log "UPS SHUTDOWN ORCHESTRAT — TERMINAT" log "$(get_ups_info)" log "========================================" if (( DRY_RUN )); then log "DRY-RUN: nodul local NU se oprește." exit 0 fi log "Opresc nodul local ($HOSTNAME) — ultimul." systemctl poweroff --no-block exit 0