#!/usr/bin/env python3 """infra_actions.py -- nucleul tabului Infrastructura (dashboard + /infra Discord). Interfata comuna (vezi planul de implementare, WP4): ACTIONS: dict[str, dict] registrul actiunilor status(fresh=False) -> dict stare cluster, cache 30s check(action_id, st, dry_run) -> (bool, str) precondi\u021bii, functie pura run(action_id, dry_run) -> dict executa o actiune stop(action_id) -> dict opreste un job systemd-run log_tail(action_id, n=60) -> dict ultimele linii din jurnal maintenance_pending(st) -> bool cluster sus, dar in mentenanta post_startup_instructions() -> str | None mesaj Discord + pin, la oprire Nu se ating nodurile decat prin `security/infra` (jurnalizat) pentru actiuni, si direct prin ssh read-only pentru status() (nu umple infra.log la fiecare 30s). Nimic din acest modul nu ruleaza vreo comanda distructiva in mod implicit: toate actiunile reale trec prin check() inainte, iar cele care opresc/pornesc ceva cer `--yes`/`confirm` mai sus, in dashboard/bot. """ from __future__ import annotations import concurrent.futures import json import re import shlex import subprocess import sys import threading import time import urllib.error import urllib.request from pathlib import Path _HERE = Path(__file__).resolve().parent if str(_HERE) not in sys.path: sys.path.insert(0, str(_HERE)) import config # noqa: E402 INFRA_BIN = str(_HERE / "security" / "infra") # Cei 3 noduri Proxmox, adrese fixe (acelasi tabel ca in security/infra). NODES: dict[str, str] = { "pve1": "10.0.20.200", "pvemini": "10.0.20.201", "pveelite": "10.0.20.202", } SSH_STATUS_OPTS = [ "-o", "BatchMode=yes", "-o", "StrictHostKeyChecking=accept-new", "-o", "ConnectTimeout=5", ] _STATUS_TIMEOUT_S = 20 _CACHE_TTL_S = 30.0 # ---------------------------------------------------------------- registrul ACTIONS: dict[str, dict] = { "stare": { "label": "Stare cluster", "desc": "Cvorum, noduri, HA, UPS, backup, SSL.", "host": "local", "cmd": None, "dry": None, "job": False, "confirm": False, "pre": None, }, "sarcini": { "label": "Sarcini active", "desc": "Job-urile systemd-run pornite din tab.", "host": "local", "cmd": None, "dry": None, "job": False, "confirm": False, "pre": None, }, "oprire": { "label": "Oprire cluster", "desc": "Salveaza starea, opreste guest-urile (mai putin CT 171), trimite instructiunile de pornire, opreste nodurile.", "host": "pvemini", "cmd": "/opt/scripts/cluster-shutdown.sh --allow-on-node --yes", "dry": "/opt/scripts/cluster-shutdown.sh --allow-on-node --dry-run", "job": True, "confirm": True, "pre": "real: 3/3 noduri online, cvorum, niciun job activ; dry: pvemini online", }, "pornire": { "label": "Pornire cluster", "desc": "Restaureaza crontab-ul si porneste guest-urile care erau active.", "host": "pvemini", "cmd": "/opt/scripts/cluster-startup.sh --allow-on-node --yes", "dry": "/opt/scripts/cluster-startup.sh --allow-on-node --dry-run", "job": True, "confirm": True, "pre": "pvemini online, cvorum, niciun job de oprire/pornire activ", }, "wol": { "label": "Trimite Wake-on-LAN", "desc": "Ruleaza C:\\wolcluster.bat de pe statia de administrare (trezeste cele 3 noduri).", "host": "oracle-prod-admin", "cmd": "cmd /c C:\\wolcluster.bat", "dry": None, "job": False, "confirm": False, "pre": "cel putin un nod offline", }, "ups-test": { "label": "Test lunar UPS", "desc": "Descarcare controlata a bateriei UPS + raport.", "host": "pvemini", "cmd": "/opt/scripts/ups-monthly-test.sh", "dry": None, "job": True, "confirm": True, "pre": "UPS OL, incarcare >= 90%, niciun job UPS activ", }, "ups-simulare": { "label": "Simulare oprire UPS", "desc": "Ruleaza scriptul de oprire la panica UPS in --dry-run --force (nimic real).", "host": "pvemini", "cmd": "runuser -u nut -- sudo -n /usr/local/bin/ups-shutdown-cluster.sh --dry-run --force", "dry": None, "job": True, "confirm": False, "pre": "pvemini online", }, "ups-istoric": { "label": "Istoric baterie UPS", "desc": "Ultimele randuri din trendul de baterie.", "host": "pvemini", "cmd": "tail -n 13 /var/log/ups-battery-trend.csv", "dry": None, "job": False, "confirm": False, "pre": "pvemini online", }, "dr-test": { "label": "Test DR saptamanal (VM 109)", "desc": "Porneste VM 109, verifica restore-ul, il opreste la final.", "host": "vm109", "cmd": ('touch /var/run/vm109-debug.flag; /opt/scripts/weekly-dr-test-proxmox.sh; ' 'rc=$?; rm -f /var/run/vm109-debug.flag; exit $rc'), "dry": None, "job": True, "confirm": True, "pre": "VM 109 oprit, niciun job dr-test/dr-patch/vm109 activ", }, "dr-patch": { "label": "Fereastra de patch VM 109", "desc": "Porneste VM 109 in fereastra de update Windows.", "host": "vm109", "cmd": "/opt/scripts/vm109-patch-window.sh --now", "dry": None, "job": True, "confirm": True, "pre": "VM 109 oprit, niciun job dr-test/dr-patch/vm109 activ", }, "vm109-pornire": { "label": "Porneste VM 109", "desc": "Pornire manuala, in afara ferestrelor programate.", "host": "vm109", "cmd": "touch /var/run/vm109-debug.flag && qm start 109", "dry": None, "job": False, "confirm": True, "pre": "VM 109 oprit, niciun job dr activ", }, "vm109-oprire": { "label": "Opreste VM 109", "desc": "Oprire manuala.", "host": "vm109", "cmd": "qm stop 109; rm -f /var/run/vm109-debug.flag", "dry": None, "job": False, "confirm": True, "pre": "VM 109 pornit, niciun job dr-test/dr-patch activ", }, "backup-pe-pvemini": { "label": "Failover backup -> pvemini", "desc": "pveelite e jos: pvemini preia rolul de tinta de backup Oracle.", "host": "pvemini", "cmd": "/opt/scripts/failover-dr-to-pvemini.sh", "dry": None, "job": True, "confirm": True, "pre": "pveelite offline, pvemini online, failover-ul nu e deja activ", }, "backup-pe-pveelite": { "label": "Failback backup -> pveelite", "desc": "Revine la topologia normala dupa ce pveelite e iar online.", "host": "pvemini", "cmd": "/opt/scripts/failback-dr-to-pveelite.sh", "dry": None, "job": True, "confirm": True, "pre": "pveelite online, failover-ul e activ", }, } _VALID_ID = re.compile(r"^[a-z0-9-]+$") for _id, _spec in ACTIONS.items(): assert _VALID_ID.match(_id), _id assert _spec["host"] in ("local", "pvemini", "pveelite", "vm109", "oracle-prod-admin"), _id STARTUP_TEXT = ( "Cluster oprit. Pornire:\n" "1. JuiceSSH (prin Tailscale) -> 10.0.20.36 port 22122\n" "2. C:\\wolcluster.bat (trezeste pve1, pvemini, pveelite prin Wake-on-LAN)\n" "3. Asteapta mesajul \u201eCluster sus\u201c aici, in Discord\n" "Rezerva daca WoL nu merge: ssh root@10.0.20.201 " "/opt/scripts/cluster-startup.sh --allow-on-node --yes" ) # ------------------------------------------------------------------- status _STATUS_SH = r""" set -o pipefail echo @@cluster pvesh get /cluster/resources --output-format json 2>&1 echo @@quorum pvecm status 2>&1 echo @@ha ha-manager status 2>&1 echo @@repl pvesh get /nodes/$(hostname)/replication --output-format json 2>&1 echo @@jobs systemctl list-units --plain --no-legend --all 'infra-*' 2>&1 ls -t /var/log/infra-actions 2>/dev/null | head -20 if command -v upsc >/dev/null 2>&1 && upsc nutdev1 >/tmp/.infra-ups 2>&1; then echo @@ups cat /tmp/.infra-ups rm -f /tmp/.infra-ups fi if command -v zfs >/dev/null 2>&1 && zfs list rpool/oracle-backups >/dev/null 2>&1; then echo @@zfs zfs get -H -o value readonly rpool/oracle-backups 2>&1 fi echo @@maint crontab -l 2>/dev/null | grep -c '^#MENTENANTA' if [ -f /opt/scripts/monitor-ssl-certificates.sh ]; then echo @@ssl for d in $(sed -n '/^DOMAINS=(/,/^)/p' /opt/scripts/monitor-ssl-certificates.sh | grep -oE '"[^"]+"' | tr -d '"'); do end=$(echo | timeout 5 openssl s_client -connect "$d:443" -servername "$d" 2>/dev/null \ | openssl x509 -noout -enddate 2>/dev/null | cut -d= -f2) if [ -n "$end" ]; then days=$(( ($(date -d "$end" +%s) - $(date +%s)) / 86400 )) echo "$d $days" fi done fi if [ -d /mnt/pve/oracle-backups/ROA/autobackup ]; then echo @@backup find /mnt/pve/oracle-backups/ROA/autobackup -maxdepth 1 -type f \ \( -name '*FULL*.BKP' -o -name 'L0_*.BKP' \) -printf '%T@ %p\n' 2>/dev/null | sort -nr | head -1 find /mnt/pve/oracle-backups/ROA/autobackup -maxdepth 1 -type f \ \( -name '*INCR*.BKP' -o -name '*INCREMENTAL*.BKP' -o -name '*CUMULATIVE*.BKP' -o -name 'L1_*.BKP' \) \ -printf '%T@ %p\n' 2>/dev/null | sort -nr | head -1 fi """ def _split_sections(text: str) -> dict[str, str]: sections: dict[str, str] = {} cur: str | None = None buf: list[str] = [] for line in text.splitlines(): if line.startswith("@@"): if cur is not None: sections[cur] = "\n".join(buf) cur = line[2:].strip() buf = [] elif cur is not None: buf.append(line) if cur is not None: sections[cur] = "\n".join(buf) return sections def _ssh_status_one(ip: str) -> str: """Read-only: ssh direct, ocoleste security/infra (nu jurnalizeaza).""" proc = subprocess.run( ["ssh", *SSH_STATUS_OPTS, f"root@{ip}", "bash", "-s"], input=_STATUS_SH, capture_output=True, text=True, timeout=_STATUS_TIMEOUT_S, ) return proc.stdout def _collect() -> dict[str, "str | Exception"]: outputs: dict[str, "str | Exception"] = {} with concurrent.futures.ThreadPoolExecutor(max_workers=len(NODES)) as pool: futures = {pool.submit(_ssh_status_one, ip): name for name, ip in NODES.items()} for fut in concurrent.futures.as_completed(futures, timeout=_STATUS_TIMEOUT_S + 5): name = futures[fut] try: outputs[name] = fut.result() except Exception as exc: # timeout, connection refused, etc. outputs[name] = exc return outputs def parse_status(outputs: dict[str, "str | Exception"]) -> dict: """Pura: transforma iesirile brute per nod in forma din contractul status().""" errors: dict[str, str] = {} sections: dict[str, dict[str, str]] = {} for host, out in outputs.items(): if isinstance(out, Exception): errors[host] = str(out) continue sections[host] = _split_sections(out) # --- cluster resources (de la primul nod care raspunde) --- cluster_json: list[dict] | None = None for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("cluster", "").strip() if not txt: continue try: cluster_json = json.loads(txt) break except (ValueError, TypeError): continue nodes: list[dict] = [] guests_running = guests_total = 0 vm109: dict | None = None if cluster_json: for name in NODES: entry = next((e for e in cluster_json if e.get("type") == "node" and e.get("node") == name), None) nodes.append({ "name": name, "ip": NODES[name], "online": name in sections, "cpu": (entry or {}).get("cpu", 0.0), "mem": (entry or {}).get("mem", 0), "maxmem": (entry or {}).get("maxmem", 0), "uptime": (entry or {}).get("uptime", 0), "guests": sum(1 for e in cluster_json if e.get("type") in ("qemu", "lxc") and e.get("node") == name), }) for e in cluster_json: if e.get("type") in ("qemu", "lxc"): guests_total += 1 if e.get("status") == "running": guests_running += 1 if e.get("type") == "qemu" and str(e.get("vmid")) == "109": vm109 = {"node": e.get("node"), "status": e.get("status")} else: for name in NODES: nodes.append({ "name": name, "ip": NODES[name], "online": name in sections, "cpu": 0.0, "mem": 0, "maxmem": 0, "uptime": 0, "guests": 0, }) # --- quorum --- quorum = {"quorate": False, "votes": 0, "expected": 0} for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("quorum", "") if not txt: continue m_q = re.search(r"^Quorate:\s*(\w+)", txt, re.MULTILINE) m_v = re.search(r"^Total votes:\s*(\d+)", txt, re.MULTILINE) m_e = re.search(r"^Expected votes:\s*(\d+)", txt, re.MULTILINE) if m_q or m_v or m_e: quorum = { "quorate": bool(m_q and m_q.group(1).lower() == "yes"), "votes": int(m_v.group(1)) if m_v else 0, "expected": int(m_e.group(1)) if m_e else 0, } break # --- HA --- ha: list[dict] = [] for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("ha", "") if not txt: continue for line in txt.splitlines(): m = re.match(r"^service\s+(\S+)\s+\((\S+),\s*(\w+)\)", line.strip()) if m: ha.append({"sid": m.group(1), "node": m.group(2), "state": m.group(3)}) if ha: break # --- replicare (merge peste toate nodurile, dedup pe id) --- repl_jobs: dict[str, dict] = {} for host, secs in sections.items(): txt = secs.get("repl", "").strip() if not txt: continue try: arr = json.loads(txt) except (ValueError, TypeError): continue if isinstance(arr, list): for job in arr: jid = job.get("id") if jid: repl_jobs[jid] = job replication = { "ok": sum(1 for j in repl_jobs.values() if not j.get("fail_count")), "total": len(repl_jobs), "last": max((j.get("last_sync", 0) for j in repl_jobs.values()), default=0), "failed": [jid for jid, j in repl_jobs.items() if j.get("fail_count")], } # --- jobs (systemd-run infra-*) --- jobs: dict[tuple[str, str], dict] = {} for host, secs in sections.items(): txt = secs.get("jobs", "") if not txt: continue for line in txt.splitlines(): m = re.match(r"^infra-(\S+)\.service\s+\S+\s+(\S+)\s+(\S+)", line.strip()) if m: jid, load_state, active_state = m.group(1), m.group(2), m.group(3) key = (host, jid) jobs.setdefault(key, {"id": jid, "host": host, "active": False, "log": None}) jobs[key]["active"] = active_state == "running" continue m2 = re.match(r"^([a-z0-9-]+)-\d{8}-\d{6}\.log$", line.strip()) if m2: jid = m2.group(1) key = (host, jid) entry = jobs.setdefault(key, {"id": jid, "host": host, "active": False, "log": None}) if entry["log"] is None: entry["log"] = f"/var/log/infra-actions/{line.strip()}" job_list = list(jobs.values()) # --- UPS --- ups = None for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("ups") if txt is None: continue kv = {} for line in txt.splitlines(): if ":" in line: k, _, v = line.partition(":") kv[k.strip()] = v.strip() if kv: try: charge = float(kv.get("battery.charge", "0")) except ValueError: charge = 0.0 runtime_raw = kv.get("battery.runtime") runtime_s = None if runtime_raw: try: runtime_s = int(float(runtime_raw)) except ValueError: runtime_s = None ups = {"status": kv.get("ups.status", ""), "charge": charge, "runtime_s": runtime_s} break # --- ZFS readonly -> dr_failover_active --- dr_failover_active = None for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("zfs") if txt is None: continue val = txt.strip() if val: dr_failover_active = val == "off" break # --- maintenance flag --- maint = 0 for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("maint") if txt is None: continue try: maint = int(txt.strip() or "0") except ValueError: maint = 0 break # --- SSL --- ssl = None best = None for host in ("pvemini", "pve1", "pveelite"): txt = sections.get(host, {}).get("ssl") if not txt: continue for line in txt.splitlines(): parts = line.split() if len(parts) == 2: domain, days_raw = parts try: days = int(days_raw) except ValueError: continue if best is None or days < best[1]: best = (domain, days) break if best: ssl = {"min_days": best[1], "domain": best[0]} # --- backup age --- backup = None for host in ("pveelite", "pvemini", "pve1"): txt = sections.get(host, {}).get("backup") if txt is None: continue lines = [l for l in txt.splitlines() if l.strip()] now = time.time() def _age_h(line: str) -> float | None: parts = line.split(None, 1) if not parts: return None try: return (now - float(parts[0])) / 3600.0 except ValueError: return None full_age = _age_h(lines[0]) if len(lines) >= 1 else None cum_age = _age_h(lines[1]) if len(lines) >= 2 else None backup = {"full_age_h": full_age, "cum_age_h": cum_age} break return { "at": time.time(), "errors": errors, "nodes": nodes, "quorum": quorum, "guests": {"running": guests_running, "total": guests_total}, "ha": ha, "replication": replication, "ups": ups, "backup": backup, "ssl": ssl, "vm109": vm109, "dr_failover_active": dr_failover_active, "maint": maint, "jobs": job_list, } _cache: dict = {"at": 0.0, "data": None} _cache_lock = threading.Lock() def _empty_status(error: str) -> dict: return { "at": time.time(), "errors": {"_": error}, "nodes": [], "quorum": {"quorate": False, "votes": 0, "expected": 0}, "guests": {"running": 0, "total": 0}, "ha": [], "replication": {"ok": 0, "total": 0, "last": 0, "failed": []}, "ups": None, "backup": None, "ssl": None, "vm109": None, "dr_failover_active": None, "maint": 0, "jobs": [], } def status(fresh: bool = False) -> dict: """Stare cluster, cache 30s. Nu arunca niciodata.""" now = time.time() cached = _cache["data"] if not fresh and cached is not None and now - _cache["at"] < _CACHE_TTL_S: return cached with _cache_lock: now2 = time.time() cached = _cache["data"] # alt thread a reimprospatat deja cat am asteptat pe lock if cached is not None and now2 - _cache["at"] < _CACHE_TTL_S: return cached try: data = parse_status(_collect()) except Exception as exc: # pragma: no cover - plasa de siguranta data = _empty_status(str(exc)) try: data["actions"] = { aid: {"ok": (r := check(aid, data, False))[0], "reason": r[1]} for aid in ACTIONS } except Exception: # pragma: no cover data["actions"] = {} data["startup_text"] = STARTUP_TEXT _cache["data"] = data _cache["at"] = time.time() return data # ------------------------------------------------------------------- check def _node(st: dict, name: str) -> dict | None: return next((n for n in st.get("nodes", []) if n["name"] == name), None) def _online(st: dict, name: str) -> bool: n = _node(st, name) return bool(n and n["online"]) def _active_jobs(st: dict) -> list[dict]: return [j for j in st.get("jobs", []) if j.get("active")] def _job_active(st: dict, *prefixes: str) -> bool: return any(j["id"].startswith(prefixes) for j in _active_jobs(st)) def check(action_id: str, st: dict, dry_run: bool) -> tuple[bool, str]: """Precondi\u021bii pentru o actiune. Functie pura, nu atinge reteaua.""" if action_id not in ACTIONS: return False, "actiune necunoscuta" if action_id in ("stare", "sarcini"): return True, "" if action_id == "oprire": if dry_run: return (True, "") if _online(st, "pvemini") else (False, "pvemini nu raspunde") nodes = st.get("nodes", []) if len(nodes) != 3 or not all(n["online"] for n in nodes): return False, "nu toate cele 3 noduri sunt online" if not st.get("quorum", {}).get("quorate"): return False, "clusterul nu are cvorum" if _active_jobs(st): return False, "exista un job activ" return True, "" if action_id == "pornire": if not _online(st, "pvemini"): return False, "pvemini nu raspunde" if dry_run: return True, "" if not st.get("quorum", {}).get("quorate"): return False, "clusterul nu are cvorum" if _job_active(st, "oprire", "pornire"): return False, "exista deja un job de oprire/pornire" return True, "" if action_id == "wol": nodes = st.get("nodes", []) if not nodes: return False, "stare necunoscuta" if len(nodes) == 3 and all(n["online"] for n in nodes): return False, "toate nodurile sunt deja online" return True, "" if action_id == "ups-test": ups = st.get("ups") if not ups: return False, "UPS necunoscut" if ups.get("status") != "OL": return False, f"UPS nu e OL (status={ups.get('status')})" if ups.get("charge", 0) < 90: return False, f"incarcare sub 90% ({ups.get('charge')}%)" if _job_active(st, "ups-test"): return False, "testul UPS ruleaza deja" return True, "" if action_id in ("ups-simulare", "ups-istoric"): return (True, "") if _online(st, "pvemini") else (False, "pvemini nu raspunde") if action_id in ("dr-test", "dr-patch"): if not st.get("vm109"): return False, "stare VM 109 necunoscuta" if st["vm109"].get("status") != "stopped": return False, "VM 109 e pornit" if _job_active(st, "dr-test", "dr-patch", "vm109"): return False, "exista deja un job DR activ" return True, "" if action_id == "vm109-pornire": if not st.get("vm109"): return False, "stare VM 109 necunoscuta" if st["vm109"].get("status") != "stopped": return False, "VM 109 e deja pornit" if _job_active(st, "dr-test", "dr-patch", "vm109"): return False, "exista un job DR activ" return True, "" if action_id == "vm109-oprire": if not st.get("vm109"): return False, "stare VM 109 necunoscuta" if st["vm109"].get("status") != "running": return False, "VM 109 e deja oprit" if _job_active(st, "dr-test", "dr-patch"): return False, "exista un job DR activ" return True, "" if action_id == "backup-pe-pvemini": if _online(st, "pveelite"): return False, "pveelite e online" if not _online(st, "pvemini"): return False, "pvemini nu raspunde" if st.get("dr_failover_active") is True: return False, "failover-ul e deja activ" return True, "" if action_id == "backup-pe-pveelite": if not _online(st, "pveelite"): return False, "pveelite nu e online" if st.get("dr_failover_active") is not True: return False, "failover-ul nu e activ" return True, "" return False, "actiune fara verificare" # pragma: no cover - completitudine # --------------------------------------------------------------------- run def _resolve_host(spec_host: str, st: dict) -> str: if spec_host == "vm109": vm109 = st.get("vm109") or {} return vm109.get("node") or "pveelite" return spec_host def run(action_id: str, dry_run: bool) -> dict: if action_id not in ACTIONS: return {"ok": False, "code": 400, "error": f"actiune necunoscuta: {action_id}"} spec = ACTIONS[action_id] st = status(fresh=True) ok, reason = check(action_id, st, dry_run) if not ok: return {"ok": False, "code": 412, "host": spec["host"], "error": reason} cmd = spec["dry"] if dry_run else spec["cmd"] if dry_run and cmd is None: return {"ok": False, "code": 400, "error": "aceasta actiune nu are simulare"} if cmd is None: return {"ok": True, "code": 200, "host": spec["host"]} host = _resolve_host(spec["host"], st) warning = None if action_id == "oprire" and not dry_run: warning = post_startup_instructions() if spec["job"]: result = _run_job(action_id, host, cmd) else: result = _run_sync(host, cmd) if warning: result["warning"] = warning return result def _run_job(action_id: str, host: str, cmd: str) -> dict: remote_cmd = cmd + '; echo "[infra] cod iesire: $?"' launcher = ( f"mkdir -p /var/log/infra-actions && " f"L=/var/log/infra-actions/{action_id}-$(date +%Y%m%d-%H%M%S).log && " f"systemd-run --unit=infra-{action_id} --collect --quiet " f"--setenv=INFRA_LOG=$L -p StandardOutput=append:$L -p StandardError=append:$L " f"/bin/bash -c {shlex.quote(remote_cmd)} && echo $L" ) try: proc = subprocess.run( [sys.executable, INFRA_BIN, host, launcher], capture_output=True, text=True, timeout=30, ) except subprocess.TimeoutExpired: return {"ok": False, "code": 500, "host": host, "error": "timeout la lansarea job-ului"} combined = proc.stdout + proc.stderr if proc.returncode != 0: if "already exists" in combined or "already loaded" in combined: return {"ok": False, "code": 409, "host": host, "error": "job deja activ"} return {"ok": False, "code": 500, "host": host, "error": combined.strip()[:2000]} lines = [l for l in proc.stdout.strip().splitlines() if l.strip()] log_path = lines[-1] if lines else None return {"ok": True, "code": 200, "host": host, "log": log_path} def _run_sync(host: str, cmd: str) -> dict: try: proc = subprocess.run( [sys.executable, INFRA_BIN, host, cmd], capture_output=True, text=True, timeout=120, ) except subprocess.TimeoutExpired: return {"ok": False, "code": 500, "host": host, "error": "timeout"} ok = proc.returncode == 0 out = {"ok": ok, "code": 200 if ok else 500, "host": host, "output": proc.stdout} if not ok: out["error"] = proc.stderr.strip()[:2000] return out def stop(action_id: str) -> dict: if action_id not in ACTIONS: return {"ok": False, "code": 400, "error": "actiune necunoscuta"} st = status() host = _resolve_host(ACTIONS[action_id]["host"], st) try: proc = subprocess.run( [sys.executable, INFRA_BIN, host, f"systemctl stop infra-{action_id}"], capture_output=True, text=True, timeout=30, ) except subprocess.TimeoutExpired: return {"ok": False, "code": 500, "host": host, "error": "timeout"} ok = proc.returncode == 0 out = {"ok": ok, "code": 200 if ok else 500, "host": host} if not ok: out["error"] = proc.stderr.strip()[:2000] return out def log_tail(action_id: str, n: int = 60) -> dict: if action_id not in ACTIONS: return {"file": None, "lines": [], "error": "actiune necunoscuta"} st = status() host = _resolve_host(ACTIONS[action_id]["host"], st) remote = ( f'F=$(ls -t /var/log/infra-actions/{action_id}-*.log 2>/dev/null | head -1); ' f'[ -n "$F" ] && echo "@@file $F" && tail -n {int(n)} "$F"' ) try: proc = subprocess.run( [sys.executable, INFRA_BIN, host, remote], capture_output=True, text=True, timeout=30, ) except subprocess.TimeoutExpired: return {"file": None, "lines": [], "error": "timeout"} out = proc.stdout if not out.startswith("@@file "): return {"file": None, "lines": []} first, _, rest = out.partition("\n") return {"file": first[len("@@file "):].strip(), "lines": rest.splitlines()} def maintenance_pending(st: dict) -> bool: """Cluster sus 3/3, cvorum complet, dar inca in starea de dupa oprire (crontab-ul are inca marcajul #MENTENANTA) si niciun job de oprire/pornire in desfasurare -- adica e nevoie sa se dea 'pornire' ca sa se termine.""" q = st.get("quorum", {}) if not q.get("quorate"): return False if q.get("votes", 0) < 3: return False if not st.get("maint", 0): return False if _job_active(st, "oprire", "pornire"): return False return True def post_startup_instructions() -> str | None: """Posteaza (si incearca sa fixeze) instructiunile de pornire in Discord. Nu blocheaza actiunea de oprire: orice eroare devine mesaj de avertisment. """ token = config.get("DISCORD_TOKEN", "") chans = config.get_list("DISCORD_CHANNEL_IDS") or config.get_list("DISCORD_CHANNEL_ID") if not token or not chans: return "lipseste DISCORD_TOKEN sau canalul de Discord" channel_id = chans[0] try: req = urllib.request.Request( f"https://discord.com/api/v10/channels/{channel_id}/messages", data=json.dumps({"content": STARTUP_TEXT}).encode("utf-8"), method="POST", headers={"Authorization": f"Bot {token}", "Content-Type": "application/json"}, ) with urllib.request.urlopen(req, timeout=10) as resp: msg = json.loads(resp.read().decode("utf-8")) except (urllib.error.URLError, urllib.error.HTTPError, ValueError, OSError) as exc: return f"nu s-a putut posta mesajul de pornire: {exc}" msg_id = msg.get("id") if not msg_id: return "raspuns Discord fara id de mesaj" try: pin_req = urllib.request.Request( f"https://discord.com/api/v10/channels/{channel_id}/pins/{msg_id}", method="PUT", headers={"Authorization": f"Bot {token}"}, ) urllib.request.urlopen(pin_req, timeout=10) except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc: return f"mesaj postat, dar fixarea (pin) a esuat: {exc}" return None