Files
atm-curs-dl/download.py
Claude Agent b239f04eb6 Proiect descărcare + transcriere curs ATM (trading, Bogdan Jinga)
Pipeline audio-only (fără video pe disc) prin ffmpeg direct din URL CloudFront,
transcriere cu faster-whisper (CPU), sumarizări structurate cu diagrame SVG
la scară reală (STYLE.md documentează toate cerințele de format).
2026-09-11 07:36:17 +00:00

163 lines
5.7 KiB
Python

"""
Download all lecture videos from the ATM trading course (systeme.io).
Login is a JSON POST (session cookie auth), course structure comes from
systeme.io's own JSON API (no HTML scraping needed), and each lecture's
video is an unsigned CloudFront MP4 URL -> plain streaming HTTP download
with resume support (Range header) since files can exceed 1GB.
"""
import argparse
import json
import logging
import os
import sys
import time
from pathlib import Path
import requests
from dotenv import load_dotenv
BASE = "https://bogdan-jinga.systeme.io"
COURSE_SLUG = "atm"
MAX_RETRIES = 3
RETRY_BACKOFF = [5, 15, 30]
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
log = logging.getLogger(__name__)
def login(session: requests.Session, email: str, password: str) -> None:
resp = session.post(f"{BASE}/api/security/login", json={"email": email, "password": password})
resp.raise_for_status()
def get_course_id(session: requests.Session) -> int:
resp = session.get(f"{BASE}/api/membership/course/{COURSE_SLUG}")
resp.raise_for_status()
return resp.json()["id"]
def get_menu(session: requests.Session, course_id: int) -> list[dict]:
resp = session.get(f"{BASE}/api/membership/course/{course_id}/menu")
resp.raise_for_status()
return resp.json()
MP4_RE = __import__("re").compile(r'https://[^"\\]+\.mp4')
def get_lecture_video_url(session: requests.Session, lecture_id: int) -> str | None:
"""
The lecture API sometimes returns a "files" dict with a CloudFront path,
sometimes inlines the same URL in the raw lecture HTML instead -> check
both, HTML regex as the reliable fallback.
"""
resp = session.get(f"{BASE}/api/membership/lecture/{lecture_id}")
resp.raise_for_status()
data = resp.json()
for f in (data.get("files") or {}).values():
if f.get("type") == "video":
return f["path"]
m = MP4_RE.search(data.get("html") or "")
return m.group(0) if m else None
def slugify(text: str) -> str:
import re
text = re.sub(r"[^\w\s-]", "", text.strip(), flags=re.UNICODE)
return re.sub(r"[\s_-]+", "_", text)[:100] or "untitled"
def download_video(session: requests.Session, url: str, dest: Path) -> bool:
"""Streaming download with resume (Range) and retry."""
tmp = dest.with_suffix(dest.suffix + ".tmp")
for attempt in range(MAX_RETRIES):
try:
headers = {}
mode = "wb"
existing = tmp.stat().st_size if tmp.exists() else 0
if existing:
headers["Range"] = f"bytes={existing}-"
mode = "ab"
with session.get(url, headers=headers, stream=True, timeout=300) as resp:
if resp.status_code not in (200, 206):
resp.raise_for_status()
with open(tmp, mode) as f:
for chunk in resp.iter_content(chunk_size=4 * 1024 * 1024):
f.write(chunk)
tmp.rename(dest)
log.info(f" Downloaded: {dest.name} ({dest.stat().st_size / 1_000_000:.1f} MB)")
return True
except Exception as e:
wait = RETRY_BACKOFF[attempt] if attempt < len(RETRY_BACKOFF) else 30
log.warning(f" Attempt {attempt + 1}/{MAX_RETRIES} failed for {dest.name}: {e}")
if attempt < MAX_RETRIES - 1:
time.sleep(wait)
log.error(f" FAILED after {MAX_RETRIES} attempts: {dest.name}")
return False
def main():
p = argparse.ArgumentParser(description="Download ATM course lecture videos")
p.add_argument("--out", default="videos", help="Output directory")
args = p.parse_args()
load_dotenv()
email = os.getenv("ATM_EMAIL", "")
password = os.getenv("ATM_PASSWORD", "")
if not email or not password:
log.error("Set ATM_EMAIL and ATM_PASSWORD in .env")
sys.exit(1)
out_dir = Path(args.out)
out_dir.mkdir(parents=True, exist_ok=True)
manifest_path = out_dir / "manifest.json"
manifest = json.loads(manifest_path.read_text()) if manifest_path.exists() else {}
session = requests.Session()
session.headers.update({"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"})
log.info("Logging in...")
login(session, email, password)
course_id = get_course_id(session)
log.info(f"Course id: {course_id}")
modules = get_menu(session, course_id)
downloaded = skipped = failed = 0
for mod_idx, mod in enumerate(modules, 1):
mod_dir = out_dir / f"{mod_idx:02d}_{slugify(mod['name'])}"
for lec_idx, lec in enumerate(mod["lectures"], 1):
title = lec["name"]
stem = f"{lec_idx:02d}_{slugify(title)}"
key = str(lec["id"])
video_url = get_lecture_video_url(session, lec["id"])
if not video_url:
log.info(f" No video (text/quiz/etc): {mod['name']} / {title}")
continue
dest = mod_dir / f"{stem}{Path(video_url).suffix}"
if dest.exists() and dest.stat().st_size > 1_000_000:
log.info(f" Skipping (exists): {dest.name}")
skipped += 1
manifest[key] = {"title": title, "path": str(dest), "status": "complete"}
continue
mod_dir.mkdir(parents=True, exist_ok=True)
ok = download_video(session, video_url, dest)
manifest[key] = {"title": title, "path": str(dest), "status": "complete" if ok else "failed"}
manifest_path.write_text(json.dumps(manifest, indent=2, ensure_ascii=False))
downloaded += 1 if ok else 0
failed += 0 if ok else 1
log.info("=" * 60)
log.info(f"Downloaded {downloaded}, skipped {skipped}, failed {failed}.")
if failed:
sys.exit(1)
if __name__ == "__main__":
main()