""" Download all lecture videos from the ATM trading course (systeme.io). Login is a JSON POST (session cookie auth), course structure comes from systeme.io's own JSON API (no HTML scraping needed), and each lecture's video is an unsigned CloudFront MP4 URL -> plain streaming HTTP download with resume support (Range header) since files can exceed 1GB. """ import argparse import json import logging import os import sys import time from pathlib import Path import requests from dotenv import load_dotenv BASE = "https://bogdan-jinga.systeme.io" COURSE_SLUG = "atm" MAX_RETRIES = 3 RETRY_BACKOFF = [5, 15, 30] logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s") log = logging.getLogger(__name__) def login(session: requests.Session, email: str, password: str) -> None: resp = session.post(f"{BASE}/api/security/login", json={"email": email, "password": password}) resp.raise_for_status() def get_course_id(session: requests.Session) -> int: resp = session.get(f"{BASE}/api/membership/course/{COURSE_SLUG}") resp.raise_for_status() return resp.json()["id"] def get_menu(session: requests.Session, course_id: int) -> list[dict]: resp = session.get(f"{BASE}/api/membership/course/{course_id}/menu") resp.raise_for_status() return resp.json() MP4_RE = __import__("re").compile(r'https://[^"\\]+\.mp4') def get_lecture_video_url(session: requests.Session, lecture_id: int) -> str | None: """ The lecture API sometimes returns a "files" dict with a CloudFront path, sometimes inlines the same URL in the raw lecture HTML instead -> check both, HTML regex as the reliable fallback. """ resp = session.get(f"{BASE}/api/membership/lecture/{lecture_id}") resp.raise_for_status() data = resp.json() for f in (data.get("files") or {}).values(): if f.get("type") == "video": return f["path"] m = MP4_RE.search(data.get("html") or "") return m.group(0) if m else None def slugify(text: str) -> str: import re text = re.sub(r"[^\w\s-]", "", text.strip(), flags=re.UNICODE) return re.sub(r"[\s_-]+", "_", text)[:100] or "untitled" def download_video(session: requests.Session, url: str, dest: Path) -> bool: """Streaming download with resume (Range) and retry.""" tmp = dest.with_suffix(dest.suffix + ".tmp") for attempt in range(MAX_RETRIES): try: headers = {} mode = "wb" existing = tmp.stat().st_size if tmp.exists() else 0 if existing: headers["Range"] = f"bytes={existing}-" mode = "ab" with session.get(url, headers=headers, stream=True, timeout=300) as resp: if resp.status_code not in (200, 206): resp.raise_for_status() with open(tmp, mode) as f: for chunk in resp.iter_content(chunk_size=4 * 1024 * 1024): f.write(chunk) tmp.rename(dest) log.info(f" Downloaded: {dest.name} ({dest.stat().st_size / 1_000_000:.1f} MB)") return True except Exception as e: wait = RETRY_BACKOFF[attempt] if attempt < len(RETRY_BACKOFF) else 30 log.warning(f" Attempt {attempt + 1}/{MAX_RETRIES} failed for {dest.name}: {e}") if attempt < MAX_RETRIES - 1: time.sleep(wait) log.error(f" FAILED after {MAX_RETRIES} attempts: {dest.name}") return False def main(): p = argparse.ArgumentParser(description="Download ATM course lecture videos") p.add_argument("--out", default="videos", help="Output directory") args = p.parse_args() load_dotenv() email = os.getenv("ATM_EMAIL", "") password = os.getenv("ATM_PASSWORD", "") if not email or not password: log.error("Set ATM_EMAIL and ATM_PASSWORD in .env") sys.exit(1) out_dir = Path(args.out) out_dir.mkdir(parents=True, exist_ok=True) manifest_path = out_dir / "manifest.json" manifest = json.loads(manifest_path.read_text()) if manifest_path.exists() else {} session = requests.Session() session.headers.update({"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}) log.info("Logging in...") login(session, email, password) course_id = get_course_id(session) log.info(f"Course id: {course_id}") modules = get_menu(session, course_id) downloaded = skipped = failed = 0 for mod_idx, mod in enumerate(modules, 1): mod_dir = out_dir / f"{mod_idx:02d}_{slugify(mod['name'])}" for lec_idx, lec in enumerate(mod["lectures"], 1): title = lec["name"] stem = f"{lec_idx:02d}_{slugify(title)}" key = str(lec["id"]) video_url = get_lecture_video_url(session, lec["id"]) if not video_url: log.info(f" No video (text/quiz/etc): {mod['name']} / {title}") continue dest = mod_dir / f"{stem}{Path(video_url).suffix}" if dest.exists() and dest.stat().st_size > 1_000_000: log.info(f" Skipping (exists): {dest.name}") skipped += 1 manifest[key] = {"title": title, "path": str(dest), "status": "complete"} continue mod_dir.mkdir(parents=True, exist_ok=True) ok = download_video(session, video_url, dest) manifest[key] = {"title": title, "path": str(dest), "status": "complete" if ok else "failed"} manifest_path.write_text(json.dumps(manifest, indent=2, ensure_ascii=False)) downloaded += 1 if ok else 0 failed += 0 if ok else 1 log.info("=" * 60) log.info(f"Downloaded {downloaded}, skipped {skipped}, failed {failed}.") if failed: sys.exit(1) if __name__ == "__main__": main()