Pipeline audio-only (fără video pe disc) prin ffmpeg direct din URL CloudFront, transcriere cu faster-whisper (CPU), sumarizări structurate cu diagrame SVG la scară reală (STYLE.md documentează toate cerințele de format).
163 lines
5.7 KiB
Python
163 lines
5.7 KiB
Python
"""
|
|
Download all lecture videos from the ATM trading course (systeme.io).
|
|
|
|
Login is a JSON POST (session cookie auth), course structure comes from
|
|
systeme.io's own JSON API (no HTML scraping needed), and each lecture's
|
|
video is an unsigned CloudFront MP4 URL -> plain streaming HTTP download
|
|
with resume support (Range header) since files can exceed 1GB.
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import logging
|
|
import os
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
import requests
|
|
from dotenv import load_dotenv
|
|
|
|
BASE = "https://bogdan-jinga.systeme.io"
|
|
COURSE_SLUG = "atm"
|
|
MAX_RETRIES = 3
|
|
RETRY_BACKOFF = [5, 15, 30]
|
|
|
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
|
|
log = logging.getLogger(__name__)
|
|
|
|
|
|
def login(session: requests.Session, email: str, password: str) -> None:
|
|
resp = session.post(f"{BASE}/api/security/login", json={"email": email, "password": password})
|
|
resp.raise_for_status()
|
|
|
|
|
|
def get_course_id(session: requests.Session) -> int:
|
|
resp = session.get(f"{BASE}/api/membership/course/{COURSE_SLUG}")
|
|
resp.raise_for_status()
|
|
return resp.json()["id"]
|
|
|
|
|
|
def get_menu(session: requests.Session, course_id: int) -> list[dict]:
|
|
resp = session.get(f"{BASE}/api/membership/course/{course_id}/menu")
|
|
resp.raise_for_status()
|
|
return resp.json()
|
|
|
|
|
|
MP4_RE = __import__("re").compile(r'https://[^"\\]+\.mp4')
|
|
|
|
|
|
def get_lecture_video_url(session: requests.Session, lecture_id: int) -> str | None:
|
|
"""
|
|
The lecture API sometimes returns a "files" dict with a CloudFront path,
|
|
sometimes inlines the same URL in the raw lecture HTML instead -> check
|
|
both, HTML regex as the reliable fallback.
|
|
"""
|
|
resp = session.get(f"{BASE}/api/membership/lecture/{lecture_id}")
|
|
resp.raise_for_status()
|
|
data = resp.json()
|
|
for f in (data.get("files") or {}).values():
|
|
if f.get("type") == "video":
|
|
return f["path"]
|
|
m = MP4_RE.search(data.get("html") or "")
|
|
return m.group(0) if m else None
|
|
|
|
|
|
def slugify(text: str) -> str:
|
|
import re
|
|
text = re.sub(r"[^\w\s-]", "", text.strip(), flags=re.UNICODE)
|
|
return re.sub(r"[\s_-]+", "_", text)[:100] or "untitled"
|
|
|
|
|
|
def download_video(session: requests.Session, url: str, dest: Path) -> bool:
|
|
"""Streaming download with resume (Range) and retry."""
|
|
tmp = dest.with_suffix(dest.suffix + ".tmp")
|
|
for attempt in range(MAX_RETRIES):
|
|
try:
|
|
headers = {}
|
|
mode = "wb"
|
|
existing = tmp.stat().st_size if tmp.exists() else 0
|
|
if existing:
|
|
headers["Range"] = f"bytes={existing}-"
|
|
mode = "ab"
|
|
with session.get(url, headers=headers, stream=True, timeout=300) as resp:
|
|
if resp.status_code not in (200, 206):
|
|
resp.raise_for_status()
|
|
with open(tmp, mode) as f:
|
|
for chunk in resp.iter_content(chunk_size=4 * 1024 * 1024):
|
|
f.write(chunk)
|
|
tmp.rename(dest)
|
|
log.info(f" Downloaded: {dest.name} ({dest.stat().st_size / 1_000_000:.1f} MB)")
|
|
return True
|
|
except Exception as e:
|
|
wait = RETRY_BACKOFF[attempt] if attempt < len(RETRY_BACKOFF) else 30
|
|
log.warning(f" Attempt {attempt + 1}/{MAX_RETRIES} failed for {dest.name}: {e}")
|
|
if attempt < MAX_RETRIES - 1:
|
|
time.sleep(wait)
|
|
log.error(f" FAILED after {MAX_RETRIES} attempts: {dest.name}")
|
|
return False
|
|
|
|
|
|
def main():
|
|
p = argparse.ArgumentParser(description="Download ATM course lecture videos")
|
|
p.add_argument("--out", default="videos", help="Output directory")
|
|
args = p.parse_args()
|
|
|
|
load_dotenv()
|
|
email = os.getenv("ATM_EMAIL", "")
|
|
password = os.getenv("ATM_PASSWORD", "")
|
|
if not email or not password:
|
|
log.error("Set ATM_EMAIL and ATM_PASSWORD in .env")
|
|
sys.exit(1)
|
|
|
|
out_dir = Path(args.out)
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
manifest_path = out_dir / "manifest.json"
|
|
manifest = json.loads(manifest_path.read_text()) if manifest_path.exists() else {}
|
|
|
|
session = requests.Session()
|
|
session.headers.update({"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"})
|
|
|
|
log.info("Logging in...")
|
|
login(session, email, password)
|
|
course_id = get_course_id(session)
|
|
log.info(f"Course id: {course_id}")
|
|
|
|
modules = get_menu(session, course_id)
|
|
downloaded = skipped = failed = 0
|
|
|
|
for mod_idx, mod in enumerate(modules, 1):
|
|
mod_dir = out_dir / f"{mod_idx:02d}_{slugify(mod['name'])}"
|
|
for lec_idx, lec in enumerate(mod["lectures"], 1):
|
|
title = lec["name"]
|
|
stem = f"{lec_idx:02d}_{slugify(title)}"
|
|
key = str(lec["id"])
|
|
|
|
video_url = get_lecture_video_url(session, lec["id"])
|
|
if not video_url:
|
|
log.info(f" No video (text/quiz/etc): {mod['name']} / {title}")
|
|
continue
|
|
|
|
dest = mod_dir / f"{stem}{Path(video_url).suffix}"
|
|
if dest.exists() and dest.stat().st_size > 1_000_000:
|
|
log.info(f" Skipping (exists): {dest.name}")
|
|
skipped += 1
|
|
manifest[key] = {"title": title, "path": str(dest), "status": "complete"}
|
|
continue
|
|
|
|
mod_dir.mkdir(parents=True, exist_ok=True)
|
|
ok = download_video(session, video_url, dest)
|
|
manifest[key] = {"title": title, "path": str(dest), "status": "complete" if ok else "failed"}
|
|
manifest_path.write_text(json.dumps(manifest, indent=2, ensure_ascii=False))
|
|
downloaded += 1 if ok else 0
|
|
failed += 0 if ok else 1
|
|
|
|
log.info("=" * 60)
|
|
log.info(f"Downloaded {downloaded}, skipped {skipped}, failed {failed}.")
|
|
if failed:
|
|
sys.exit(1)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|