from __future__ import annotations import json import logging import os import re import urllib.request from dataclasses import dataclass from pathlib import Path from typing import Any, Mapping import yt_dlp from ._yt_http import yt_get from .parse import Segment, parse_auto_dump from .ratelimit import GLOBAL_PACER, ydl_throttle_opts log = logging.getLogger(__name__) @dataclass class SubtitlePick: url: str ext: str lang: str source: str # "manual" | "auto" @dataclass class VideoData: info: dict[str, Any] segments: list[Segment] subtitle: SubtitlePick | None has_chapters: bool # Why `segments` came back empty. "No transcript" has several very # different causes — the video genuinely has no captions, the language # policy rejected the tracks that do exist, or the download was throttled — # and collapsing them into one terminal status hides recoverable failures. skip_reason: str | None = None _LANGUAGE_MODES = ("manual", "auto", "any") def _sources_for(mode: str, prefer_manual: bool, manual: dict, auto: dict) -> list[tuple[str, dict]]: """Return ordered list of (label, tracks_dict) to try for ``mode``. ``any`` defers to the legacy :data:`prefer_manual` global default. ``manual`` / ``auto`` force the track family even if the other has a higher-priority language elsewhere in the iteration. """ if mode == "manual": return [("manual", manual)] if mode == "auto": return [("auto", auto)] if prefer_manual: return [("manual", manual), ("auto", auto)] return [("auto", auto), ("manual", manual)] def extract_video( video_url: str, languages: Mapping[str, str] | list[str], retries: int = 10, sleep_subrequests: float = 2.0, prefer_manual: bool = True, cookies_file: str | None = None, cookies_from_browser: str | None = None, extractor_retries: int = 3, socket_timeout: float = 30.0, ) -> VideoData: """Run yt-dlp on ``video_url`` and pull the preferred subtitle track. ``languages`` may be either a list (legacy, every entry uses ``prefer_manual``) or a mapping ``{lang: mode}`` where ``mode`` is one of ``"manual"``, ``"auto"`` or ``"any"``. The mapping form is the preferred interface because it lets you mix per-language policies such as ``{"en": "manual", "es": "auto", "pt": "any"}``. """ languages_dict = _coerce_languages(languages, prefer_manual) ydl_opts: dict[str, Any] = { "writesubtitles": True, "writeautomaticsub": True, "subtitleslangs": list(languages_dict.keys()), "skip_download": True, "quiet": True, "no_warnings": True, "retries": retries, "noprogress": True, # Never let yt-dlp probe formats: it costs one HTTP request per format # and we only ever want captions and metadata. "check_formats": None, **ydl_throttle_opts( sleep_subrequests, extractor_retries=extractor_retries, socket_timeout=socket_timeout, ), } if cookies_file: ydl_opts["cookiefile"] = cookies_file if cookies_from_browser: ydl_opts["cookiesfrombrowser"] = (cookies_from_browser,) # One video extraction is two requests: the watch page and the InnerTube # player call. yt-dlp spaces them itself; the pacer needs to know they exist. GLOBAL_PACER.wait(cost=2) try: with yt_dlp.YoutubeDL(ydl_opts) as ydl: info = ydl.extract_info(video_url, download=False) except Exception: # Logged-in sessions on current YouTube increasingly end here # ("The page needs to be reloaded" / format-availability failures). # The session itself is usually fine — the watch page still hands # metadata and caption tracks to a plain cookie'd GET — so try that # before giving up. Without cookies there is nothing to fall back to. if cookies_file: fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual) if fallback is not None: return fallback raise pick = pick_subtitle(info, languages_dict, prefer_manual) segments: list[Segment] = [] skip_reason: str | None = None if pick: raw, dl_error = _download_subtitle(pick.url) if raw: segments = parse_auto_dump(raw) if not segments: log.warning("Could not parse subtitle for %s (format=%s)", video_url, pick.ext) skip_reason = f"subtitle downloaded but parsed empty (lang={pick.lang}, format={pick.ext})" else: skip_reason = ( f"subtitle track found (lang={pick.lang}, {pick.source}) but the download failed " f"— usually throttling; retry later [{dl_error}]" ) else: skip_reason = describe_missing_subtitle(info, languages_dict) has_chapters = bool(info.get("chapters")) return VideoData( info=info, segments=segments, subtitle=pick, has_chapters=has_chapters, skip_reason=skip_reason, ) _WATCH_PAGE_UA = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36" ) # Precompiled: the fallback can fire on every video of a throttled batch, and # re-compiling per call showed up under those runs. _INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;") # One opener per (cookie file, mtime): the jar parse is per-call work that is # pure waste inside a batch. Keyed on mtime so a re-imported cookie file under # the same path still gets a fresh jar; only the newest entry is kept. _OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {} def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0): path = str(Path(cookies_file).resolve()) try: mtime = os.path.getmtime(path) except OSError: mtime = -1.0 key = (path, mtime) opener = _OPENER_CACHE.get(key) if opener is None: jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file) jar.load(ignore_discard=True, ignore_expires=True) opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar)) opener.addheaders = [ ("User-Agent", _WATCH_PAGE_UA), ("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"), ] _OPENER_CACHE.clear() _OPENER_CACHE[key] = opener return opener.open(url, timeout=timeout) def extract_via_watch_page( video_url: str, cookies_file: str, languages: Mapping[str, str], prefer_manual: bool = True, ) -> VideoData | None: """Session-cookie fallback: scrape the watch page directly. yt-dlp's InnerTube clients reject logged-in sessions that lack a PO token (playability "The page needs to be reloaded") or return no formats/captions, which kills cookie-authenticated videos — members being the case this exists for. The plain watch page served to the logged-in browser still carries `ytInitialPlayerResponse` with metadata and caption tracks, so GET it with the vault cookie and reuse the normal subtitle picker. Returns None when the page holds no caption tracks at all, so callers keep their own error semantics. """ GLOBAL_PACER.wait() html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace") m = _INITIAL_PLAYER_RE.search(html) if not m: log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url) return None try: pr = json.loads(m.group(1)) except json.JSONDecodeError: log.warning("watch-page fallback: unparseable player response for %s", video_url) return None status = (pr.get("playabilityStatus") or {}).get("status") if status != "OK": reason = (pr.get("playabilityStatus") or {}).get("reason") or status raise RuntimeError(f"watch-page fallback: video not playable ({reason})") details = pr.get("videoDetails") or {} micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {} tracks = ( (pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {} ).get("captionTracks") or [] if not tracks: return None # Reuse pick_subtitle by shaping the tracks as an info dict. info: dict[str, Any] = { "title": details.get("title"), "channel": details.get("author"), "duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None, "view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None, "description": details.get("shortDescription") or "", "tags": details.get("keywords") or [], "thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"), "upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None, # The watch page carries no availability signal; leaving it unset # keeps the (more informed) discovery value in the store. "availability": None, "subtitles": {}, "automatic_captions": {}, } for t in tracks: base = t.get("baseUrl") or "" if not base: continue entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}] if t.get("kind") == "asr": info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry) else: info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry) pick = pick_subtitle(info, languages, prefer_manual) segments: list[Segment] = [] skip_reason: str | None = None if pick: try: GLOBAL_PACER.wait() raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace") segments = parse_auto_dump(raw) if not segments: skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)" except Exception as exc: # pylint: disable=broad-except skip_reason = f"caption download failed via watch-page fallback: {exc}" else: skip_reason = describe_missing_subtitle(info, languages) return VideoData( info=info, segments=segments, subtitle=pick, has_chapters=bool(info.get("chapters")), skip_reason=skip_reason, ) def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str: """Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'. A channel that only publishes auto-generated captions scanned under a manual-only policy yields nothing — which is a config problem, not a property of the video, and the message has to say so. """ manual = {k: v for k, v in (info.get("subtitles") or {}).items() if v} auto = {k: v for k, v in (info.get("automatic_captions") or {}).items() if v} if not manual and not auto: return "no caption tracks published for this video" wanted = ", ".join(f"{lang}={mode}" for lang, mode in languages.items()) or "(none configured)" modes = {str(m).lower() for m in languages.values()} parts = [f"no track matched the language policy ({wanted})"] parts.append(f"available: {len(manual)} manual, {len(auto)} auto") if auto and not manual and modes == {"manual"}: parts.append( "this video has ONLY auto-generated captions — set the language mode " "to 'any' or 'auto' to use them" ) return "; ".join(parts) def _coerce_languages(languages: Mapping[str, str] | list[str] | None, prefer_manual: bool) -> dict[str, str]: """Normalise legacy list / new dict / None into ``{lang: mode}``.""" default = "manual" if prefer_manual else "auto" if languages is None: return {} if isinstance(languages, Mapping): out: dict[str, str] = {} for lang, mode in languages.items(): m = str(mode).lower().strip() if m not in _LANGUAGE_MODES: m = "any" out[str(lang)] = m return out if isinstance(languages, (list, tuple)): return {str(l): default for l in languages} return {} def pick_subtitle(info: dict[str, Any], languages: Mapping[str, str], prefer_manual: bool = True) -> SubtitlePick | None: """Pick the best subtitle track for ``info`` honouring per-language mode. See :func:`extract_video` for the ``languages`` schema. ``prefer_manual`` is only consulted for entries whose mode is ``"any"``. """ manual = info.get("subtitles") or {} auto = info.get("automatic_captions") or {} if isinstance(languages, (list, tuple)): # legacy path: convert on the fly default = "manual" if prefer_manual else "auto" languages = {l: default for l in languages} # Config order is a preference between languages we can read, not an # instruction to accept a machine translation when the real transcript is # sitting right there. An English channel scanned under {es, es-419, en} # was yielding Spanish auto-translations of English speech. # # Two passes rather than a reorder. The reorder alone needed to know the # spoken language, and when neither an `-orig` key nor `info["language"]` # was present it silently fell back to config order and reintroduced the # bug. Rejecting translations outright in the first pass needs no such # knowledge: whatever language it lands on, it is the one actually spoken. ordered = list(languages.items()) spoken = original_language(info) if spoken and any(_normalize_lang(l) == spoken for l, _ in ordered): ordered.sort(key=lambda kv: _normalize_lang(kv[0]) != spoken) for allow_translations in (False, True): for lang, mode in ordered: if mode not in _LANGUAGE_MODES: mode = "any" sources = _sources_for(mode, prefer_manual, manual, auto) normalized = _normalize_lang(lang) for source_label, tracks in sources: for track_lang, formats in _ordered_tracks(tracks, normalized): if not allow_translations and _is_translation(track_lang, formats): continue pick = _pick_best_format(formats) if pick: return SubtitlePick( url=pick["url"], ext=pick["ext"], lang=track_lang, source=source_label, ) return None def _pick_best_format(formats: list[dict[str, Any]]) -> dict[str, Any] | None: priority = ["json3", "srv1", "srv3", "vtt", "ttml"] for ext in priority: for fmt in formats: if fmt.get("ext") == ext and fmt.get("url"): return fmt for fmt in formats: if fmt.get("url"): return fmt return None def _normalize_lang(code: str) -> str: base = code.replace("_", "-").split("-")[0].lower() return base def _is_original_track(code: str) -> bool: """True for YouTube's original-ASR track, which it suffixes with ``-orig``. YouTube publishes the speech-recognised track as ``-orig`` and then a long tail of machine translations keyed by bare language code — including a translation *into the video's own language*. So on a Spanish video both ``es-orig`` and ``es`` exist, and only the first is the real transcript. """ return code.replace("_", "-").lower().endswith("-orig") def original_language(info: dict[str, Any]) -> str | None: """The language actually spoken in the video, normalised, or None. Prefers the ``-orig`` track that YouTube itself publishes over ``info`` keys, because the ``-orig`` suffix is direct evidence from the caption list while ``language`` is metadata that YouTube localises along with the title. """ for code in (info.get("automatic_captions") or {}): if _is_original_track(code): return _normalize_lang(code) for key in ("language", "original_language"): val = info.get(key) if isinstance(val, str) and val: return _normalize_lang(val) return None def _is_translation(code: str, formats: list[dict[str, Any]] | None) -> bool: """True when this track is YouTube machine-translating some other track. The caption URL says so outright: yt-dlp builds a translated track by appending ``tlang=`` to the base track's URL and omits it when the target equals the source language. That is direct evidence, unlike the ``-orig`` naming convention, and it is what lets us reject a translation even for a video whose spoken language we could not otherwise determine. """ if _is_original_track(code): return False for fmt in formats or []: url = fmt.get("url") or "" if "tlang=" in url: return True return False def _ordered_tracks(tracks: dict, normalized: str) -> list[tuple[str, Any]]: """Tracks matching `normalized`, original-ASR first. Without this the picker took whichever key yt-dlp happened to list first. That silently returned the right thing on Spanish channels (``es-orig`` sorts before ``es``) and the wrong thing everywhere else. """ matches = [(c, f) for c, f in tracks.items() if _normalize_lang(c) == normalized and f] matches.sort(key=lambda kv: not _is_original_track(kv[0])) return matches def _download_subtitle(url: str, *, timeout: float = 15.0) -> tuple[str | None, str | None]: """Fetch a caption track. Returns (text, error_description). The error text is returned rather than only logged because a 429 here is how YouTube throttling most often shows up on this path, and the circuit breaker upstream can only see it if it survives into the stored reason. """ try: GLOBAL_PACER.wait() resp = yt_get(url, timeout=timeout) resp.raise_for_status() return resp.text, None except Exception as exc: # pylint: disable=broad-except log.error("Failed to download subtitle from %s: %s", url, exc) # Lead with a normalised "HTTP Error " token. Downstream, the # throttle detector has to recognise a 429 here, and depending on the # prose is fragile: a 429 served without a reason phrase (routine over # HTTP/2) says nothing about "too many requests". status = getattr(getattr(exc, "response", None), "status_code", None) prefix = f"HTTP Error {status}: " if status else "" return None, f"{prefix}{type(exc).__name__}: {exc}"