Files
yt-channel-scraper/src/yt_scraper/extract.py
T
urieljareth b3b27ce883 feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo
Extraccion autenticada:
- import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20
- extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla
  con sesion logueada (members-only); regex y opener cacheados
- js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts
- rutas de Brave multiplataforma (Windows/macOS/Linux)

Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL,
memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics

Rendimiento:
- entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota)
- _rank_unranked con guarda (antes full-scan en cada arranque/import)
- upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL)
- thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool)
- reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md),
  healthz expone reconcile_done
- Store.transaction(): escrituras por video agrupadas (~6 commits -> 3)

Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render,
order_pending->store, keep_ref->config, seconds_to_ts solo en segments);
re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done);
fuera wrappers muertos de segments.py
2026-09-10 00:32:19 -06:00

464 lines
19 KiB
Python

from __future__ import annotations
import json
import logging
import os
import re
import urllib.request
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Mapping
import yt_dlp
from ._yt_http import yt_get
from .parse import Segment, parse_auto_dump
from .ratelimit import GLOBAL_PACER, ydl_throttle_opts
log = logging.getLogger(__name__)
@dataclass
class SubtitlePick:
url: str
ext: str
lang: str
source: str # "manual" | "auto"
@dataclass
class VideoData:
info: dict[str, Any]
segments: list[Segment]
subtitle: SubtitlePick | None
has_chapters: bool
# Why `segments` came back empty. "No transcript" has several very
# different causes — the video genuinely has no captions, the language
# policy rejected the tracks that do exist, or the download was throttled —
# and collapsing them into one terminal status hides recoverable failures.
skip_reason: str | None = None
_LANGUAGE_MODES = ("manual", "auto", "any")
def _sources_for(mode: str, prefer_manual: bool, manual: dict, auto: dict) -> list[tuple[str, dict]]:
"""Return ordered list of (label, tracks_dict) to try for ``mode``.
``any`` defers to the legacy :data:`prefer_manual` global default.
``manual`` / ``auto`` force the track family even if the other has
a higher-priority language elsewhere in the iteration.
"""
if mode == "manual":
return [("manual", manual)]
if mode == "auto":
return [("auto", auto)]
if prefer_manual:
return [("manual", manual), ("auto", auto)]
return [("auto", auto), ("manual", manual)]
def extract_video(
video_url: str,
languages: Mapping[str, str] | list[str],
retries: int = 10,
sleep_subrequests: float = 2.0,
prefer_manual: bool = True,
cookies_file: str | None = None,
cookies_from_browser: str | None = None,
extractor_retries: int = 3,
socket_timeout: float = 30.0,
) -> VideoData:
"""Run yt-dlp on ``video_url`` and pull the preferred subtitle track.
``languages`` may be either a list (legacy, every entry uses
``prefer_manual``) or a mapping ``{lang: mode}`` where ``mode`` is one
of ``"manual"``, ``"auto"`` or ``"any"``. The mapping form is the
preferred interface because it lets you mix per-language policies such
as ``{"en": "manual", "es": "auto", "pt": "any"}``.
"""
languages_dict = _coerce_languages(languages, prefer_manual)
ydl_opts: dict[str, Any] = {
"writesubtitles": True,
"writeautomaticsub": True,
"subtitleslangs": list(languages_dict.keys()),
"skip_download": True,
"quiet": True,
"no_warnings": True,
"retries": retries,
"noprogress": True,
# Never let yt-dlp probe formats: it costs one HTTP request per format
# and we only ever want captions and metadata.
"check_formats": None,
**ydl_throttle_opts(
sleep_subrequests,
extractor_retries=extractor_retries,
socket_timeout=socket_timeout,
),
}
if cookies_file:
ydl_opts["cookiefile"] = cookies_file
if cookies_from_browser:
ydl_opts["cookiesfrombrowser"] = (cookies_from_browser,)
# One video extraction is two requests: the watch page and the InnerTube
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
GLOBAL_PACER.wait(cost=2)
try:
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(video_url, download=False)
except Exception:
# Logged-in sessions on current YouTube increasingly end here
# ("The page needs to be reloaded" / format-availability failures).
# The session itself is usually fine — the watch page still hands
# metadata and caption tracks to a plain cookie'd GET — so try that
# before giving up. Without cookies there is nothing to fall back to.
if cookies_file:
fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual)
if fallback is not None:
return fallback
raise
pick = pick_subtitle(info, languages_dict, prefer_manual)
segments: list[Segment] = []
skip_reason: str | None = None
if pick:
raw, dl_error = _download_subtitle(pick.url)
if raw:
segments = parse_auto_dump(raw)
if not segments:
log.warning("Could not parse subtitle for %s (format=%s)", video_url, pick.ext)
skip_reason = f"subtitle downloaded but parsed empty (lang={pick.lang}, format={pick.ext})"
else:
skip_reason = (
f"subtitle track found (lang={pick.lang}, {pick.source}) but the download failed "
f"— usually throttling; retry later [{dl_error}]"
)
else:
skip_reason = describe_missing_subtitle(info, languages_dict)
has_chapters = bool(info.get("chapters"))
return VideoData(
info=info, segments=segments, subtitle=pick,
has_chapters=has_chapters, skip_reason=skip_reason,
)
_WATCH_PAGE_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36"
)
# Precompiled: the fallback can fire on every video of a throttled batch, and
# re-compiling per call showed up under those runs.
_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;")
# One opener per (cookie file, mtime): the jar parse is per-call work that is
# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under
# the same path still gets a fresh jar; only the newest entry is kept.
_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {}
def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0):
path = str(Path(cookies_file).resolve())
try:
mtime = os.path.getmtime(path)
except OSError:
mtime = -1.0
key = (path, mtime)
opener = _OPENER_CACHE.get(key)
if opener is None:
jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file)
jar.load(ignore_discard=True, ignore_expires=True)
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
opener.addheaders = [
("User-Agent", _WATCH_PAGE_UA),
("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"),
]
_OPENER_CACHE.clear()
_OPENER_CACHE[key] = opener
return opener.open(url, timeout=timeout)
def extract_via_watch_page(
video_url: str,
cookies_file: str,
languages: Mapping[str, str],
prefer_manual: bool = True,
) -> VideoData | None:
"""Session-cookie fallback: scrape the watch page directly.
yt-dlp's InnerTube clients reject logged-in sessions that lack a PO
token (playability "The page needs to be reloaded") or return no
formats/captions, which kills cookie-authenticated videos — members
being the case this exists for. The plain watch page served to the
logged-in browser still carries `ytInitialPlayerResponse` with
metadata and caption tracks, so GET it with the vault cookie and
reuse the normal subtitle picker. Returns None when the page holds
no caption tracks at all, so callers keep their own error semantics.
"""
GLOBAL_PACER.wait()
html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace")
m = _INITIAL_PLAYER_RE.search(html)
if not m:
log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url)
return None
try:
pr = json.loads(m.group(1))
except json.JSONDecodeError:
log.warning("watch-page fallback: unparseable player response for %s", video_url)
return None
status = (pr.get("playabilityStatus") or {}).get("status")
if status != "OK":
reason = (pr.get("playabilityStatus") or {}).get("reason") or status
raise RuntimeError(f"watch-page fallback: video not playable ({reason})")
details = pr.get("videoDetails") or {}
micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {}
tracks = (
(pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {}
).get("captionTracks") or []
if not tracks:
return None
# Reuse pick_subtitle by shaping the tracks as an info dict.
info: dict[str, Any] = {
"title": details.get("title"),
"channel": details.get("author"),
"duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None,
"view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None,
"description": details.get("shortDescription") or "",
"tags": details.get("keywords") or [],
"thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"),
"upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None,
# The watch page carries no availability signal; leaving it unset
# keeps the (more informed) discovery value in the store.
"availability": None,
"subtitles": {},
"automatic_captions": {},
}
for t in tracks:
base = t.get("baseUrl") or ""
if not base:
continue
entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}]
if t.get("kind") == "asr":
info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry)
else:
info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry)
pick = pick_subtitle(info, languages, prefer_manual)
segments: list[Segment] = []
skip_reason: str | None = None
if pick:
try:
GLOBAL_PACER.wait()
raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace")
segments = parse_auto_dump(raw)
if not segments:
skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)"
except Exception as exc: # pylint: disable=broad-except
skip_reason = f"caption download failed via watch-page fallback: {exc}"
else:
skip_reason = describe_missing_subtitle(info, languages)
return VideoData(
info=info, segments=segments, subtitle=pick,
has_chapters=bool(info.get("chapters")), skip_reason=skip_reason,
)
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
A channel that only publishes auto-generated captions scanned under a
manual-only policy yields nothing — which is a config problem, not a
property of the video, and the message has to say so.
"""
manual = {k: v for k, v in (info.get("subtitles") or {}).items() if v}
auto = {k: v for k, v in (info.get("automatic_captions") or {}).items() if v}
if not manual and not auto:
return "no caption tracks published for this video"
wanted = ", ".join(f"{lang}={mode}" for lang, mode in languages.items()) or "(none configured)"
modes = {str(m).lower() for m in languages.values()}
parts = [f"no track matched the language policy ({wanted})"]
parts.append(f"available: {len(manual)} manual, {len(auto)} auto")
if auto and not manual and modes == {"manual"}:
parts.append(
"this video has ONLY auto-generated captions — set the language mode "
"to 'any' or 'auto' to use them"
)
return "; ".join(parts)
def _coerce_languages(languages: Mapping[str, str] | list[str] | None,
prefer_manual: bool) -> dict[str, str]:
"""Normalise legacy list / new dict / None into ``{lang: mode}``."""
default = "manual" if prefer_manual else "auto"
if languages is None:
return {}
if isinstance(languages, Mapping):
out: dict[str, str] = {}
for lang, mode in languages.items():
m = str(mode).lower().strip()
if m not in _LANGUAGE_MODES:
m = "any"
out[str(lang)] = m
return out
if isinstance(languages, (list, tuple)):
return {str(l): default for l in languages}
return {}
def pick_subtitle(info: dict[str, Any],
languages: Mapping[str, str],
prefer_manual: bool = True) -> SubtitlePick | None:
"""Pick the best subtitle track for ``info`` honouring per-language mode.
See :func:`extract_video` for the ``languages`` schema. ``prefer_manual``
is only consulted for entries whose mode is ``"any"``.
"""
manual = info.get("subtitles") or {}
auto = info.get("automatic_captions") or {}
if isinstance(languages, (list, tuple)):
# legacy path: convert on the fly
default = "manual" if prefer_manual else "auto"
languages = {l: default for l in languages}
# Config order is a preference between languages we can read, not an
# instruction to accept a machine translation when the real transcript is
# sitting right there. An English channel scanned under {es, es-419, en}
# was yielding Spanish auto-translations of English speech.
#
# Two passes rather than a reorder. The reorder alone needed to know the
# spoken language, and when neither an `-orig` key nor `info["language"]`
# was present it silently fell back to config order and reintroduced the
# bug. Rejecting translations outright in the first pass needs no such
# knowledge: whatever language it lands on, it is the one actually spoken.
ordered = list(languages.items())
spoken = original_language(info)
if spoken and any(_normalize_lang(l) == spoken for l, _ in ordered):
ordered.sort(key=lambda kv: _normalize_lang(kv[0]) != spoken)
for allow_translations in (False, True):
for lang, mode in ordered:
if mode not in _LANGUAGE_MODES:
mode = "any"
sources = _sources_for(mode, prefer_manual, manual, auto)
normalized = _normalize_lang(lang)
for source_label, tracks in sources:
for track_lang, formats in _ordered_tracks(tracks, normalized):
if not allow_translations and _is_translation(track_lang, formats):
continue
pick = _pick_best_format(formats)
if pick:
return SubtitlePick(
url=pick["url"],
ext=pick["ext"],
lang=track_lang,
source=source_label,
)
return None
def _pick_best_format(formats: list[dict[str, Any]]) -> dict[str, Any] | None:
priority = ["json3", "srv1", "srv3", "vtt", "ttml"]
for ext in priority:
for fmt in formats:
if fmt.get("ext") == ext and fmt.get("url"):
return fmt
for fmt in formats:
if fmt.get("url"):
return fmt
return None
def _normalize_lang(code: str) -> str:
base = code.replace("_", "-").split("-")[0].lower()
return base
def _is_original_track(code: str) -> bool:
"""True for YouTube's original-ASR track, which it suffixes with ``-orig``.
YouTube publishes the speech-recognised track as ``<lang>-orig`` and then a
long tail of machine translations keyed by bare language code — including a
translation *into the video's own language*. So on a Spanish video both
``es-orig`` and ``es`` exist, and only the first is the real transcript.
"""
return code.replace("_", "-").lower().endswith("-orig")
def original_language(info: dict[str, Any]) -> str | None:
"""The language actually spoken in the video, normalised, or None.
Prefers the ``-orig`` track that YouTube itself publishes over ``info`` keys,
because the ``-orig`` suffix is direct evidence from the caption list while
``language`` is metadata that YouTube localises along with the title.
"""
for code in (info.get("automatic_captions") or {}):
if _is_original_track(code):
return _normalize_lang(code)
for key in ("language", "original_language"):
val = info.get(key)
if isinstance(val, str) and val:
return _normalize_lang(val)
return None
def _is_translation(code: str, formats: list[dict[str, Any]] | None) -> bool:
"""True when this track is YouTube machine-translating some other track.
The caption URL says so outright: yt-dlp builds a translated track by
appending ``tlang=`` to the base track's URL and omits it when the target
equals the source language. That is direct evidence, unlike the ``-orig``
naming convention, and it is what lets us reject a translation even for a
video whose spoken language we could not otherwise determine.
"""
if _is_original_track(code):
return False
for fmt in formats or []:
url = fmt.get("url") or ""
if "tlang=" in url:
return True
return False
def _ordered_tracks(tracks: dict, normalized: str) -> list[tuple[str, Any]]:
"""Tracks matching `normalized`, original-ASR first.
Without this the picker took whichever key yt-dlp happened to list first.
That silently returned the right thing on Spanish channels (``es-orig``
sorts before ``es``) and the wrong thing everywhere else.
"""
matches = [(c, f) for c, f in tracks.items() if _normalize_lang(c) == normalized and f]
matches.sort(key=lambda kv: not _is_original_track(kv[0]))
return matches
def _download_subtitle(url: str, *, timeout: float = 15.0) -> tuple[str | None, str | None]:
"""Fetch a caption track. Returns (text, error_description).
The error text is returned rather than only logged because a 429 here is
how YouTube throttling most often shows up on this path, and the circuit
breaker upstream can only see it if it survives into the stored reason.
"""
try:
GLOBAL_PACER.wait()
resp = yt_get(url, timeout=timeout)
resp.raise_for_status()
return resp.text, None
except Exception as exc: # pylint: disable=broad-except
log.error("Failed to download subtitle from %s: %s", url, exc)
# Lead with a normalised "HTTP Error <status>" token. Downstream, the
# throttle detector has to recognise a 429 here, and depending on the
# prose is fragile: a 429 served without a reason phrase (routine over
# HTTP/2) says nothing about "too many requests".
status = getattr(getattr(exc, "response", None), "status_code", None)
prefix = f"HTTP Error {status}: " if status else ""
return None, f"{prefix}{type(exc).__name__}: {exc}"