Extraccion autenticada: - import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20 - extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla con sesion logueada (members-only); regex y opener cacheados - js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts - rutas de Brave multiplataforma (Windows/macOS/Linux) Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL, memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics Rendimiento: - entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota) - _rank_unranked con guarda (antes full-scan en cada arranque/import) - upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL) - thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool) - reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md), healthz expone reconcile_done - Store.transaction(): escrituras por video agrupadas (~6 commits -> 3) Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render, order_pending->store, keep_ref->config, seconds_to_ts solo en segments); re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done); fuera wrappers muertos de segments.py
464 lines
19 KiB
Python
464 lines
19 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import urllib.request
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Any, Mapping
|
|
|
|
import yt_dlp
|
|
|
|
from ._yt_http import yt_get
|
|
from .parse import Segment, parse_auto_dump
|
|
from .ratelimit import GLOBAL_PACER, ydl_throttle_opts
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
|
|
@dataclass
|
|
class SubtitlePick:
|
|
url: str
|
|
ext: str
|
|
lang: str
|
|
source: str # "manual" | "auto"
|
|
|
|
|
|
@dataclass
|
|
class VideoData:
|
|
info: dict[str, Any]
|
|
segments: list[Segment]
|
|
subtitle: SubtitlePick | None
|
|
has_chapters: bool
|
|
# Why `segments` came back empty. "No transcript" has several very
|
|
# different causes — the video genuinely has no captions, the language
|
|
# policy rejected the tracks that do exist, or the download was throttled —
|
|
# and collapsing them into one terminal status hides recoverable failures.
|
|
skip_reason: str | None = None
|
|
|
|
|
|
_LANGUAGE_MODES = ("manual", "auto", "any")
|
|
|
|
|
|
def _sources_for(mode: str, prefer_manual: bool, manual: dict, auto: dict) -> list[tuple[str, dict]]:
|
|
"""Return ordered list of (label, tracks_dict) to try for ``mode``.
|
|
|
|
``any`` defers to the legacy :data:`prefer_manual` global default.
|
|
``manual`` / ``auto`` force the track family even if the other has
|
|
a higher-priority language elsewhere in the iteration.
|
|
"""
|
|
if mode == "manual":
|
|
return [("manual", manual)]
|
|
if mode == "auto":
|
|
return [("auto", auto)]
|
|
if prefer_manual:
|
|
return [("manual", manual), ("auto", auto)]
|
|
return [("auto", auto), ("manual", manual)]
|
|
|
|
|
|
def extract_video(
|
|
video_url: str,
|
|
languages: Mapping[str, str] | list[str],
|
|
retries: int = 10,
|
|
sleep_subrequests: float = 2.0,
|
|
prefer_manual: bool = True,
|
|
cookies_file: str | None = None,
|
|
cookies_from_browser: str | None = None,
|
|
extractor_retries: int = 3,
|
|
socket_timeout: float = 30.0,
|
|
) -> VideoData:
|
|
"""Run yt-dlp on ``video_url`` and pull the preferred subtitle track.
|
|
|
|
``languages`` may be either a list (legacy, every entry uses
|
|
``prefer_manual``) or a mapping ``{lang: mode}`` where ``mode`` is one
|
|
of ``"manual"``, ``"auto"`` or ``"any"``. The mapping form is the
|
|
preferred interface because it lets you mix per-language policies such
|
|
as ``{"en": "manual", "es": "auto", "pt": "any"}``.
|
|
"""
|
|
languages_dict = _coerce_languages(languages, prefer_manual)
|
|
|
|
ydl_opts: dict[str, Any] = {
|
|
"writesubtitles": True,
|
|
"writeautomaticsub": True,
|
|
"subtitleslangs": list(languages_dict.keys()),
|
|
"skip_download": True,
|
|
"quiet": True,
|
|
"no_warnings": True,
|
|
"retries": retries,
|
|
"noprogress": True,
|
|
# Never let yt-dlp probe formats: it costs one HTTP request per format
|
|
# and we only ever want captions and metadata.
|
|
"check_formats": None,
|
|
**ydl_throttle_opts(
|
|
sleep_subrequests,
|
|
extractor_retries=extractor_retries,
|
|
socket_timeout=socket_timeout,
|
|
),
|
|
}
|
|
if cookies_file:
|
|
ydl_opts["cookiefile"] = cookies_file
|
|
if cookies_from_browser:
|
|
ydl_opts["cookiesfrombrowser"] = (cookies_from_browser,)
|
|
|
|
# One video extraction is two requests: the watch page and the InnerTube
|
|
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
|
|
GLOBAL_PACER.wait(cost=2)
|
|
try:
|
|
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|
info = ydl.extract_info(video_url, download=False)
|
|
except Exception:
|
|
# Logged-in sessions on current YouTube increasingly end here
|
|
# ("The page needs to be reloaded" / format-availability failures).
|
|
# The session itself is usually fine — the watch page still hands
|
|
# metadata and caption tracks to a plain cookie'd GET — so try that
|
|
# before giving up. Without cookies there is nothing to fall back to.
|
|
if cookies_file:
|
|
fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual)
|
|
if fallback is not None:
|
|
return fallback
|
|
raise
|
|
|
|
pick = pick_subtitle(info, languages_dict, prefer_manual)
|
|
segments: list[Segment] = []
|
|
skip_reason: str | None = None
|
|
if pick:
|
|
raw, dl_error = _download_subtitle(pick.url)
|
|
if raw:
|
|
segments = parse_auto_dump(raw)
|
|
if not segments:
|
|
log.warning("Could not parse subtitle for %s (format=%s)", video_url, pick.ext)
|
|
skip_reason = f"subtitle downloaded but parsed empty (lang={pick.lang}, format={pick.ext})"
|
|
else:
|
|
skip_reason = (
|
|
f"subtitle track found (lang={pick.lang}, {pick.source}) but the download failed "
|
|
f"— usually throttling; retry later [{dl_error}]"
|
|
)
|
|
else:
|
|
skip_reason = describe_missing_subtitle(info, languages_dict)
|
|
|
|
has_chapters = bool(info.get("chapters"))
|
|
return VideoData(
|
|
info=info, segments=segments, subtitle=pick,
|
|
has_chapters=has_chapters, skip_reason=skip_reason,
|
|
)
|
|
|
|
|
|
_WATCH_PAGE_UA = (
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|
"(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36"
|
|
)
|
|
|
|
# Precompiled: the fallback can fire on every video of a throttled batch, and
|
|
# re-compiling per call showed up under those runs.
|
|
_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;")
|
|
|
|
# One opener per (cookie file, mtime): the jar parse is per-call work that is
|
|
# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under
|
|
# the same path still gets a fresh jar; only the newest entry is kept.
|
|
_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {}
|
|
|
|
|
|
def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0):
|
|
path = str(Path(cookies_file).resolve())
|
|
try:
|
|
mtime = os.path.getmtime(path)
|
|
except OSError:
|
|
mtime = -1.0
|
|
key = (path, mtime)
|
|
opener = _OPENER_CACHE.get(key)
|
|
if opener is None:
|
|
jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file)
|
|
jar.load(ignore_discard=True, ignore_expires=True)
|
|
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
|
|
opener.addheaders = [
|
|
("User-Agent", _WATCH_PAGE_UA),
|
|
("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"),
|
|
]
|
|
_OPENER_CACHE.clear()
|
|
_OPENER_CACHE[key] = opener
|
|
return opener.open(url, timeout=timeout)
|
|
|
|
|
|
def extract_via_watch_page(
|
|
video_url: str,
|
|
cookies_file: str,
|
|
languages: Mapping[str, str],
|
|
prefer_manual: bool = True,
|
|
) -> VideoData | None:
|
|
"""Session-cookie fallback: scrape the watch page directly.
|
|
|
|
yt-dlp's InnerTube clients reject logged-in sessions that lack a PO
|
|
token (playability "The page needs to be reloaded") or return no
|
|
formats/captions, which kills cookie-authenticated videos — members
|
|
being the case this exists for. The plain watch page served to the
|
|
logged-in browser still carries `ytInitialPlayerResponse` with
|
|
metadata and caption tracks, so GET it with the vault cookie and
|
|
reuse the normal subtitle picker. Returns None when the page holds
|
|
no caption tracks at all, so callers keep their own error semantics.
|
|
"""
|
|
GLOBAL_PACER.wait()
|
|
html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace")
|
|
m = _INITIAL_PLAYER_RE.search(html)
|
|
if not m:
|
|
log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url)
|
|
return None
|
|
try:
|
|
pr = json.loads(m.group(1))
|
|
except json.JSONDecodeError:
|
|
log.warning("watch-page fallback: unparseable player response for %s", video_url)
|
|
return None
|
|
|
|
status = (pr.get("playabilityStatus") or {}).get("status")
|
|
if status != "OK":
|
|
reason = (pr.get("playabilityStatus") or {}).get("reason") or status
|
|
raise RuntimeError(f"watch-page fallback: video not playable ({reason})")
|
|
|
|
details = pr.get("videoDetails") or {}
|
|
micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {}
|
|
tracks = (
|
|
(pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {}
|
|
).get("captionTracks") or []
|
|
if not tracks:
|
|
return None
|
|
|
|
# Reuse pick_subtitle by shaping the tracks as an info dict.
|
|
info: dict[str, Any] = {
|
|
"title": details.get("title"),
|
|
"channel": details.get("author"),
|
|
"duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None,
|
|
"view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None,
|
|
"description": details.get("shortDescription") or "",
|
|
"tags": details.get("keywords") or [],
|
|
"thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"),
|
|
"upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None,
|
|
# The watch page carries no availability signal; leaving it unset
|
|
# keeps the (more informed) discovery value in the store.
|
|
"availability": None,
|
|
"subtitles": {},
|
|
"automatic_captions": {},
|
|
}
|
|
for t in tracks:
|
|
base = t.get("baseUrl") or ""
|
|
if not base:
|
|
continue
|
|
entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}]
|
|
if t.get("kind") == "asr":
|
|
info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry)
|
|
else:
|
|
info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry)
|
|
|
|
pick = pick_subtitle(info, languages, prefer_manual)
|
|
segments: list[Segment] = []
|
|
skip_reason: str | None = None
|
|
if pick:
|
|
try:
|
|
GLOBAL_PACER.wait()
|
|
raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace")
|
|
segments = parse_auto_dump(raw)
|
|
if not segments:
|
|
skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)"
|
|
except Exception as exc: # pylint: disable=broad-except
|
|
skip_reason = f"caption download failed via watch-page fallback: {exc}"
|
|
else:
|
|
skip_reason = describe_missing_subtitle(info, languages)
|
|
|
|
return VideoData(
|
|
info=info, segments=segments, subtitle=pick,
|
|
has_chapters=bool(info.get("chapters")), skip_reason=skip_reason,
|
|
)
|
|
|
|
|
|
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
|
|
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
|
|
|
|
A channel that only publishes auto-generated captions scanned under a
|
|
manual-only policy yields nothing — which is a config problem, not a
|
|
property of the video, and the message has to say so.
|
|
"""
|
|
manual = {k: v for k, v in (info.get("subtitles") or {}).items() if v}
|
|
auto = {k: v for k, v in (info.get("automatic_captions") or {}).items() if v}
|
|
if not manual and not auto:
|
|
return "no caption tracks published for this video"
|
|
|
|
wanted = ", ".join(f"{lang}={mode}" for lang, mode in languages.items()) or "(none configured)"
|
|
modes = {str(m).lower() for m in languages.values()}
|
|
parts = [f"no track matched the language policy ({wanted})"]
|
|
parts.append(f"available: {len(manual)} manual, {len(auto)} auto")
|
|
if auto and not manual and modes == {"manual"}:
|
|
parts.append(
|
|
"this video has ONLY auto-generated captions — set the language mode "
|
|
"to 'any' or 'auto' to use them"
|
|
)
|
|
return "; ".join(parts)
|
|
|
|
|
|
def _coerce_languages(languages: Mapping[str, str] | list[str] | None,
|
|
prefer_manual: bool) -> dict[str, str]:
|
|
"""Normalise legacy list / new dict / None into ``{lang: mode}``."""
|
|
default = "manual" if prefer_manual else "auto"
|
|
if languages is None:
|
|
return {}
|
|
if isinstance(languages, Mapping):
|
|
out: dict[str, str] = {}
|
|
for lang, mode in languages.items():
|
|
m = str(mode).lower().strip()
|
|
if m not in _LANGUAGE_MODES:
|
|
m = "any"
|
|
out[str(lang)] = m
|
|
return out
|
|
if isinstance(languages, (list, tuple)):
|
|
return {str(l): default for l in languages}
|
|
return {}
|
|
|
|
|
|
def pick_subtitle(info: dict[str, Any],
|
|
languages: Mapping[str, str],
|
|
prefer_manual: bool = True) -> SubtitlePick | None:
|
|
"""Pick the best subtitle track for ``info`` honouring per-language mode.
|
|
|
|
See :func:`extract_video` for the ``languages`` schema. ``prefer_manual``
|
|
is only consulted for entries whose mode is ``"any"``.
|
|
"""
|
|
manual = info.get("subtitles") or {}
|
|
auto = info.get("automatic_captions") or {}
|
|
|
|
if isinstance(languages, (list, tuple)):
|
|
# legacy path: convert on the fly
|
|
default = "manual" if prefer_manual else "auto"
|
|
languages = {l: default for l in languages}
|
|
|
|
# Config order is a preference between languages we can read, not an
|
|
# instruction to accept a machine translation when the real transcript is
|
|
# sitting right there. An English channel scanned under {es, es-419, en}
|
|
# was yielding Spanish auto-translations of English speech.
|
|
#
|
|
# Two passes rather than a reorder. The reorder alone needed to know the
|
|
# spoken language, and when neither an `-orig` key nor `info["language"]`
|
|
# was present it silently fell back to config order and reintroduced the
|
|
# bug. Rejecting translations outright in the first pass needs no such
|
|
# knowledge: whatever language it lands on, it is the one actually spoken.
|
|
ordered = list(languages.items())
|
|
spoken = original_language(info)
|
|
if spoken and any(_normalize_lang(l) == spoken for l, _ in ordered):
|
|
ordered.sort(key=lambda kv: _normalize_lang(kv[0]) != spoken)
|
|
|
|
for allow_translations in (False, True):
|
|
for lang, mode in ordered:
|
|
if mode not in _LANGUAGE_MODES:
|
|
mode = "any"
|
|
sources = _sources_for(mode, prefer_manual, manual, auto)
|
|
normalized = _normalize_lang(lang)
|
|
for source_label, tracks in sources:
|
|
for track_lang, formats in _ordered_tracks(tracks, normalized):
|
|
if not allow_translations and _is_translation(track_lang, formats):
|
|
continue
|
|
pick = _pick_best_format(formats)
|
|
if pick:
|
|
return SubtitlePick(
|
|
url=pick["url"],
|
|
ext=pick["ext"],
|
|
lang=track_lang,
|
|
source=source_label,
|
|
)
|
|
return None
|
|
|
|
|
|
def _pick_best_format(formats: list[dict[str, Any]]) -> dict[str, Any] | None:
|
|
priority = ["json3", "srv1", "srv3", "vtt", "ttml"]
|
|
for ext in priority:
|
|
for fmt in formats:
|
|
if fmt.get("ext") == ext and fmt.get("url"):
|
|
return fmt
|
|
for fmt in formats:
|
|
if fmt.get("url"):
|
|
return fmt
|
|
return None
|
|
|
|
|
|
def _normalize_lang(code: str) -> str:
|
|
base = code.replace("_", "-").split("-")[0].lower()
|
|
return base
|
|
|
|
|
|
def _is_original_track(code: str) -> bool:
|
|
"""True for YouTube's original-ASR track, which it suffixes with ``-orig``.
|
|
|
|
YouTube publishes the speech-recognised track as ``<lang>-orig`` and then a
|
|
long tail of machine translations keyed by bare language code — including a
|
|
translation *into the video's own language*. So on a Spanish video both
|
|
``es-orig`` and ``es`` exist, and only the first is the real transcript.
|
|
"""
|
|
return code.replace("_", "-").lower().endswith("-orig")
|
|
|
|
|
|
def original_language(info: dict[str, Any]) -> str | None:
|
|
"""The language actually spoken in the video, normalised, or None.
|
|
|
|
Prefers the ``-orig`` track that YouTube itself publishes over ``info`` keys,
|
|
because the ``-orig`` suffix is direct evidence from the caption list while
|
|
``language`` is metadata that YouTube localises along with the title.
|
|
"""
|
|
for code in (info.get("automatic_captions") or {}):
|
|
if _is_original_track(code):
|
|
return _normalize_lang(code)
|
|
for key in ("language", "original_language"):
|
|
val = info.get(key)
|
|
if isinstance(val, str) and val:
|
|
return _normalize_lang(val)
|
|
return None
|
|
|
|
|
|
def _is_translation(code: str, formats: list[dict[str, Any]] | None) -> bool:
|
|
"""True when this track is YouTube machine-translating some other track.
|
|
|
|
The caption URL says so outright: yt-dlp builds a translated track by
|
|
appending ``tlang=`` to the base track's URL and omits it when the target
|
|
equals the source language. That is direct evidence, unlike the ``-orig``
|
|
naming convention, and it is what lets us reject a translation even for a
|
|
video whose spoken language we could not otherwise determine.
|
|
"""
|
|
if _is_original_track(code):
|
|
return False
|
|
for fmt in formats or []:
|
|
url = fmt.get("url") or ""
|
|
if "tlang=" in url:
|
|
return True
|
|
return False
|
|
|
|
|
|
def _ordered_tracks(tracks: dict, normalized: str) -> list[tuple[str, Any]]:
|
|
"""Tracks matching `normalized`, original-ASR first.
|
|
|
|
Without this the picker took whichever key yt-dlp happened to list first.
|
|
That silently returned the right thing on Spanish channels (``es-orig``
|
|
sorts before ``es``) and the wrong thing everywhere else.
|
|
"""
|
|
matches = [(c, f) for c, f in tracks.items() if _normalize_lang(c) == normalized and f]
|
|
matches.sort(key=lambda kv: not _is_original_track(kv[0]))
|
|
return matches
|
|
|
|
|
|
def _download_subtitle(url: str, *, timeout: float = 15.0) -> tuple[str | None, str | None]:
|
|
"""Fetch a caption track. Returns (text, error_description).
|
|
|
|
The error text is returned rather than only logged because a 429 here is
|
|
how YouTube throttling most often shows up on this path, and the circuit
|
|
breaker upstream can only see it if it survives into the stored reason.
|
|
"""
|
|
try:
|
|
GLOBAL_PACER.wait()
|
|
resp = yt_get(url, timeout=timeout)
|
|
resp.raise_for_status()
|
|
return resp.text, None
|
|
except Exception as exc: # pylint: disable=broad-except
|
|
log.error("Failed to download subtitle from %s: %s", url, exc)
|
|
# Lead with a normalised "HTTP Error <status>" token. Downstream, the
|
|
# throttle detector has to recognise a 429 here, and depending on the
|
|
# prose is fragile: a 429 served without a reason phrase (routine over
|
|
# HTTP/2) says nothing about "too many requests".
|
|
status = getattr(getattr(exc, "response", None), "status_code", None)
|
|
prefix = f"HTTP Error {status}: " if status else ""
|
|
return None, f"{prefix}{type(exc).__name__}: {exc}"
|