wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)

This commit is contained in:
urieljareth
2026-08-22 19:37:23 -06:00
parent 4f5a68b572
commit 8a59b39c98
103 changed files with 70954 additions and 1825 deletions
+230 -34
View File
@@ -2,12 +2,13 @@ from __future__ import annotations
import logging
from dataclasses import dataclass
from typing import Any
from typing import Any, Mapping
import requests
import yt_dlp
from ._yt_http import yt_get
from .parse import Segment, parse_auto_dump
from .ratelimit import GLOBAL_PACER, ydl_throttle_opts
log = logging.getLogger(__name__)
@@ -26,75 +27,199 @@ class VideoData:
segments: list[Segment]
subtitle: SubtitlePick | None
has_chapters: bool
# Why `segments` came back empty. "No transcript" has several very
# different causes — the video genuinely has no captions, the language
# policy rejected the tracks that do exist, or the download was throttled —
# and collapsing them into one terminal status hides recoverable failures.
skip_reason: str | None = None
_LANGUAGE_MODES = ("manual", "auto", "any")
def _sources_for(mode: str, prefer_manual: bool, manual: dict, auto: dict) -> list[tuple[str, dict]]:
"""Return ordered list of (label, tracks_dict) to try for ``mode``.
``any`` defers to the legacy :data:`prefer_manual` global default.
``manual`` / ``auto`` force the track family even if the other has
a higher-priority language elsewhere in the iteration.
"""
if mode == "manual":
return [("manual", manual)]
if mode == "auto":
return [("auto", auto)]
if prefer_manual:
return [("manual", manual), ("auto", auto)]
return [("auto", auto), ("manual", manual)]
def extract_video(
video_url: str,
languages: list[str],
languages: Mapping[str, str] | list[str],
retries: int = 10,
sleep_subrequests: float = 2.0,
prefer_manual: bool = True,
cookies_file: str | None = None,
cookies_from_browser: str | None = None,
extractor_retries: int = 3,
socket_timeout: float = 30.0,
) -> VideoData:
"""Run yt-dlp on ``video_url`` and pull the preferred subtitle track.
``languages`` may be either a list (legacy, every entry uses
``prefer_manual``) or a mapping ``{lang: mode}`` where ``mode`` is one
of ``"manual"``, ``"auto"`` or ``"any"``. The mapping form is the
preferred interface because it lets you mix per-language policies such
as ``{"en": "manual", "es": "auto", "pt": "any"}``.
"""
languages_dict = _coerce_languages(languages, prefer_manual)
ydl_opts: dict[str, Any] = {
"writesubtitles": True,
"writeautomaticsub": True,
"subtitleslangs": languages,
"subtitleslangs": list(languages_dict.keys()),
"skip_download": True,
"quiet": True,
"no_warnings": True,
"retries": retries,
"sleep_subrequests": sleep_subrequests,
"noprogress": True,
# Never let yt-dlp probe formats: it costs one HTTP request per format
# and we only ever want captions and metadata.
"check_formats": None,
**ydl_throttle_opts(
sleep_subrequests,
extractor_retries=extractor_retries,
socket_timeout=socket_timeout,
),
}
if cookies_file:
ydl_opts["cookiefile"] = cookies_file
if cookies_from_browser:
ydl_opts["cookiesfrombrowser"] = (cookies_from_browser,)
# One video extraction is two requests: the watch page and the InnerTube
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
GLOBAL_PACER.wait(cost=2)
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(video_url, download=False)
pick = pick_subtitle(info, languages, prefer_manual)
pick = pick_subtitle(info, languages_dict, prefer_manual)
segments: list[Segment] = []
skip_reason: str | None = None
if pick:
raw = _download_subtitle(pick.url)
raw, dl_error = _download_subtitle(pick.url)
if raw:
segments = parse_auto_dump(raw)
if not segments:
log.warning("Could not parse subtitle for %s (format=%s)", video_url, pick.ext)
skip_reason = f"subtitle downloaded but parsed empty (lang={pick.lang}, format={pick.ext})"
else:
skip_reason = (
f"subtitle track found (lang={pick.lang}, {pick.source}) but the download failed "
f"— usually throttling; retry later [{dl_error}]"
)
else:
skip_reason = describe_missing_subtitle(info, languages_dict)
has_chapters = bool(info.get("chapters"))
return VideoData(info=info, segments=segments, subtitle=pick, has_chapters=has_chapters)
return VideoData(
info=info, segments=segments, subtitle=pick,
has_chapters=has_chapters, skip_reason=skip_reason,
)
def pick_subtitle(info: dict[str, Any], languages: list[str], prefer_manual: bool = True) -> SubtitlePick | None:
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
A channel that only publishes auto-generated captions scanned under a
manual-only policy yields nothing — which is a config problem, not a
property of the video, and the message has to say so.
"""
manual = {k: v for k, v in (info.get("subtitles") or {}).items() if v}
auto = {k: v for k, v in (info.get("automatic_captions") or {}).items() if v}
if not manual and not auto:
return "no caption tracks published for this video"
wanted = ", ".join(f"{lang}={mode}" for lang, mode in languages.items()) or "(none configured)"
modes = {str(m).lower() for m in languages.values()}
parts = [f"no track matched the language policy ({wanted})"]
parts.append(f"available: {len(manual)} manual, {len(auto)} auto")
if auto and not manual and modes == {"manual"}:
parts.append(
"this video has ONLY auto-generated captions — set the language mode "
"to 'any' or 'auto' to use them"
)
return "; ".join(parts)
def _coerce_languages(languages: Mapping[str, str] | list[str] | None,
prefer_manual: bool) -> dict[str, str]:
"""Normalise legacy list / new dict / None into ``{lang: mode}``."""
default = "manual" if prefer_manual else "auto"
if languages is None:
return {}
if isinstance(languages, Mapping):
out: dict[str, str] = {}
for lang, mode in languages.items():
m = str(mode).lower().strip()
if m not in _LANGUAGE_MODES:
m = "any"
out[str(lang)] = m
return out
if isinstance(languages, (list, tuple)):
return {str(l): default for l in languages}
return {}
def pick_subtitle(info: dict[str, Any],
languages: Mapping[str, str],
prefer_manual: bool = True) -> SubtitlePick | None:
"""Pick the best subtitle track for ``info`` honouring per-language mode.
See :func:`extract_video` for the ``languages`` schema. ``prefer_manual``
is only consulted for entries whose mode is ``"any"``.
"""
manual = info.get("subtitles") or {}
auto = info.get("automatic_captions") or {}
ordered_sources: list[tuple[str, dict[str, Any]]]
if prefer_manual:
ordered_sources = [("manual", manual), ("auto", auto)]
else:
ordered_sources = [("auto", auto), ("manual", manual)]
if isinstance(languages, (list, tuple)):
# legacy path: convert on the fly
default = "manual" if prefer_manual else "auto"
languages = {l: default for l in languages}
for source_label, tracks in ordered_sources:
for lang in languages:
# Config order is a preference between languages we can read, not an
# instruction to accept a machine translation when the real transcript is
# sitting right there. An English channel scanned under {es, es-419, en}
# was yielding Spanish auto-translations of English speech.
#
# Two passes rather than a reorder. The reorder alone needed to know the
# spoken language, and when neither an `-orig` key nor `info["language"]`
# was present it silently fell back to config order and reintroduced the
# bug. Rejecting translations outright in the first pass needs no such
# knowledge: whatever language it lands on, it is the one actually spoken.
ordered = list(languages.items())
spoken = original_language(info)
if spoken and any(_normalize_lang(l) == spoken for l, _ in ordered):
ordered.sort(key=lambda kv: _normalize_lang(kv[0]) != spoken)
for allow_translations in (False, True):
for lang, mode in ordered:
if mode not in _LANGUAGE_MODES:
mode = "any"
sources = _sources_for(mode, prefer_manual, manual, auto)
normalized = _normalize_lang(lang)
for track_lang, formats in tracks.items():
if _normalize_lang(track_lang) != normalized:
continue
if not formats:
continue
pick = _pick_best_format(formats)
if pick:
return SubtitlePick(
url=pick["url"],
ext=pick["ext"],
lang=track_lang,
source="manual" if source_label == "manual" else "auto",
)
for source_label, tracks in sources:
for track_lang, formats in _ordered_tracks(tracks, normalized):
if not allow_translations and _is_translation(track_lang, formats):
continue
pick = _pick_best_format(formats)
if pick:
return SubtitlePick(
url=pick["url"],
ext=pick["ext"],
lang=track_lang,
source=source_label,
)
return None
@@ -115,11 +240,82 @@ def _normalize_lang(code: str) -> str:
return base
def _download_subtitle(url: str) -> str | None:
def _is_original_track(code: str) -> bool:
"""True for YouTube's original-ASR track, which it suffixes with ``-orig``.
YouTube publishes the speech-recognised track as ``<lang>-orig`` and then a
long tail of machine translations keyed by bare language code — including a
translation *into the video's own language*. So on a Spanish video both
``es-orig`` and ``es`` exist, and only the first is the real transcript.
"""
return code.replace("_", "-").lower().endswith("-orig")
def original_language(info: dict[str, Any]) -> str | None:
"""The language actually spoken in the video, normalised, or None.
Prefers the ``-orig`` track that YouTube itself publishes over ``info`` keys,
because the ``-orig`` suffix is direct evidence from the caption list while
``language`` is metadata that YouTube localises along with the title.
"""
for code in (info.get("automatic_captions") or {}):
if _is_original_track(code):
return _normalize_lang(code)
for key in ("language", "original_language"):
val = info.get(key)
if isinstance(val, str) and val:
return _normalize_lang(val)
return None
def _is_translation(code: str, formats: list[dict[str, Any]] | None) -> bool:
"""True when this track is YouTube machine-translating some other track.
The caption URL says so outright: yt-dlp builds a translated track by
appending ``tlang=`` to the base track's URL and omits it when the target
equals the source language. That is direct evidence, unlike the ``-orig``
naming convention, and it is what lets us reject a translation even for a
video whose spoken language we could not otherwise determine.
"""
if _is_original_track(code):
return False
for fmt in formats or []:
url = fmt.get("url") or ""
if "tlang=" in url:
return True
return False
def _ordered_tracks(tracks: dict, normalized: str) -> list[tuple[str, Any]]:
"""Tracks matching `normalized`, original-ASR first.
Without this the picker took whichever key yt-dlp happened to list first.
That silently returned the right thing on Spanish channels (``es-orig``
sorts before ``es``) and the wrong thing everywhere else.
"""
matches = [(c, f) for c, f in tracks.items() if _normalize_lang(c) == normalized and f]
matches.sort(key=lambda kv: not _is_original_track(kv[0]))
return matches
def _download_subtitle(url: str, *, timeout: float = 15.0) -> tuple[str | None, str | None]:
"""Fetch a caption track. Returns (text, error_description).
The error text is returned rather than only logged because a 429 here is
how YouTube throttling most often shows up on this path, and the circuit
breaker upstream can only see it if it survives into the stored reason.
"""
try:
resp = requests.get(url, timeout=15, headers={"User-Agent": "Mozilla/5.0"})
GLOBAL_PACER.wait()
resp = yt_get(url, timeout=timeout)
resp.raise_for_status()
return resp.text
except requests.RequestException as exc:
return resp.text, None
except Exception as exc: # pylint: disable=broad-except
log.error("Failed to download subtitle from %s: %s", url, exc)
return None
# Lead with a normalised "HTTP Error <status>" token. Downstream, the
# throttle detector has to recognise a 429 here, and depending on the
# prose is fragile: a 429 served without a reason phrase (routine over
# HTTP/2) says nothing about "too many requests".
status = getattr(getattr(exc, "response", None), "status_code", None)
prefix = f"HTTP Error {status}: " if status else ""
return None, f"{prefix}{type(exc).__name__}: {exc}"