wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)
This commit is contained in:
+230
-34
@@ -2,12 +2,13 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
from typing import Any, Mapping
|
||||
|
||||
import requests
|
||||
import yt_dlp
|
||||
|
||||
from ._yt_http import yt_get
|
||||
from .parse import Segment, parse_auto_dump
|
||||
from .ratelimit import GLOBAL_PACER, ydl_throttle_opts
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -26,75 +27,199 @@ class VideoData:
|
||||
segments: list[Segment]
|
||||
subtitle: SubtitlePick | None
|
||||
has_chapters: bool
|
||||
# Why `segments` came back empty. "No transcript" has several very
|
||||
# different causes — the video genuinely has no captions, the language
|
||||
# policy rejected the tracks that do exist, or the download was throttled —
|
||||
# and collapsing them into one terminal status hides recoverable failures.
|
||||
skip_reason: str | None = None
|
||||
|
||||
|
||||
_LANGUAGE_MODES = ("manual", "auto", "any")
|
||||
|
||||
|
||||
def _sources_for(mode: str, prefer_manual: bool, manual: dict, auto: dict) -> list[tuple[str, dict]]:
|
||||
"""Return ordered list of (label, tracks_dict) to try for ``mode``.
|
||||
|
||||
``any`` defers to the legacy :data:`prefer_manual` global default.
|
||||
``manual`` / ``auto`` force the track family even if the other has
|
||||
a higher-priority language elsewhere in the iteration.
|
||||
"""
|
||||
if mode == "manual":
|
||||
return [("manual", manual)]
|
||||
if mode == "auto":
|
||||
return [("auto", auto)]
|
||||
if prefer_manual:
|
||||
return [("manual", manual), ("auto", auto)]
|
||||
return [("auto", auto), ("manual", manual)]
|
||||
|
||||
|
||||
def extract_video(
|
||||
video_url: str,
|
||||
languages: list[str],
|
||||
languages: Mapping[str, str] | list[str],
|
||||
retries: int = 10,
|
||||
sleep_subrequests: float = 2.0,
|
||||
prefer_manual: bool = True,
|
||||
cookies_file: str | None = None,
|
||||
cookies_from_browser: str | None = None,
|
||||
extractor_retries: int = 3,
|
||||
socket_timeout: float = 30.0,
|
||||
) -> VideoData:
|
||||
"""Run yt-dlp on ``video_url`` and pull the preferred subtitle track.
|
||||
|
||||
``languages`` may be either a list (legacy, every entry uses
|
||||
``prefer_manual``) or a mapping ``{lang: mode}`` where ``mode`` is one
|
||||
of ``"manual"``, ``"auto"`` or ``"any"``. The mapping form is the
|
||||
preferred interface because it lets you mix per-language policies such
|
||||
as ``{"en": "manual", "es": "auto", "pt": "any"}``.
|
||||
"""
|
||||
languages_dict = _coerce_languages(languages, prefer_manual)
|
||||
|
||||
ydl_opts: dict[str, Any] = {
|
||||
"writesubtitles": True,
|
||||
"writeautomaticsub": True,
|
||||
"subtitleslangs": languages,
|
||||
"subtitleslangs": list(languages_dict.keys()),
|
||||
"skip_download": True,
|
||||
"quiet": True,
|
||||
"no_warnings": True,
|
||||
"retries": retries,
|
||||
"sleep_subrequests": sleep_subrequests,
|
||||
"noprogress": True,
|
||||
# Never let yt-dlp probe formats: it costs one HTTP request per format
|
||||
# and we only ever want captions and metadata.
|
||||
"check_formats": None,
|
||||
**ydl_throttle_opts(
|
||||
sleep_subrequests,
|
||||
extractor_retries=extractor_retries,
|
||||
socket_timeout=socket_timeout,
|
||||
),
|
||||
}
|
||||
if cookies_file:
|
||||
ydl_opts["cookiefile"] = cookies_file
|
||||
if cookies_from_browser:
|
||||
ydl_opts["cookiesfrombrowser"] = (cookies_from_browser,)
|
||||
|
||||
# One video extraction is two requests: the watch page and the InnerTube
|
||||
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
|
||||
GLOBAL_PACER.wait(cost=2)
|
||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||
info = ydl.extract_info(video_url, download=False)
|
||||
|
||||
pick = pick_subtitle(info, languages, prefer_manual)
|
||||
pick = pick_subtitle(info, languages_dict, prefer_manual)
|
||||
segments: list[Segment] = []
|
||||
skip_reason: str | None = None
|
||||
if pick:
|
||||
raw = _download_subtitle(pick.url)
|
||||
raw, dl_error = _download_subtitle(pick.url)
|
||||
if raw:
|
||||
segments = parse_auto_dump(raw)
|
||||
if not segments:
|
||||
log.warning("Could not parse subtitle for %s (format=%s)", video_url, pick.ext)
|
||||
skip_reason = f"subtitle downloaded but parsed empty (lang={pick.lang}, format={pick.ext})"
|
||||
else:
|
||||
skip_reason = (
|
||||
f"subtitle track found (lang={pick.lang}, {pick.source}) but the download failed "
|
||||
f"— usually throttling; retry later [{dl_error}]"
|
||||
)
|
||||
else:
|
||||
skip_reason = describe_missing_subtitle(info, languages_dict)
|
||||
|
||||
has_chapters = bool(info.get("chapters"))
|
||||
return VideoData(info=info, segments=segments, subtitle=pick, has_chapters=has_chapters)
|
||||
return VideoData(
|
||||
info=info, segments=segments, subtitle=pick,
|
||||
has_chapters=has_chapters, skip_reason=skip_reason,
|
||||
)
|
||||
|
||||
|
||||
def pick_subtitle(info: dict[str, Any], languages: list[str], prefer_manual: bool = True) -> SubtitlePick | None:
|
||||
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
|
||||
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
|
||||
|
||||
A channel that only publishes auto-generated captions scanned under a
|
||||
manual-only policy yields nothing — which is a config problem, not a
|
||||
property of the video, and the message has to say so.
|
||||
"""
|
||||
manual = {k: v for k, v in (info.get("subtitles") or {}).items() if v}
|
||||
auto = {k: v for k, v in (info.get("automatic_captions") or {}).items() if v}
|
||||
if not manual and not auto:
|
||||
return "no caption tracks published for this video"
|
||||
|
||||
wanted = ", ".join(f"{lang}={mode}" for lang, mode in languages.items()) or "(none configured)"
|
||||
modes = {str(m).lower() for m in languages.values()}
|
||||
parts = [f"no track matched the language policy ({wanted})"]
|
||||
parts.append(f"available: {len(manual)} manual, {len(auto)} auto")
|
||||
if auto and not manual and modes == {"manual"}:
|
||||
parts.append(
|
||||
"this video has ONLY auto-generated captions — set the language mode "
|
||||
"to 'any' or 'auto' to use them"
|
||||
)
|
||||
return "; ".join(parts)
|
||||
|
||||
|
||||
def _coerce_languages(languages: Mapping[str, str] | list[str] | None,
|
||||
prefer_manual: bool) -> dict[str, str]:
|
||||
"""Normalise legacy list / new dict / None into ``{lang: mode}``."""
|
||||
default = "manual" if prefer_manual else "auto"
|
||||
if languages is None:
|
||||
return {}
|
||||
if isinstance(languages, Mapping):
|
||||
out: dict[str, str] = {}
|
||||
for lang, mode in languages.items():
|
||||
m = str(mode).lower().strip()
|
||||
if m not in _LANGUAGE_MODES:
|
||||
m = "any"
|
||||
out[str(lang)] = m
|
||||
return out
|
||||
if isinstance(languages, (list, tuple)):
|
||||
return {str(l): default for l in languages}
|
||||
return {}
|
||||
|
||||
|
||||
def pick_subtitle(info: dict[str, Any],
|
||||
languages: Mapping[str, str],
|
||||
prefer_manual: bool = True) -> SubtitlePick | None:
|
||||
"""Pick the best subtitle track for ``info`` honouring per-language mode.
|
||||
|
||||
See :func:`extract_video` for the ``languages`` schema. ``prefer_manual``
|
||||
is only consulted for entries whose mode is ``"any"``.
|
||||
"""
|
||||
manual = info.get("subtitles") or {}
|
||||
auto = info.get("automatic_captions") or {}
|
||||
|
||||
ordered_sources: list[tuple[str, dict[str, Any]]]
|
||||
if prefer_manual:
|
||||
ordered_sources = [("manual", manual), ("auto", auto)]
|
||||
else:
|
||||
ordered_sources = [("auto", auto), ("manual", manual)]
|
||||
if isinstance(languages, (list, tuple)):
|
||||
# legacy path: convert on the fly
|
||||
default = "manual" if prefer_manual else "auto"
|
||||
languages = {l: default for l in languages}
|
||||
|
||||
for source_label, tracks in ordered_sources:
|
||||
for lang in languages:
|
||||
# Config order is a preference between languages we can read, not an
|
||||
# instruction to accept a machine translation when the real transcript is
|
||||
# sitting right there. An English channel scanned under {es, es-419, en}
|
||||
# was yielding Spanish auto-translations of English speech.
|
||||
#
|
||||
# Two passes rather than a reorder. The reorder alone needed to know the
|
||||
# spoken language, and when neither an `-orig` key nor `info["language"]`
|
||||
# was present it silently fell back to config order and reintroduced the
|
||||
# bug. Rejecting translations outright in the first pass needs no such
|
||||
# knowledge: whatever language it lands on, it is the one actually spoken.
|
||||
ordered = list(languages.items())
|
||||
spoken = original_language(info)
|
||||
if spoken and any(_normalize_lang(l) == spoken for l, _ in ordered):
|
||||
ordered.sort(key=lambda kv: _normalize_lang(kv[0]) != spoken)
|
||||
|
||||
for allow_translations in (False, True):
|
||||
for lang, mode in ordered:
|
||||
if mode not in _LANGUAGE_MODES:
|
||||
mode = "any"
|
||||
sources = _sources_for(mode, prefer_manual, manual, auto)
|
||||
normalized = _normalize_lang(lang)
|
||||
for track_lang, formats in tracks.items():
|
||||
if _normalize_lang(track_lang) != normalized:
|
||||
continue
|
||||
if not formats:
|
||||
continue
|
||||
pick = _pick_best_format(formats)
|
||||
if pick:
|
||||
return SubtitlePick(
|
||||
url=pick["url"],
|
||||
ext=pick["ext"],
|
||||
lang=track_lang,
|
||||
source="manual" if source_label == "manual" else "auto",
|
||||
)
|
||||
for source_label, tracks in sources:
|
||||
for track_lang, formats in _ordered_tracks(tracks, normalized):
|
||||
if not allow_translations and _is_translation(track_lang, formats):
|
||||
continue
|
||||
pick = _pick_best_format(formats)
|
||||
if pick:
|
||||
return SubtitlePick(
|
||||
url=pick["url"],
|
||||
ext=pick["ext"],
|
||||
lang=track_lang,
|
||||
source=source_label,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
@@ -115,11 +240,82 @@ def _normalize_lang(code: str) -> str:
|
||||
return base
|
||||
|
||||
|
||||
def _download_subtitle(url: str) -> str | None:
|
||||
def _is_original_track(code: str) -> bool:
|
||||
"""True for YouTube's original-ASR track, which it suffixes with ``-orig``.
|
||||
|
||||
YouTube publishes the speech-recognised track as ``<lang>-orig`` and then a
|
||||
long tail of machine translations keyed by bare language code — including a
|
||||
translation *into the video's own language*. So on a Spanish video both
|
||||
``es-orig`` and ``es`` exist, and only the first is the real transcript.
|
||||
"""
|
||||
return code.replace("_", "-").lower().endswith("-orig")
|
||||
|
||||
|
||||
def original_language(info: dict[str, Any]) -> str | None:
|
||||
"""The language actually spoken in the video, normalised, or None.
|
||||
|
||||
Prefers the ``-orig`` track that YouTube itself publishes over ``info`` keys,
|
||||
because the ``-orig`` suffix is direct evidence from the caption list while
|
||||
``language`` is metadata that YouTube localises along with the title.
|
||||
"""
|
||||
for code in (info.get("automatic_captions") or {}):
|
||||
if _is_original_track(code):
|
||||
return _normalize_lang(code)
|
||||
for key in ("language", "original_language"):
|
||||
val = info.get(key)
|
||||
if isinstance(val, str) and val:
|
||||
return _normalize_lang(val)
|
||||
return None
|
||||
|
||||
|
||||
def _is_translation(code: str, formats: list[dict[str, Any]] | None) -> bool:
|
||||
"""True when this track is YouTube machine-translating some other track.
|
||||
|
||||
The caption URL says so outright: yt-dlp builds a translated track by
|
||||
appending ``tlang=`` to the base track's URL and omits it when the target
|
||||
equals the source language. That is direct evidence, unlike the ``-orig``
|
||||
naming convention, and it is what lets us reject a translation even for a
|
||||
video whose spoken language we could not otherwise determine.
|
||||
"""
|
||||
if _is_original_track(code):
|
||||
return False
|
||||
for fmt in formats or []:
|
||||
url = fmt.get("url") or ""
|
||||
if "tlang=" in url:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _ordered_tracks(tracks: dict, normalized: str) -> list[tuple[str, Any]]:
|
||||
"""Tracks matching `normalized`, original-ASR first.
|
||||
|
||||
Without this the picker took whichever key yt-dlp happened to list first.
|
||||
That silently returned the right thing on Spanish channels (``es-orig``
|
||||
sorts before ``es``) and the wrong thing everywhere else.
|
||||
"""
|
||||
matches = [(c, f) for c, f in tracks.items() if _normalize_lang(c) == normalized and f]
|
||||
matches.sort(key=lambda kv: not _is_original_track(kv[0]))
|
||||
return matches
|
||||
|
||||
|
||||
def _download_subtitle(url: str, *, timeout: float = 15.0) -> tuple[str | None, str | None]:
|
||||
"""Fetch a caption track. Returns (text, error_description).
|
||||
|
||||
The error text is returned rather than only logged because a 429 here is
|
||||
how YouTube throttling most often shows up on this path, and the circuit
|
||||
breaker upstream can only see it if it survives into the stored reason.
|
||||
"""
|
||||
try:
|
||||
resp = requests.get(url, timeout=15, headers={"User-Agent": "Mozilla/5.0"})
|
||||
GLOBAL_PACER.wait()
|
||||
resp = yt_get(url, timeout=timeout)
|
||||
resp.raise_for_status()
|
||||
return resp.text
|
||||
except requests.RequestException as exc:
|
||||
return resp.text, None
|
||||
except Exception as exc: # pylint: disable=broad-except
|
||||
log.error("Failed to download subtitle from %s: %s", url, exc)
|
||||
return None
|
||||
# Lead with a normalised "HTTP Error <status>" token. Downstream, the
|
||||
# throttle detector has to recognise a 429 here, and depending on the
|
||||
# prose is fragile: a 429 served without a reason phrase (routine over
|
||||
# HTTP/2) says nothing about "too many requests".
|
||||
status = getattr(getattr(exc, "response", None), "status_code", None)
|
||||
prefix = f"HTTP Error {status}: " if status else ""
|
||||
return None, f"{prefix}{type(exc).__name__}: {exc}"
|
||||
|
||||
Reference in New Issue
Block a user