feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo
Extraccion autenticada: - import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20 - extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla con sesion logueada (members-only); regex y opener cacheados - js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts - rutas de Brave multiplataforma (Windows/macOS/Linux) Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL, memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics Rendimiento: - entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota) - _rank_unranked con guarda (antes full-scan en cada arranque/import) - upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL) - thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool) - reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md), healthz expone reconcile_done - Store.transaction(): escrituras por video agrupadas (~6 commits -> 3) Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render, order_pending->store, keep_ref->config, seconds_to_ts solo en segments); re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done); fuera wrappers muertos de segments.py
This commit is contained in:
+144
-2
@@ -1,7 +1,12 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import urllib.request
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any, Mapping
|
||||
|
||||
import yt_dlp
|
||||
@@ -100,8 +105,20 @@ def extract_video(
|
||||
# One video extraction is two requests: the watch page and the InnerTube
|
||||
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
|
||||
GLOBAL_PACER.wait(cost=2)
|
||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||
info = ydl.extract_info(video_url, download=False)
|
||||
try:
|
||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||
info = ydl.extract_info(video_url, download=False)
|
||||
except Exception:
|
||||
# Logged-in sessions on current YouTube increasingly end here
|
||||
# ("The page needs to be reloaded" / format-availability failures).
|
||||
# The session itself is usually fine — the watch page still hands
|
||||
# metadata and caption tracks to a plain cookie'd GET — so try that
|
||||
# before giving up. Without cookies there is nothing to fall back to.
|
||||
if cookies_file:
|
||||
fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual)
|
||||
if fallback is not None:
|
||||
return fallback
|
||||
raise
|
||||
|
||||
pick = pick_subtitle(info, languages_dict, prefer_manual)
|
||||
segments: list[Segment] = []
|
||||
@@ -128,6 +145,131 @@ def extract_video(
|
||||
)
|
||||
|
||||
|
||||
_WATCH_PAGE_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36"
|
||||
)
|
||||
|
||||
# Precompiled: the fallback can fire on every video of a throttled batch, and
|
||||
# re-compiling per call showed up under those runs.
|
||||
_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;")
|
||||
|
||||
# One opener per (cookie file, mtime): the jar parse is per-call work that is
|
||||
# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under
|
||||
# the same path still gets a fresh jar; only the newest entry is kept.
|
||||
_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {}
|
||||
|
||||
|
||||
def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0):
|
||||
path = str(Path(cookies_file).resolve())
|
||||
try:
|
||||
mtime = os.path.getmtime(path)
|
||||
except OSError:
|
||||
mtime = -1.0
|
||||
key = (path, mtime)
|
||||
opener = _OPENER_CACHE.get(key)
|
||||
if opener is None:
|
||||
jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file)
|
||||
jar.load(ignore_discard=True, ignore_expires=True)
|
||||
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
|
||||
opener.addheaders = [
|
||||
("User-Agent", _WATCH_PAGE_UA),
|
||||
("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"),
|
||||
]
|
||||
_OPENER_CACHE.clear()
|
||||
_OPENER_CACHE[key] = opener
|
||||
return opener.open(url, timeout=timeout)
|
||||
|
||||
|
||||
def extract_via_watch_page(
|
||||
video_url: str,
|
||||
cookies_file: str,
|
||||
languages: Mapping[str, str],
|
||||
prefer_manual: bool = True,
|
||||
) -> VideoData | None:
|
||||
"""Session-cookie fallback: scrape the watch page directly.
|
||||
|
||||
yt-dlp's InnerTube clients reject logged-in sessions that lack a PO
|
||||
token (playability "The page needs to be reloaded") or return no
|
||||
formats/captions, which kills cookie-authenticated videos — members
|
||||
being the case this exists for. The plain watch page served to the
|
||||
logged-in browser still carries `ytInitialPlayerResponse` with
|
||||
metadata and caption tracks, so GET it with the vault cookie and
|
||||
reuse the normal subtitle picker. Returns None when the page holds
|
||||
no caption tracks at all, so callers keep their own error semantics.
|
||||
"""
|
||||
GLOBAL_PACER.wait()
|
||||
html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace")
|
||||
m = _INITIAL_PLAYER_RE.search(html)
|
||||
if not m:
|
||||
log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url)
|
||||
return None
|
||||
try:
|
||||
pr = json.loads(m.group(1))
|
||||
except json.JSONDecodeError:
|
||||
log.warning("watch-page fallback: unparseable player response for %s", video_url)
|
||||
return None
|
||||
|
||||
status = (pr.get("playabilityStatus") or {}).get("status")
|
||||
if status != "OK":
|
||||
reason = (pr.get("playabilityStatus") or {}).get("reason") or status
|
||||
raise RuntimeError(f"watch-page fallback: video not playable ({reason})")
|
||||
|
||||
details = pr.get("videoDetails") or {}
|
||||
micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {}
|
||||
tracks = (
|
||||
(pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {}
|
||||
).get("captionTracks") or []
|
||||
if not tracks:
|
||||
return None
|
||||
|
||||
# Reuse pick_subtitle by shaping the tracks as an info dict.
|
||||
info: dict[str, Any] = {
|
||||
"title": details.get("title"),
|
||||
"channel": details.get("author"),
|
||||
"duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None,
|
||||
"view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None,
|
||||
"description": details.get("shortDescription") or "",
|
||||
"tags": details.get("keywords") or [],
|
||||
"thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"),
|
||||
"upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None,
|
||||
# The watch page carries no availability signal; leaving it unset
|
||||
# keeps the (more informed) discovery value in the store.
|
||||
"availability": None,
|
||||
"subtitles": {},
|
||||
"automatic_captions": {},
|
||||
}
|
||||
for t in tracks:
|
||||
base = t.get("baseUrl") or ""
|
||||
if not base:
|
||||
continue
|
||||
entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}]
|
||||
if t.get("kind") == "asr":
|
||||
info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry)
|
||||
else:
|
||||
info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry)
|
||||
|
||||
pick = pick_subtitle(info, languages, prefer_manual)
|
||||
segments: list[Segment] = []
|
||||
skip_reason: str | None = None
|
||||
if pick:
|
||||
try:
|
||||
GLOBAL_PACER.wait()
|
||||
raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace")
|
||||
segments = parse_auto_dump(raw)
|
||||
if not segments:
|
||||
skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)"
|
||||
except Exception as exc: # pylint: disable=broad-except
|
||||
skip_reason = f"caption download failed via watch-page fallback: {exc}"
|
||||
else:
|
||||
skip_reason = describe_missing_subtitle(info, languages)
|
||||
|
||||
return VideoData(
|
||||
info=info, segments=segments, subtitle=pick,
|
||||
has_chapters=bool(info.get("chapters")), skip_reason=skip_reason,
|
||||
)
|
||||
|
||||
|
||||
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
|
||||
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user