feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo

Extraccion autenticada:
- import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20
- extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla
  con sesion logueada (members-only); regex y opener cacheados
- js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts
- rutas de Brave multiplataforma (Windows/macOS/Linux)

Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL,
memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics

Rendimiento:
- entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota)
- _rank_unranked con guarda (antes full-scan en cada arranque/import)
- upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL)
- thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool)
- reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md),
  healthz expone reconcile_done
- Store.transaction(): escrituras por video agrupadas (~6 commits -> 3)

Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render,
order_pending->store, keep_ref->config, seconds_to_ts solo en segments);
re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done);
fuera wrappers muertos de segments.py
This commit is contained in:
urieljareth
2026-09-10 00:32:19 -06:00
parent 1190228a81
commit b3b27ce883
21 changed files with 1494 additions and 326 deletions
+144 -2
View File
@@ -1,7 +1,12 @@
from __future__ import annotations
import json
import logging
import os
import re
import urllib.request
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Mapping
import yt_dlp
@@ -100,8 +105,20 @@ def extract_video(
# One video extraction is two requests: the watch page and the InnerTube
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
GLOBAL_PACER.wait(cost=2)
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(video_url, download=False)
try:
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(video_url, download=False)
except Exception:
# Logged-in sessions on current YouTube increasingly end here
# ("The page needs to be reloaded" / format-availability failures).
# The session itself is usually fine — the watch page still hands
# metadata and caption tracks to a plain cookie'd GET — so try that
# before giving up. Without cookies there is nothing to fall back to.
if cookies_file:
fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual)
if fallback is not None:
return fallback
raise
pick = pick_subtitle(info, languages_dict, prefer_manual)
segments: list[Segment] = []
@@ -128,6 +145,131 @@ def extract_video(
)
_WATCH_PAGE_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36"
)
# Precompiled: the fallback can fire on every video of a throttled batch, and
# re-compiling per call showed up under those runs.
_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;")
# One opener per (cookie file, mtime): the jar parse is per-call work that is
# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under
# the same path still gets a fresh jar; only the newest entry is kept.
_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {}
def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0):
path = str(Path(cookies_file).resolve())
try:
mtime = os.path.getmtime(path)
except OSError:
mtime = -1.0
key = (path, mtime)
opener = _OPENER_CACHE.get(key)
if opener is None:
jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file)
jar.load(ignore_discard=True, ignore_expires=True)
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
opener.addheaders = [
("User-Agent", _WATCH_PAGE_UA),
("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"),
]
_OPENER_CACHE.clear()
_OPENER_CACHE[key] = opener
return opener.open(url, timeout=timeout)
def extract_via_watch_page(
video_url: str,
cookies_file: str,
languages: Mapping[str, str],
prefer_manual: bool = True,
) -> VideoData | None:
"""Session-cookie fallback: scrape the watch page directly.
yt-dlp's InnerTube clients reject logged-in sessions that lack a PO
token (playability "The page needs to be reloaded") or return no
formats/captions, which kills cookie-authenticated videos — members
being the case this exists for. The plain watch page served to the
logged-in browser still carries `ytInitialPlayerResponse` with
metadata and caption tracks, so GET it with the vault cookie and
reuse the normal subtitle picker. Returns None when the page holds
no caption tracks at all, so callers keep their own error semantics.
"""
GLOBAL_PACER.wait()
html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace")
m = _INITIAL_PLAYER_RE.search(html)
if not m:
log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url)
return None
try:
pr = json.loads(m.group(1))
except json.JSONDecodeError:
log.warning("watch-page fallback: unparseable player response for %s", video_url)
return None
status = (pr.get("playabilityStatus") or {}).get("status")
if status != "OK":
reason = (pr.get("playabilityStatus") or {}).get("reason") or status
raise RuntimeError(f"watch-page fallback: video not playable ({reason})")
details = pr.get("videoDetails") or {}
micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {}
tracks = (
(pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {}
).get("captionTracks") or []
if not tracks:
return None
# Reuse pick_subtitle by shaping the tracks as an info dict.
info: dict[str, Any] = {
"title": details.get("title"),
"channel": details.get("author"),
"duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None,
"view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None,
"description": details.get("shortDescription") or "",
"tags": details.get("keywords") or [],
"thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"),
"upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None,
# The watch page carries no availability signal; leaving it unset
# keeps the (more informed) discovery value in the store.
"availability": None,
"subtitles": {},
"automatic_captions": {},
}
for t in tracks:
base = t.get("baseUrl") or ""
if not base:
continue
entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}]
if t.get("kind") == "asr":
info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry)
else:
info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry)
pick = pick_subtitle(info, languages, prefer_manual)
segments: list[Segment] = []
skip_reason: str | None = None
if pick:
try:
GLOBAL_PACER.wait()
raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace")
segments = parse_auto_dump(raw)
if not segments:
skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)"
except Exception as exc: # pylint: disable=broad-except
skip_reason = f"caption download failed via watch-page fallback: {exc}"
else:
skip_reason = describe_missing_subtitle(info, languages)
return VideoData(
info=info, segments=segments, subtitle=pick,
has_chapters=bool(info.get("chapters")), skip_reason=skip_reason,
)
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.