feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo

Extraccion autenticada:
- import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20
- extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla
  con sesion logueada (members-only); regex y opener cacheados
- js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts
- rutas de Brave multiplataforma (Windows/macOS/Linux)

Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL,
memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics

Rendimiento:
- entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota)
- _rank_unranked con guarda (antes full-scan en cada arranque/import)
- upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL)
- thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool)
- reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md),
  healthz expone reconcile_done
- Store.transaction(): escrituras por video agrupadas (~6 commits -> 3)

Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render,
order_pending->store, keep_ref->config, seconds_to_ts solo en segments);
re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done);
fuera wrappers muertos de segments.py
This commit is contained in:
urieljareth
2026-09-10 00:32:19 -06:00
parent 1190228a81
commit b3b27ce883
21 changed files with 1494 additions and 326 deletions
+24 -71
View File
@@ -13,11 +13,10 @@ from rich.progress import Progress, SpinnerColumn, TextColumn, BarColumn, TaskPr
from rich.table import Table from rich.table import Table
from .config import Config, load_config, parse_languages from .config import Config, load_config, parse_languages
from .store import Store, VideoRef, VideoRow from .store import Store, order_pending
from .discover import discover_channel, discover_incremental from .discover import discover_channel, discover_incremental, extract_handle
from .chapters import align_chapters, chapters_from_info, Chapter, Section from .segments import seconds_to_ts
from .parse import Segment from .render import safe_filename
from .render import build_filename_stem, render_markdown
from .ratelimit import ThrottleGuard, configure_global_pacer, polite_sleep from .ratelimit import ThrottleGuard, configure_global_pacer, polite_sleep
from .pipeline import process_video from .pipeline import process_video
from .cookies import auto_import_dir, resolve_active_path from .cookies import auto_import_dir, resolve_active_path
@@ -157,7 +156,7 @@ def _run_scrape(obj, limit, since, languages, no_auto, no_shorts, include_shorts
channel_id, channel_name, refs = result.channel_id, result.channel_name, result.refs channel_id, channel_name, refs = result.channel_id, result.channel_name, result.refs
if result.full_scan: if result.full_scan:
store.upsert_channel(channel_id, _extract_handle(cfg.channel_url), channel_name, len(refs)) store.upsert_channel(channel_id, extract_handle(cfg.channel_url), channel_name, len(refs))
else: else:
store.update_channel_meta(channel_id, name=channel_name) store.update_channel_meta(channel_id, name=channel_name)
console.print( console.print(
@@ -190,7 +189,7 @@ def _run_scrape(obj, limit, since, languages, no_auto, no_shorts, include_shorts
else: else:
# Discovery only saw the newest slice; keep the older backlog reachable # Discovery only saw the newest slice; keep the older backlog reachable
# but put this run's videos first so --limit still means "los mas nuevos". # but put this run's videos first so --limit still means "los mas nuevos".
pending = _order_pending(pending, refs) pending = order_pending(pending, refs)
if limit: if limit:
pending = pending[:limit] pending = pending[:limit]
if not pending: if not pending:
@@ -244,7 +243,7 @@ def _known_channel_for(store: Store, channel_url: str) -> str | None:
""" """
if not channel_url: if not channel_url:
return None return None
handle = _extract_handle(channel_url).lstrip("@").lower() handle = extract_handle(channel_url).lstrip("@").lower()
tail = channel_url.rstrip("/").split("/")[-1] tail = channel_url.rstrip("/").split("/")[-1]
for ch in store.list_channels(): for ch in store.list_channels():
cid = ch.get("channel_id") or "" cid = ch.get("channel_id") or ""
@@ -256,14 +255,6 @@ def _known_channel_for(store: Store, channel_url: str) -> str | None:
return None return None
def _order_pending(rows: list, refs: list) -> list:
"""Videos from this run's window first, then the rest of the backlog."""
by_id = {r.video_id: r for r in rows}
ordered = [by_id.pop(r.video_id) for r in refs if r.video_id in by_id]
ordered.extend(by_id.values())
return ordered
def _print_dry_run(refs): def _print_dry_run(refs):
table = Table(show_lines=False) table = Table(show_lines=False)
table.add_column("Fecha", style="dim") table.add_column("Fecha", style="dim")
@@ -388,7 +379,13 @@ def export_cmd(obj, fmt, out_dir):
def audio_cmd(obj, limit, since, force): def audio_cmd(obj, limit, since, force):
"""Descarga audio MP3 (requiere ffmpeg).""" """Descarga audio MP3 (requiere ffmpeg)."""
if not shutil.which("ffmpeg"): if not shutil.which("ffmpeg"):
console.print("[red]ffmpeg no encontrado.[/red] Instala: winget install ffmpeg") if sys.platform == "win32":
hint = "winget install ffmpeg (o choco install ffmpeg)"
elif sys.platform == "darwin":
hint = "brew install ffmpeg"
else:
hint = "apt install ffmpeg (Debian/Ubuntu) / dnf install ffmpeg (Fedora) / pacman -S ffmpeg (Arch)"
console.print(f"[red]ffmpeg no encontrado.[/red] Instala: {hint}")
sys.exit(1) sys.exit(1)
store: Store = obj.store store: Store = obj.store
cfg: Config = obj.cfg cfg: Config = obj.cfg
@@ -408,6 +405,7 @@ def audio_cmd(obj, limit, since, force):
"outtmpl": str(out_dir / "%(title)s.%(ext)s"), "outtmpl": str(out_dir / "%(title)s.%(ext)s"),
"postprocessors": [{"key": "FFmpegExtractAudio", "preferredcodec": "mp3", "preferredquality": "128"}], "postprocessors": [{"key": "FFmpegExtractAudio", "preferredcodec": "mp3", "preferredquality": "128"}],
"quiet": True, "no_warnings": True, "noprogress": True, "quiet": True, "no_warnings": True, "noprogress": True,
"js_runtimes": {"node": {}, "deno": {}, "bun": {}, "quickjs": {}},
} }
if cookie_path: if cookie_path:
ydl_opts["cookiefile"] = cookie_path ydl_opts["cookiefile"] = cookie_path
@@ -415,7 +413,7 @@ def audio_cmd(obj, limit, since, force):
ydl_opts["cookiesfrombrowser"] = (obj.cookies_from_browser,) ydl_opts["cookiesfrombrowser"] = (obj.cookies_from_browser,)
with yt_dlp.YoutubeDL(ydl_opts) as ydl: with yt_dlp.YoutubeDL(ydl_opts) as ydl:
for v in videos: for v in videos:
target = out_dir / f"{_safe_filename(v.title or v.video_id)}.mp3" target = out_dir / f"{safe_filename(v.title or v.video_id)}.mp3"
if target.exists() and not force: if target.exists() and not force:
continue continue
try: try:
@@ -454,7 +452,7 @@ def channels_add(obj, url):
cfg.channel_url = url cfg.channel_url = url
console.print("[cyan]Resolviendo canal...[/cyan]") console.print("[cyan]Resolviendo canal...[/cyan]")
channel_id, name, avatar, refs = discover_channel(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests) channel_id, name, avatar, refs = discover_channel(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
obj.store.upsert_channel(channel_id, _extract_handle(url), name, len(refs), avatar=avatar) obj.store.upsert_channel(channel_id, extract_handle(url), name, len(refs), avatar=avatar)
obj.store.upsert_videos(refs) obj.store.upsert_videos(refs)
console.print(f"[green]Added:[/green] {name} ({channel_id}) — {len(refs)} videos") console.print(f"[green]Added:[/green] {name} ({channel_id}) — {len(refs)} videos")
@@ -549,6 +547,7 @@ def re_render_cmd(obj, backfill):
"""Regenerar Markdown desde segmentos almacenados.""" """Regenerar Markdown desde segmentos almacenados."""
store: Store = obj.store store: Store = obj.store
cfg: Config = obj.cfg cfg: Config = obj.cfg
from .pipeline import re_render_videos
from .segments import backfill_from_markdown from .segments import backfill_from_markdown
if backfill: if backfill:
md_root = Path(cfg.output_dir_resolved) md_root = Path(cfg.output_dir_resolved)
@@ -559,26 +558,13 @@ def re_render_cmd(obj, backfill):
if not videos: if not videos:
console.print("[yellow]No hay videos con segments_json. Usa --backfill.[/yellow]") console.print("[yellow]No hay videos con segments_json. Usa --backfill.[/yellow]")
return return
import json
from .render import render_markdown
console.print(f"[cyan]Re-renderizando {len(videos)} videos...[/cyan]") console.print(f"[cyan]Re-renderizando {len(videos)} videos...[/cyan]")
for v in videos: # Delegate to the pipeline implementation instead of a local loop: it
segs = [Segment(start=s["start"], end=s["end"], text=s["text"]) for s in json.loads(v.segments_json)] # retires the .md a re-render supersedes (renamed titles used to leave
chapters = [Chapter(title=c["title"], start_time=c["start"], end_time=c.get("end", c["start"])) for c in json.loads(v.chapters_json or "[]")] # orphans), re-points markdown_path via mark_done and fills channel_name
sections = align_chapters(segs, chapters) # from the channels table.
context = { n = sum(re_render_videos(store, cfg, cid) for cid in targets)
"video_id": v.video_id, "title": v.title or v.video_id, "channel_name": "", console.print(f"[green]Re-render completo:[/green] {n} videos")
"channel_id": v.channel_id, "channel_url": "", "upload_date": v.upload_date or "",
"duration": v.duration or 0, "url": v.url, "transcript_lang": v.transcript_lang or "",
"transcript_src": v.transcript_src or "", "view_count": v.view_count, "like_count": v.like_count,
"tags": _parse_tags(v.tags), "thumbnail": v.thumbnail or "", "description": v.description or "",
"sections": sections,
}
stem = build_filename_stem(v.upload_date, v.title or v.video_id, cfg.filename_template)
ch = store.get_channel(v.channel_id)
out_subdir = Path(cfg.output_dir_resolved) / _safe_dirname((ch or {}).get("name") or "unknown")
render_markdown(cfg.template_path_resolved, out_subdir, stem, context)
console.print(f"[green]Re-render completo.[/green]")
# --------------------------------------------------------------------------- helpers # --------------------------------------------------------------------------- helpers
@@ -612,12 +598,6 @@ def _title_for(store: Store, video_id: str) -> str | None:
return v.title if v else None return v.title if v else None
def _extract_handle(url: str) -> str:
if "@" in url:
return "@" + url.split("@", 1)[1].split("/", 1)[0]
return ""
def _fmt_date(d: str | None) -> str: def _fmt_date(d: str | None) -> str:
if not d: if not d:
return "" return ""
@@ -634,33 +614,6 @@ def _fmt_duration(seconds: int | None) -> str:
return f"{h}:{m:02d}:{s:02d}" if h else f"{m}:{s:02d}" return f"{h}:{m:02d}:{s:02d}" if h else f"{m}:{s:02d}"
def seconds_to_ts(sec: float) -> str:
total = int(sec)
h, rem = divmod(total, 3600)
m, s = divmod(rem, 60)
return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}"
def _safe_dirname(name: str) -> str:
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
return safe.strip().strip(".") or "unknown"
def _safe_filename(name: str) -> str:
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
return safe.strip().strip(".") or "untitled"
def _parse_tags(tags_json: str | None) -> list[str]:
if not tags_json:
return []
import json
try:
return json.loads(tags_json)
except (json.JSONDecodeError, TypeError):
return []
def _parse_interval(s: str) -> float: def _parse_interval(s: str) -> float:
s = s.strip().lower() s = s.strip().lower()
if s.endswith("s"): if s.endswith("s"):
+24
View File
@@ -105,6 +105,30 @@ class Config:
LANGUAGE_MODES = ("manual", "auto", "any") LANGUAGE_MODES = ("manual", "auto", "any")
def keep_ref(cfg: Config, include_shorts: bool | None = None, no_live: bool | None = None):
"""Predicate matching the shorts/live rules that decide what reaches the DB.
Lives next to the `Config` flags it reads so the policy has one home.
Incremental discovery needs the same filter its stored ids were created
under, otherwise the tail of a window is full of entries that can never be
recognised as known and the window keeps widening for nothing. The optional
overrides serve the job runner, whose per-run opts may disagree with the
config the store was populated under.
"""
shorts = cfg.include_shorts if include_shorts is None else include_shorts
skip_live = (not cfg.include_live) if no_live is None else no_live
def keep(r) -> bool: # VideoRef or VideoRow — both carry `.url`
url = r.url or ""
if not shorts and "/shorts/" in url:
return False
if skip_live and url.startswith("https://www.youtube.com/live/"):
return False
return True
return keep
def parse_languages(raw: Any, prefer_manual: bool) -> dict[str, str]: def parse_languages(raw: Any, prefer_manual: bool) -> dict[str, str]:
"""Normalise legacy list / new dict / None into ``{lang: mode}``.""" """Normalise legacy list / new dict / None into ``{lang: mode}``."""
default = "manual" if prefer_manual else "auto" default = "manual" if prefer_manual else "auto"
+238 -13
View File
@@ -1,6 +1,7 @@
from __future__ import annotations from __future__ import annotations
import logging import logging
import os
import uuid import uuid
from datetime import datetime, timezone from datetime import datetime, timezone
from pathlib import Path from pathlib import Path
@@ -10,7 +11,12 @@ from .store import CookieRow, Store
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
# A logged-in YouTube session always sets this core set. A lone
# `__Secure-3PSID` (partial extension export) is NOT a session — YouTube
# treats the request as anonymous and members-only content stays locked,
# which is how an "active membership" cookie once failed invisibly.
SESSION_COOKIE_NAMES = {"SID", "SAPISID", "__Secure-3PSID", "SSID", "LOGIN_INFO", "HSID", "APISID"} SESSION_COOKIE_NAMES = {"SID", "SAPISID", "__Secure-3PSID", "SSID", "LOGIN_INFO", "HSID", "APISID"}
FULL_SESSION_NAMES = {"SID", "HSID", "SSID"}
_DEFAULT_DIR = Path("cookies") _DEFAULT_DIR = Path("cookies")
@@ -28,10 +34,16 @@ def parse_netscape(text: str) -> tuple[bool, dict]:
""" """
lines = text.splitlines() lines = text.splitlines()
expiries: list[int] = [] expiries: list[int] = []
session_expiries: list[int] = []
names: set[str] = set() names: set[str] = set()
count = 0 count = 0
for line in lines: for line in lines:
line = line.rstrip("\n") line = line.rstrip("\n")
# "#HttpOnly_" is a data prefix, not a comment — and the session
# cookies themselves (SID, HSID, ...) are HttpOnly, so skipping
# these lines would silently strip the login out of an export.
if line.startswith("#HttpOnly_"):
line = line[len("#HttpOnly_"):]
if not line.strip() or line.startswith("#"): if not line.strip() or line.startswith("#"):
continue continue
parts = line.split("\t") parts = line.split("\t")
@@ -48,14 +60,20 @@ def parse_netscape(text: str) -> tuple[bool, dict]:
count += 1 count += 1
if expiry: if expiry:
expiries.append(expiry) expiries.append(expiry)
if name in SESSION_COOKIE_NAMES:
session_expiries.append(expiry)
if count == 0: if count == 0:
return False, {"count": 0, "has_session": False, "expires_at": None, "names": set()} return False, {"count": 0, "has_session": False, "expires_at": None, "names": set()}
has_session = bool(names & SESSION_COOKIE_NAMES) has_session = FULL_SESSION_NAMES <= names or "LOGIN_INFO" in names
# What the user needs to know is when the LOGIN dies, not when the
# earliest throwaway cookie (YSC and friends live hours) lapses — so
# report the session cookies' own expiry, falling back to the file max.
relevant = session_expiries or expiries
expires_at = None expires_at = None
if expiries: if relevant:
earliest = min(expiries) latest = max(relevant)
if earliest > 0: if latest > 0:
expires_at = datetime.fromtimestamp(earliest, tz=timezone.utc).isoformat(timespec="seconds") expires_at = datetime.fromtimestamp(latest, tz=timezone.utc).isoformat(timespec="seconds")
return True, {"count": count, "has_session": has_session, "expires_at": expires_at, "names": names} return True, {"count": count, "has_session": has_session, "expires_at": expires_at, "names": names}
@@ -99,10 +117,15 @@ def import_text(store: Store, text: str, label: str, cookie_dir: str | Path | No
def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int: def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int:
"""Import any loose .txt Netscape files in dir that aren't tracked yet. Returns count.""" """Import any loose .txt Netscape files in dir that aren't tracked yet. Returns count."""
cdir = cookies_dir(dir_path) cdir = cookies_dir(dir_path)
tracked = {c.filename for c in store.list_cookies()} tracked = store.list_cookies()
for c in tracked:
if not (cdir / c.filename).exists():
store.delete_cookie(c.id)
tracked_filenames = {c.filename for c in store.list_cookies()}
n = 0 n = 0
for f in sorted(cdir.glob("*.txt")): for f in sorted(cdir.glob("*.txt")):
if f.name in tracked: if f.name in tracked_filenames:
continue continue
ok, info = parse_netscape_file(f) ok, info = parse_netscape_file(f)
if not ok: if not ok:
@@ -118,14 +141,210 @@ def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int:
cookie_count=info["count"], cookie_count=info["count"],
) )
n += 1 n += 1
# activate first cookie if none active # activate first cookie if none active or active file missing
if not store.get_active_cookie(): active = store.get_active_cookie()
if not active or not (cdir / active.filename).exists():
cookies = store.list_cookies() cookies = store.list_cookies()
if cookies: if cookies:
store.set_active_cookie(cookies[0].id) store.set_active_cookie(cookies[0].id)
return n return n
class BrowserCookieLockedError(RuntimeError):
"""The browser's cookie database could not be read — it is running.
Chromium opens its Cookies file without sharing read access, so the
browser (including its background/tray processes) must be fully closed
for the extraction to succeed.
"""
# Where the Brave executable lives on a default install, in preference
# order. The %VAR% placeholders only expand on Windows (elsewhere they stay
# literal and simply never match), so one flat cross-platform list works.
_BRAVE_EXE_CANDIDATES = (
# Windows
r"%ProgramFiles%\BraveSoftware\Brave-Browser\Application\brave.exe",
r"%LocalAppData%\BraveSoftware\Brave-Browser\Application\brave.exe",
# macOS (system-wide Applications and per-user ~/Applications)
"/Applications/Brave Browser.app/Contents/MacOS/Brave Browser",
"~/Applications/Brave Browser.app/Contents/MacOS/Brave Browser",
# Linux (official deb/rpm packages, distro builds, snap, manual installs)
"/usr/bin/brave-browser",
"/usr/bin/brave",
"/opt/brave.com/brave/brave-browser",
"/snap/bin/brave",
"/usr/local/bin/brave-browser",
)
# Default profile ("User Data") directories per OS — same flat-list trick:
# only the one for the current OS exists, the rest never match.
_BRAVE_USER_DATA_CANDIDATES = (
r"%LocalAppData%\BraveSoftware\Brave-Browser\User Data", # Windows
"~/Library/Application Support/BraveSoftware/Brave-Browser", # macOS
"~/.config/BraveSoftware/Brave-Browser", # Linux
)
def _expand_path(candidate: str) -> str:
"""Expand %VAR% (Windows) and ~ (macOS/Linux) in a path candidate."""
return os.path.expanduser(os.path.expandvars(candidate))
def _find_brave() -> tuple[str | None, str | None]:
"""(exe_path, user_data_dir) for a default Brave install on Windows/macOS/Linux."""
exe = next(
(p for p in (_expand_path(c) for c in _BRAVE_EXE_CANDIDATES) if os.path.isfile(p)),
None,
)
ud = next(
(p for p in (_expand_path(c) for c in _BRAVE_USER_DATA_CANDIDATES) if os.path.isdir(p)),
None,
)
return exe, ud
def _cdp_cookie(cookie: dict):
"""DevTools cookie dict -> http.cookiejar.Cookie."""
import http.cookiejar
domain = cookie.get("domain") or ".youtube.com"
http_only = bool(cookie.get("httpOnly"))
return http.cookiejar.Cookie(
version=0, name=cookie["name"], value=cookie.get("value") or "",
port=None, port_specified=False,
domain=domain, domain_specified=True, domain_initial_dot=domain.startswith("."),
path=cookie.get("path") or "/", path_specified=True,
secure=bool(cookie.get("secure")),
expires=int(cookie.get("expires")) if cookie.get("expires") else None,
discard=False, comment=None, comment_url=None,
rest={"HttpOnly": None} if http_only else {},
)
def _extract_brave_cdp(timeout: float = 45.0) -> list:
"""Launch Brave headless on its REAL profile and read decrypted cookies
over DevTools.
Chromium 127+ encrypts new cookies "app-bound" (v20): only the browser
itself can decrypt them, which is why yt-dlp's file-based extraction
dies with "Failed to decrypt with DPAPI". Launching the browser with
its user-data-dir passed EXPLICITLY on the command line keeps remote
debugging allowed (Chromium 136+ blocks it for the implicit default
dir), and the browser hands its own cookies over in plaintext. The
browser must still be closed — the profile is single-writer.
"""
import json
import socket
import subprocess
import time
import urllib.request
exe, user_data = _find_brave()
if not exe or not user_data:
raise RuntimeError(
"Brave installation not found (looked in the default install "
"locations for Windows, macOS and Linux)"
)
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
port = s.getsockname()[1]
proc = subprocess.Popen(
[
exe, "--headless=new", f"--remote-debugging-port={port}",
f"--user-data-dir={user_data}", "--no-first-run",
"--no-default-browser-check", "about:blank",
],
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
)
try:
deadline = time.monotonic() + timeout
version = None
while time.monotonic() < deadline:
try:
version = json.load(urllib.request.urlopen(
f"http://127.0.0.1:{port}/json/version", timeout=2))
break
except Exception:
time.sleep(0.5)
if version is None:
# Most common cause: Brave is already running and the new
# process just delegated to it without opening a debug port.
raise BrowserCookieLockedError(
"could not attach to Brave — close Brave completely "
"(including background processes) and try again"
)
from websockets.sync.client import connect
with connect(version["webSocketDebuggerUrl"], open_timeout=10) as ws:
ws.send(json.dumps({"id": 1, "method": "Storage.getCookies"}))
reply = json.loads(ws.recv())
return [_cdp_cookie(c) for c in reply.get("result", {}).get("cookies", [])]
finally:
proc.kill()
def import_from_browser(
store: Store,
browser: str = "brave",
profile: str | None = None,
label: str | None = None,
cookie_dir: str | Path | None = None,
) -> str:
"""Pull youtube.com cookies straight from a local browser profile.
Tries yt-dlp's file-based extraction first; on Chromium 127+ profiles
whose cookies are app-bound (v20 — "Failed to decrypt with DPAPI"),
falls back to launching the browser itself headless and reading the
decrypted cookies over DevTools. Only youtube.com cookies are kept and
stored in the vault as a normal Netscape file, so activation, expiry
tracking and every extraction path work unchanged.
"""
from yt_dlp.cookies import extract_cookies_from_browser
cookies = None
try:
cookies = extract_cookies_from_browser(browser, profile or None)
except Exception as exc:
msg = str(exc)
if "Could not copy Chrome cookie database" in msg or "database is locked" in msg:
raise BrowserCookieLockedError(
f"could not read {browser}'s cookie database — close {browser} completely "
"(including background processes) and try again"
) from exc
# App-bound (v20) cookies: only the browser can decrypt them.
if browser == "brave" and ("decrypt with DPAPI" in msg or "decrypt" in msg.lower()):
log.info("Brave cookies are app-bound; falling back to DevTools extraction")
cookies = _extract_brave_cdp()
else:
raise
lines = []
for c in cookies:
if "youtube.com" not in (c.domain or ""):
continue
# Netscape format; the #HttpOnly_ prefix is stripped again on parse.
prefix = "#HttpOnly_" if c.has_nonstandard_attr("httponly") else ""
domain = c.domain or ".youtube.com"
include_subdomains = "TRUE" if domain.startswith(".") else "FALSE"
expiry = int(c.expires) if c.expires else 0
lines.append(
f"{prefix}{domain}\t{include_subdomains}\t{c.path or '/'}\t"
f"{'TRUE' if c.secure else 'FALSE'}\t{expiry}\t{c.name}\t{c.value}"
)
if not lines:
raise ValueError(f"no youtube.com cookies found in {browser} (not logged in?)")
text = (
"# Netscape HTTP Cookie File\n"
f"# Extracted from {browser} profile {profile or 'default'}\n"
+ "\n".join(lines) + "\n"
)
return import_text(store, text, label=label or f"{browser}", cookie_dir=cookie_dir)
def set_active(store: Store, cookie_id: str) -> None: def set_active(store: Store, cookie_id: str) -> None:
store.set_active_cookie(cookie_id) store.set_active_cookie(cookie_id)
@@ -143,12 +362,18 @@ def delete(store: Store, cookie_id: str, cookie_dir: str | Path | None = None) -
def resolve_active_path(store: Store, cookie_dir: str | Path | None = None) -> str | None: def resolve_active_path(store: Store, cookie_dir: str | Path | None = None) -> str | None:
row = store.get_active_cookie()
if not row:
return None
cdir = cookies_dir(cookie_dir) cdir = cookies_dir(cookie_dir)
row = store.get_active_cookie()
if row:
path = cdir / row.filename path = cdir / row.filename
return str(path) if path.exists() else None if path.exists():
return str(path)
for c in store.list_cookies():
path = cdir / c.filename
if path.exists():
store.set_active_cookie(c.id)
return str(path)
return None
def is_expired(row: CookieRow) -> bool: def is_expired(row: CookieRow) -> bool:
+11
View File
@@ -211,6 +211,17 @@ def discover_incremental(
size = min(size * 2, max_window) size = min(size * 2, max_window)
def extract_handle(url: str) -> str:
"""@handle embedded in a channel URL, or "" for /channel/<id> URLs.
Every caller that records a channel (CLI, webapp add-channel, jobs, watch)
normalises the handle the same way; this is that one shared definition.
"""
if "@" in url:
return "@" + url.split("@", 1)[1].split("/", 1)[0]
return ""
def _pick_channel_avatar(info: dict[str, Any]) -> str | None: def _pick_channel_avatar(info: dict[str, Any]) -> str | None:
"""Best-effort channel avatar URL from a yt-dlp channel info dict. """Best-effort channel avatar URL from a yt-dlp channel info dict.
+142
View File
@@ -1,7 +1,12 @@
from __future__ import annotations from __future__ import annotations
import json
import logging import logging
import os
import re
import urllib.request
from dataclasses import dataclass from dataclasses import dataclass
from pathlib import Path
from typing import Any, Mapping from typing import Any, Mapping
import yt_dlp import yt_dlp
@@ -100,8 +105,20 @@ def extract_video(
# One video extraction is two requests: the watch page and the InnerTube # One video extraction is two requests: the watch page and the InnerTube
# player call. yt-dlp spaces them itself; the pacer needs to know they exist. # player call. yt-dlp spaces them itself; the pacer needs to know they exist.
GLOBAL_PACER.wait(cost=2) GLOBAL_PACER.wait(cost=2)
try:
with yt_dlp.YoutubeDL(ydl_opts) as ydl: with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(video_url, download=False) info = ydl.extract_info(video_url, download=False)
except Exception:
# Logged-in sessions on current YouTube increasingly end here
# ("The page needs to be reloaded" / format-availability failures).
# The session itself is usually fine — the watch page still hands
# metadata and caption tracks to a plain cookie'd GET — so try that
# before giving up. Without cookies there is nothing to fall back to.
if cookies_file:
fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual)
if fallback is not None:
return fallback
raise
pick = pick_subtitle(info, languages_dict, prefer_manual) pick = pick_subtitle(info, languages_dict, prefer_manual)
segments: list[Segment] = [] segments: list[Segment] = []
@@ -128,6 +145,131 @@ def extract_video(
) )
_WATCH_PAGE_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36"
)
# Precompiled: the fallback can fire on every video of a throttled batch, and
# re-compiling per call showed up under those runs.
_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;")
# One opener per (cookie file, mtime): the jar parse is per-call work that is
# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under
# the same path still gets a fresh jar; only the newest entry is kept.
_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {}
def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0):
path = str(Path(cookies_file).resolve())
try:
mtime = os.path.getmtime(path)
except OSError:
mtime = -1.0
key = (path, mtime)
opener = _OPENER_CACHE.get(key)
if opener is None:
jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file)
jar.load(ignore_discard=True, ignore_expires=True)
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
opener.addheaders = [
("User-Agent", _WATCH_PAGE_UA),
("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"),
]
_OPENER_CACHE.clear()
_OPENER_CACHE[key] = opener
return opener.open(url, timeout=timeout)
def extract_via_watch_page(
video_url: str,
cookies_file: str,
languages: Mapping[str, str],
prefer_manual: bool = True,
) -> VideoData | None:
"""Session-cookie fallback: scrape the watch page directly.
yt-dlp's InnerTube clients reject logged-in sessions that lack a PO
token (playability "The page needs to be reloaded") or return no
formats/captions, which kills cookie-authenticated videos — members
being the case this exists for. The plain watch page served to the
logged-in browser still carries `ytInitialPlayerResponse` with
metadata and caption tracks, so GET it with the vault cookie and
reuse the normal subtitle picker. Returns None when the page holds
no caption tracks at all, so callers keep their own error semantics.
"""
GLOBAL_PACER.wait()
html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace")
m = _INITIAL_PLAYER_RE.search(html)
if not m:
log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url)
return None
try:
pr = json.loads(m.group(1))
except json.JSONDecodeError:
log.warning("watch-page fallback: unparseable player response for %s", video_url)
return None
status = (pr.get("playabilityStatus") or {}).get("status")
if status != "OK":
reason = (pr.get("playabilityStatus") or {}).get("reason") or status
raise RuntimeError(f"watch-page fallback: video not playable ({reason})")
details = pr.get("videoDetails") or {}
micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {}
tracks = (
(pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {}
).get("captionTracks") or []
if not tracks:
return None
# Reuse pick_subtitle by shaping the tracks as an info dict.
info: dict[str, Any] = {
"title": details.get("title"),
"channel": details.get("author"),
"duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None,
"view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None,
"description": details.get("shortDescription") or "",
"tags": details.get("keywords") or [],
"thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"),
"upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None,
# The watch page carries no availability signal; leaving it unset
# keeps the (more informed) discovery value in the store.
"availability": None,
"subtitles": {},
"automatic_captions": {},
}
for t in tracks:
base = t.get("baseUrl") or ""
if not base:
continue
entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}]
if t.get("kind") == "asr":
info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry)
else:
info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry)
pick = pick_subtitle(info, languages, prefer_manual)
segments: list[Segment] = []
skip_reason: str | None = None
if pick:
try:
GLOBAL_PACER.wait()
raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace")
segments = parse_auto_dump(raw)
if not segments:
skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)"
except Exception as exc: # pylint: disable=broad-except
skip_reason = f"caption download failed via watch-page fallback: {exc}"
else:
skip_reason = describe_missing_subtitle(info, languages)
return VideoData(
info=info, segments=segments, subtitle=pick,
has_chapters=bool(info.get("chapters")), skip_reason=skip_reason,
)
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str: def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'. """Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
+4 -24
View File
@@ -4,9 +4,9 @@ import logging
import time import time
from typing import Callable from typing import Callable
from .config import Config from .config import Config, keep_ref
from .cookies import resolve_active_path from .cookies import resolve_active_path
from .discover import discover_incremental from .discover import discover_incremental, extract_handle
from .pipeline import process_video from .pipeline import process_video
from .ratelimit import ThrottleGuard, polite_sleep from .ratelimit import ThrottleGuard, polite_sleep
from .store import Store from .store import Store
@@ -77,13 +77,13 @@ def _run_once(
max_window=cfg.sync.max_window, max_window=cfg.sync.max_window,
overlap=cfg.sync.overlap, overlap=cfg.sync.overlap,
since=store.latest_upload_date(channel_id) if known else None, since=store.latest_upload_date(channel_id) if known else None,
keep=_keep_ref(cfg), keep=keep_ref(cfg),
) )
channel_name, refs = result.channel_name, result.refs channel_name, refs = result.channel_name, result.refs
target_channel = channel_id or result.channel_id target_channel = channel_id or result.channel_id
if result.full_scan: if result.full_scan:
store.upsert_channel( store.upsert_channel(
target_channel, _extract_handle(cfg.channel_url), channel_name, len(refs), avatar=result.avatar target_channel, extract_handle(cfg.channel_url), channel_name, len(refs), avatar=result.avatar
) )
else: else:
store.update_channel_meta(target_channel, name=channel_name, avatar=result.avatar) store.update_channel_meta(target_channel, name=channel_name, avatar=result.avatar)
@@ -133,23 +133,3 @@ def _run_once(
_emit(f"watch: throttled, waiting {wait:.1f}s") _emit(f"watch: throttled, waiting {wait:.1f}s")
time.sleep(wait) time.sleep(wait)
polite_sleep(cfg.delay.min_seconds, cfg.delay.max_seconds) polite_sleep(cfg.delay.min_seconds, cfg.delay.max_seconds)
def _keep_ref(cfg: Config):
"""Same shorts/live rules the store was populated under — see jobs._keep_ref."""
def keep(r) -> bool:
url = r.url or ""
if not cfg.include_shorts and "/shorts/" in url:
return False
if not cfg.include_live and url.startswith("https://www.youtube.com/live/"):
return False
return True
return keep
def _extract_handle(url: str) -> str:
if "@" in url:
return "@" + url.split("@", 1)[1].split("/", 1)[0]
return ""
+8 -10
View File
@@ -9,7 +9,7 @@ from .chapters import align_chapters, chapters_from_info
from .config import Config from .config import Config
from .extract import extract_video from .extract import extract_video
from .ratelimit import is_rate_limited from .ratelimit import is_rate_limited
from .render import build_filename_stem, render_markdown from .render import build_filename_stem, render_markdown, safe_dirname
from .store import Store, VideoRow from .store import Store, VideoRow
from ._yt_http import yt_get from ._yt_http import yt_get
@@ -44,7 +44,7 @@ def _render_and_retire(
template=cfg.filename_template, template=cfg.filename_template,
video_id=video_id, video_id=video_id,
) )
out_subdir = Path(cfg.output_dir_resolved) / _safe_dirname(channel_name) out_subdir = Path(cfg.output_dir_resolved) / safe_dirname(channel_name)
md_path = render_markdown(cfg.template_path_resolved, out_subdir, stem, context) md_path = render_markdown(cfg.template_path_resolved, out_subdir, stem, context)
out_root = Path(cfg.output_dir_resolved).parent out_root = Path(cfg.output_dir_resolved).parent
@@ -122,13 +122,15 @@ def process_video(
return "error" return "error"
info = data.info info = data.info
store.set_availability(row.video_id, info.get("availability"))
# Metadata is persisted BEFORE the no-transcript exit. The extraction already # Metadata is persisted BEFORE the no-transcript exit. The extraction already
# cost its requests and the info dict is in hand; discarding it because the # cost its requests and the info dict is in hand; discarding it because the
# separate caption fetch failed means a retry re-spends them for data we # separate caption fetch failed means a retry re-spends them for data we
# already had. Measured after a throttling incident: five rows left with # already had. Measured after a throttling incident: five rows left with
# upload_date, view_count, description and thumbnail all NULL. # upload_date, view_count, description and thumbnail all NULL.
# Agrupadas en una transaccion: una conexion/commit en vez de tres.
with store.transaction():
store.set_availability(row.video_id, info.get("availability"))
_store_metadata(store, row, info) _store_metadata(store, row, info)
if not data.segments: if not data.segments:
@@ -148,7 +150,9 @@ def process_video(
chapters = chapters_from_info(data.info) chapters = chapters_from_info(data.info)
sections = align_chapters(data.segments, chapters) sections = align_chapters(data.segments, chapters)
# persist segments + rich metadata to DB (for search, stats, webapp) # persist segments + rich metadata to DB (for search, stats, webapp);
# una transaccion: delete+inserts+update atomicos y un solo commit
with store.transaction():
store.store_segments(row.video_id, data.segments) store.store_segments(row.video_id, data.segments)
seg_json = json.dumps( seg_json = json.dumps(
[{"start": s.start, "end": s.end, "text": s.text} for s in data.segments], [{"start": s.start, "end": s.end, "text": s.text} for s in data.segments],
@@ -205,11 +209,6 @@ def _normalize_date(d: str | None) -> str:
return d return d
def _safe_dirname(name: str) -> str:
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
return safe.strip().strip(".") or "unknown"
def thumbnail_url_for(video_row) -> str: def thumbnail_url_for(video_row) -> str:
"""Thumbnail URL for a video: stored URL, else the canonical YouTube one derived from its id.""" """Thumbnail URL for a video: stored URL, else the canonical YouTube one derived from its id."""
if video_row and getattr(video_row, "thumbnail", None): if video_row and getattr(video_row, "thumbnail", None):
@@ -259,7 +258,6 @@ def re_render_videos(store: Store, cfg: Config, channel_id: str | None = None) -
import json import json
from .chapters import align_chapters, Chapter from .chapters import align_chapters, Chapter
from .parse import Segment from .parse import Segment
from .render import build_filename_stem, render_markdown
videos = [v for v in store.get_all(channel_id) if v.status == "done" and v.segments_json] videos = [v for v in store.get_all(channel_id) if v.status == "done" and v.segments_json]
n = 0 n = 0
+2
View File
@@ -191,6 +191,7 @@ def ydl_throttle_opts(
*, *,
extractor_retries: int = 3, extractor_retries: int = 3,
socket_timeout: float = 30.0, socket_timeout: float = 30.0,
js_runtimes: dict[str, dict] | None = None,
) -> dict[str, object]: ) -> dict[str, object]:
"""The politeness half of every `ydl_opts` dict in this project. """The politeness half of every `ydl_opts` dict in this project.
@@ -213,6 +214,7 @@ def ydl_throttle_opts(
# this only covers transient 5xx and network errors. # this only covers transient 5xx and network errors.
"extractor_retries": int(extractor_retries), "extractor_retries": int(extractor_retries),
"socket_timeout": float(socket_timeout), "socket_timeout": float(socket_timeout),
"js_runtimes": js_runtimes if js_runtimes is not None else {"node": {}, "deno": {}, "bun": {}, "quickjs": {}},
} }
+28
View File
@@ -32,7 +32,16 @@ def to_json(value) -> str:
return json.dumps(value, ensure_ascii=False) return json.dumps(value, ensure_ascii=False)
# One Environment per template directory, cached for the process lifetime:
# building it (loader + filters) per rendered note was the dominant cost of
# batch renders. Jinja's own per-env template cache keeps the compiled
# Template, and its default auto_reload still picks up on-disk edits.
_ENV_CACHE: dict[Path, Environment] = {}
def _make_env(template_dir: Path) -> Environment: def _make_env(template_dir: Path) -> Environment:
env = _ENV_CACHE.get(template_dir)
if env is None:
env = Environment( env = Environment(
loader=FileSystemLoader(str(template_dir)), loader=FileSystemLoader(str(template_dir)),
autoescape=select_autoescape(disabled_extensions=("j2", "txt")), autoescape=select_autoescape(disabled_extensions=("j2", "txt")),
@@ -42,6 +51,7 @@ def _make_env(template_dir: Path) -> Environment:
env.filters["format_timestamp"] = format_timestamp env.filters["format_timestamp"] = format_timestamp
env.filters["quote_yaml"] = quote_yaml env.filters["quote_yaml"] = quote_yaml
env.filters["to_json"] = to_json env.filters["to_json"] = to_json
_ENV_CACHE[template_dir] = env
return env return env
@@ -63,6 +73,24 @@ def render_markdown(
return out_file return out_file
def safe_dirname(name: str | None) -> str:
"""Channel name -> markdown subdirectory name, shared by every .md writer.
Strips the characters Windows forbids in a path segment, then leading and
trailing spaces/dots (also illegal there), falling back to "unknown" so the
output tree never grows a nameless root. Callers with a bare filename want
`safe_filename` instead ("untitled" fallback).
"""
safe = "".join(c for c in (name or "") if c not in r'\/:*?"<>|')
return safe.strip().strip(".") or "unknown"
def safe_filename(name: str) -> str:
"""Free-text (title) -> file-legal stem; "untitled" when nothing survives."""
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
return safe.strip().strip(".") or "untitled"
def build_filename_stem( def build_filename_stem(
upload_date: str | None, upload_date: str | None,
title: str, title: str,
+1 -9
View File
@@ -8,7 +8,7 @@ from pathlib import Path
from typing import Callable from typing import Callable
from .parse import Segment from .parse import Segment
from .store import SearchHit, Store from .store import Store
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
@@ -91,14 +91,6 @@ def _strip_quotes(value: str) -> str:
return v return v
def store_segments(store: Store, video_id: str, segments: list[Segment]) -> None:
store.store_segments(video_id, segments)
def search(store: Store, query: str, channel_id: str | None = None, limit: int = 50) -> list[SearchHit]:
return store.search_segments(query, channel_id=channel_id, limit=limit)
def backfill_from_markdown( def backfill_from_markdown(
store: Store, store: Store,
md_root: Path, md_root: Path,
+107 -30
View File
@@ -2,6 +2,7 @@ from __future__ import annotations
import json import json
import sqlite3 import sqlite3
import threading
from contextlib import contextmanager from contextlib import contextmanager
from dataclasses import dataclass from dataclasses import dataclass
from datetime import datetime, timezone from datetime import datetime, timezone
@@ -393,10 +394,30 @@ class JobRow:
last_error: str | None last_error: str | None
def order_pending(rows: list[VideoRow], refs: list[VideoRef]) -> list[VideoRow]:
"""Newest-window-first ordering for the pending queue.
`get_pending` is ordered by discovery time, which used to coincide with
newest-first because discovery saw the whole channel at once. With windowed
sync that no longer holds, so put the videos from this run's window (the
`refs` discovery just returned, newest first) at the front and keep the rest
of the backlog behind them. Shared by the CLI scrape and the webapp's
channel jobs, which must agree on what `--limit` / `limit` mean.
"""
by_id = {r.video_id: r for r in rows}
ordered = [by_id.pop(x.video_id) for x in refs if x.video_id in by_id]
ordered.extend(by_id.values())
return ordered
class Store: class Store:
def __init__(self, db_path: str | Path): def __init__(self, db_path: str | Path):
self.db_path = Path(db_path) self.db_path = Path(db_path)
self.db_path.parent.mkdir(parents=True, exist_ok=True) self.db_path.parent.mkdir(parents=True, exist_ok=True)
# Pila de transacciones ambientales, por hilo: permite agrupar varias
# llamadas a Store en una sola conexion/commit sin pasar la conexion
# como parametro a cada metodo.
self._local = threading.local()
self._init_schema() self._init_schema()
def _connect(self) -> sqlite3.Connection: def _connect(self) -> sqlite3.Connection:
@@ -419,13 +440,53 @@ class Store:
if col not in ch_existing: if col not in ch_existing:
conn.execute(f"ALTER TABLE channels ADD COLUMN {col} {coltype}") conn.execute(f"ALTER TABLE channels ADD COLUMN {col} {coltype}")
conn.executescript(_EXTRA_SCHEMA) conn.executescript(_EXTRA_SCHEMA)
# Unconditional, not "only when the column was just added": a row can # Unconditional in spirit, not in cost: a row can also arrive
# also arrive unranked afterwards, and an unranked row is displayed # unranked afterwards, and an unranked row is displayed in the
# in the wrong place rather than merely in an arbitrary one. # wrong place rather than merely in an arbitrary one. But
# _rank_unranked scans the whole videos table, and Store is built
# on every webapp import — so gate it behind the same predicate it
# matches on, which is a single indexed probe.
if conn.execute(
"SELECT 1 FROM videos WHERE channel_seq IS NULL LIMIT 1"
).fetchone() is not None:
_rank_unranked(conn) _rank_unranked(conn)
@contextmanager
def transaction(self):
"""Agrupa varias escrituras de Store en una sola conexion y un commit.
`process_video` costaba ~6 connect/commit por video (un fsync cada
uno en WAL). Dentro de este bloque, toda llamada a Store hecha desde
el MISMO hilo se une a la conexion ambiental; un error revierte el
grupo entero, que es la atomicidad por video que se quiere de todos
modos.
"""
conn = self._connect()
stack: list[sqlite3.Connection] = getattr(self._local, "stack", None) or []
self._local.stack = stack
stack.append(conn)
try:
yield conn
conn.commit()
except BaseException:
conn.rollback()
raise
finally:
stack.remove(conn)
conn.close()
@contextmanager @contextmanager
def _cursor(self) -> Iterator[sqlite3.Cursor]: def _cursor(self) -> Iterator[sqlite3.Cursor]:
# Dentro de transaction(): misma conexion y sin commit intermedio —
# el bloque externo decide cuando el trabajo se vuelve durable.
stack: list[sqlite3.Connection] = getattr(self._local, "stack", None) or []
if stack:
cur = stack[-1].cursor()
try:
yield cur
finally:
cur.close()
return
conn = self._connect() conn = self._connect()
try: try:
yield conn.cursor() yield conn.cursor()
@@ -576,8 +637,10 @@ class Store:
for rank, (_, r) in enumerate(ordered): for rank, (_, r) in enumerate(ordered):
seqs[r.video_id] = base + width - rank seqs[r.video_id] = base + width - rank
for r in refs: # One executemany instead of a prepared-statement round per ref:
cur.execute( # discovery hands over hundreds of rows at once and the statement
# text is identical for all of them.
cur.executemany(
"""INSERT INTO videos """INSERT INTO videos
(video_id, channel_id, title, url, upload_date, (video_id, channel_id, title, url, upload_date,
upload_date_approx, duration, availability, upload_date_approx, duration, availability,
@@ -600,10 +663,14 @@ class Store:
duration = COALESCE(excluded.duration, videos.duration), duration = COALESCE(excluded.duration, videos.duration),
availability = COALESCE(excluded.availability, videos.availability), availability = COALESCE(excluded.availability, videos.availability),
channel_seq = COALESCE(excluded.channel_seq, videos.channel_seq)""", channel_seq = COALESCE(excluded.channel_seq, videos.channel_seq)""",
[
(r.video_id, r.channel_id, r.title, r.url, r.upload_date, (r.video_id, r.channel_id, r.title, r.url, r.upload_date,
int(r.date_approx or 0), r.duration, int(r.date_approx or 0), r.duration, r.availability,
r.availability, seqs.get(r.video_id), now), seqs.get(r.video_id), now)
for r in refs
],
) )
for r in refs:
if r.video_id not in existing: if r.video_id not in existing:
inserted += 1 inserted += 1
existing.add(r.video_id) existing.add(r.video_id)
@@ -1038,34 +1105,44 @@ class Store:
def dashboard(self) -> dict: def dashboard(self) -> dict:
with self._cursor() as cur: with self._cursor() as cur:
channels = [dict(r) for r in cur.execute("SELECT * FROM channels ORDER BY name").fetchall()] channels = [dict(r) for r in cur.execute("SELECT * FROM channels ORDER BY name").fetchall()]
# One GROUP BY instead of one aggregate query per channel (N+1).
# Channels with zero videos are absent from the grouping and get
# the same zeros/NULLs the per-channel query used to return.
agg = {
r["channel_id"]: r
for r in cur.execute(
"SELECT channel_id, COUNT(*) AS n, COALESCE(SUM(duration),0) AS dur, "
"MIN(upload_date) AS mind, MAX(upload_date) AS maxd, "
"COALESCE(SUM(view_count),0) AS views, COALESCE(SUM(like_count),0) AS likes "
"FROM videos GROUP BY channel_id"
).fetchall()
}
for ch in channels: for ch in channels:
cid = ch["channel_id"] row = agg.get(ch["channel_id"])
row = cur.execute( ch["video_count_db"] = row["n"] if row else 0
"SELECT COUNT(*) AS n, COALESCE(SUM(duration),0) AS dur, MIN(upload_date) AS mind, MAX(upload_date) AS maxd, COALESCE(SUM(view_count),0) AS views, COALESCE(SUM(like_count),0) AS likes FROM videos WHERE channel_id = ?", ch["total_duration"] = row["dur"] if row else 0
(cid,), ch["date_min"] = row["mind"] if row else None
).fetchone() ch["date_max"] = row["maxd"] if row else None
ch["video_count_db"] = row["n"] ch["total_views"] = row["views"] if row else 0
ch["total_duration"] = row["dur"] ch["total_likes"] = row["likes"] if row else 0
ch["date_min"] = row["mind"]
ch["date_max"] = row["maxd"]
ch["total_views"] = row["views"]
ch["total_likes"] = row["likes"]
status_breakdown = { status_breakdown = {
r["status"]: r["n"] r["status"]: r["n"]
for r in cur.execute("SELECT status, COUNT(*) AS n FROM videos GROUP BY status").fetchall() for r in cur.execute("SELECT status, COUNT(*) AS n FROM videos GROUP BY status").fetchall()
} }
# top tags (tags is JSON array text) # top tags (tags is JSON array text). Counted in SQL via json_each
tag_rows = cur.execute("SELECT tags FROM videos WHERE tags IS NOT NULL AND tags != '[]'").fetchall() # instead of loading every tags string into Python; json_valid
tag_counts: dict[str, int] = {} # skips rows the old json.loads/except path also skipped, and the
for tr in tag_rows: # TRIM/LOWER/empty-filter mirrors the per-tag normalisation.
try: tag_counts = {
for t in json.loads(tr["tags"]): r["tag"]: r["n"]
t = (t or "").strip().lower() for r in cur.execute(
if t: "SELECT LOWER(TRIM(je.value)) AS tag, COUNT(*) AS n "
tag_counts[t] = tag_counts.get(t, 0) + 1 "FROM videos v, json_each(v.tags) je "
except (json.JSONDecodeError, TypeError): "WHERE json_valid(v.tags) AND TRIM(je.value) <> '' "
continue "GROUP BY tag ORDER BY n DESC LIMIT 20"
top_tags = sorted(tag_counts.items(), key=lambda x: x[1], reverse=True)[:20] ).fetchall()
}
top_tags = sorted(tag_counts.items(), key=lambda x: x[1], reverse=True)
return { return {
"channels": channels, "channels": channels,
"status_breakdown": status_breakdown, "status_breakdown": status_breakdown,
+67 -23
View File
@@ -1,6 +1,7 @@
from __future__ import annotations from __future__ import annotations
import json import json
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
@@ -12,7 +13,9 @@ from .. import analysis as analysis_mod
from .. import cookies as cookies_mod from .. import cookies as cookies_mod
from .. import export as export_mod from .. import export as export_mod
from ..config import Config from ..config import Config
from ..discover import extract_handle
from ..ratelimit import polite_sleep from ..ratelimit import polite_sleep
from ..render import safe_dirname
from .. import store as store_mod from .. import store as store_mod
from ..store import Store from ..store import Store
@@ -21,6 +24,11 @@ from ..store import Store
#: so this only bounds the eager burst that follows "Add channel". #: so this only bounds the eager burst that follows "Add channel".
THUMBNAIL_AUTO_LIMIT = 60 THUMBNAIL_AUTO_LIMIT = 60
#: Thumbnail fetches go to the i.ytimg.com CDN (not youtube.com), so the
#: Pacer does not apply to them; a small pool turns ~60 sequential round
#: trips into ~10 batches without hammering the CDN.
THUMBNAIL_WORKERS = 6
def build_router(store: Store, cfg: Config, jobs) -> APIRouter: def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
r = APIRouter(prefix="/api") r = APIRouter(prefix="/api")
@@ -50,7 +58,7 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
avatar_cached = False avatar_cached = False
if not avatar: if not avatar:
avatar = deep_channel_avatar(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests) avatar = deep_channel_avatar(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
store.upsert_channel(cid, _handle(url), name, len(refs), avatar=avatar) store.upsert_channel(cid, extract_handle(url), name, len(refs), avatar=avatar)
store.upsert_videos(refs) store.upsert_videos(refs)
if avatar: if avatar:
avatars_dir = Path(cfg.output_dir_resolved).parent / "avatars" avatars_dir = Path(cfg.output_dir_resolved).parent / "avatars"
@@ -229,6 +237,31 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
cookies_mod.set_active(store, imported[0]) cookies_mod.set_active(store, imported[0])
return {"imported": imported} return {"imported": imported}
@r.post("/cookies/from-browser")
def cookies_from_browser(payload: dict):
"""Extract youtube.com cookies from a local browser (default: Brave).
Requires the browser to be fully closed; Chromium keeps its cookie
database locked while running.
"""
payload = payload or {}
browser = (payload.get("browser") or "brave").strip().lower()
profile = payload.get("profile") or None
try:
cid = cookies_mod.import_from_browser(store, browser=browser, profile=profile)
except cookies_mod.BrowserCookieLockedError as exc:
raise HTTPException(409, str(exc))
except Exception as exc:
raise HTTPException(400, str(exc))
# The user imported it precisely to use it.
cookies_mod.set_active(store, cid)
row = store.get_cookie(cid)
return {
"imported": [cid], "active": cid, "browser": browser,
"cookie_count": row.cookie_count if row else None,
"has_session": row.has_session if row else None,
}
@r.post("/cookies/{cookie_id}/activate") @r.post("/cookies/{cookie_id}/activate")
def cookies_activate(cookie_id: str): def cookies_activate(cookie_id: str):
cookies_mod.set_active(store, cookie_id) cookies_mod.set_active(store, cookie_id)
@@ -289,8 +322,11 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
"database": _folder_info(Path(cfg.database_path_resolved).parent), "database": _folder_info(Path(cfg.database_path_resolved).parent),
}} }}
# `def`, not `async def`: same reason as add_channel — the body blocks on
# SQLite reads and a subprocess/os.startfile call, which would hold the
# event loop if this were async.
@r.post("/folders/open") @r.post("/folders/open")
async def open_folder(payload: dict): def open_folder(payload: dict):
kind = (payload or {}).get("kind") kind = (payload or {}).get("kind")
channel_id = (payload or {}).get("channel_id") channel_id = (payload or {}).get("channel_id")
data_root = Path(cfg.output_dir_resolved).parent data_root = Path(cfg.output_dir_resolved).parent
@@ -299,7 +335,7 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
if channel_id: if channel_id:
ch = store.get_channel(channel_id) or {} ch = store.get_channel(channel_id) or {}
if ch.get("name"): if ch.get("name"):
target = Path(cfg.output_dir_resolved) / _safe_dir(ch["name"]) target = Path(cfg.output_dir_resolved) / safe_dirname(ch["name"])
elif kind == "exports": elif kind == "exports":
target = data_root / "exports" target = data_root / "exports"
elif kind == "audio": elif kind == "audio":
@@ -375,6 +411,17 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
raise HTTPException(404, "markdown file missing on disk") raise HTTPException(404, "markdown file missing on disk")
return FileResponse(str(p), filename=p.name, media_type="text/markdown") return FileResponse(str(p), filename=p.name, media_type="text/markdown")
@r.post("/videos/{video_id}/open-markdown")
def open_video_markdown(video_id: str):
v = store.get_video(video_id)
if not v or not v.markdown_path:
raise HTTPException(404, "markdown not generated yet")
p = Path(cfg.output_dir_resolved).parent / v.markdown_path
if not p.exists():
raise HTTPException(404, "markdown file missing on disk")
_open_in_os(p)
return {"opened": str(p)}
@r.api_route("/videos/{video_id}/audio", methods=["GET", "HEAD"]) @r.api_route("/videos/{video_id}/audio", methods=["GET", "HEAD"])
def video_audio(video_id: str, request: Request): def video_audio(video_id: str, request: Request):
# audio is stored as data/audio/<video_id>.mp3 (see jobs._run_audio outtmpl) # audio is stored as data/audio/<video_id>.mp3 (see jobs._run_audio outtmpl)
@@ -423,10 +470,13 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
video_ids = video_ids[:limit] video_ids = video_ids[:limit]
out_dir = Path(cfg.output_dir_resolved).parent / "thumbnails" out_dir = Path(cfg.output_dir_resolved).parent / "thumbnails"
out_dir.mkdir(parents=True, exist_ok=True) out_dir.mkdir(parents=True, exist_ok=True)
n = 0 # CDN fetches, one per video, each independent: run them on a small
for vid in video_ids: # thread pool instead of sequentially. cache_thumbnail owns its own
if cache_thumbnail(store, vid, out_dir): # SQLite connection and writes one file per id, so this is safe;
n += 1 # failures still count as False exactly as before.
with ThreadPoolExecutor(max_workers=THUMBNAIL_WORKERS) as pool:
results = list(pool.map(lambda vid: cache_thumbnail(store, vid, out_dir), video_ids))
n = sum(1 for ok in results if ok)
return {"downloaded": n, "dir": str(out_dir), "skipped": skipped} return {"downloaded": n, "dir": str(out_dir), "skipped": skipped}
@r.get("/thumbnails/{video_id}") @r.get("/thumbnails/{video_id}")
@@ -455,8 +505,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
"permanent": counts.get("permanent", 0), "permanent": counts.get("permanent", 0),
} }
# `def`, not `async def`: the UPDATE below blocks on SQLite; run it on the
# threadpool like the other blocking handlers.
@r.post("/videos/reset") @r.post("/videos/reset")
async def reset_videos(payload: dict): def reset_videos(payload: dict):
"""Send `error` / `no_subtitles` videos back to pending so they can be retried. """Send `error` / `no_subtitles` videos back to pending so they can be retried.
`no_subtitles` is resettable on purpose: it is recorded whenever the `no_subtitles` is resettable on purpose: it is recorded whenever the
@@ -620,8 +672,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
"top_tags": d["top_tags"], "totals": totals} "top_tags": d["top_tags"], "totals": totals}
# -------------------------------------------------- audio (job) # -------------------------------------------------- audio (job)
# `def`, not `async def`: no await here, and the SQLite reads below would
# block the event loop.
@r.post("/tools/audio") @r.post("/tools/audio")
async def tools_audio(payload: dict): def tools_audio(payload: dict):
video_ids = (payload or {}).get("video_ids") or [] video_ids = (payload or {}).get("video_ids") or []
channel_id = (payload or {}).get("channel_id") channel_id = (payload or {}).get("channel_id")
if not video_ids and not channel_id: if not video_ids and not channel_id:
@@ -632,8 +686,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
job_id = jobs.enqueue(channel_id, opts) job_id = jobs.enqueue(channel_id, opts)
return {"job_id": job_id} return {"job_id": job_id}
# `def`, not `async def`: same as tools_audio — SQLite reads + filesystem
# stats only, no await.
@r.post("/tools/video") @r.post("/tools/video")
async def tools_video(payload: dict): def tools_video(payload: dict):
video_ids = (payload or {}).get("video_ids") or [] video_ids = (payload or {}).get("video_ids") or []
if len(video_ids) != 1: if len(video_ids) != 1:
raise HTTPException(400, "exactly one video_id required") raise HTTPException(400, "exactly one video_id required")
@@ -652,11 +708,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
def _video_dict(v) -> dict: def _video_dict(v) -> dict:
import json as _json
tags = [] tags = []
if v.tags: if v.tags:
try: try:
tags = _json.loads(v.tags) tags = json.loads(v.tags)
except Exception: except Exception:
tags = [] tags = []
return { return {
@@ -710,22 +765,11 @@ def _loads(s):
return {} return {}
def _handle(url: str) -> str:
if "@" in url:
return "@" + url.split("@", 1)[1].split("/", 1)[0]
return ""
def _folder_info(path: Path) -> dict: def _folder_info(path: Path) -> dict:
path = Path(path) path = Path(path)
return {"path": str(path.resolve()), "exists": path.exists()} return {"path": str(path.resolve()), "exists": path.exists()}
def _safe_dir(name: str) -> str:
safe = "".join(c for c in (name or "") if c not in r'\/:*?"<>|')
return (safe.strip().strip(".") or "unknown")
def _open_in_os(path: Path) -> None: def _open_in_os(path: Path) -> None:
import os import os
import sys import sys
+28 -1
View File
@@ -1,6 +1,7 @@
from __future__ import annotations from __future__ import annotations
import logging import logging
import threading
from pathlib import Path from pathlib import Path
from fastapi import FastAPI from fastapi import FastAPI
@@ -47,18 +48,44 @@ def create_app(cfg: Config | None = None, db_path: str | Path | None = None) ->
# Reconcile, not just backfill: backfill_from_markdown populates segments # Reconcile, not just backfill: backfill_from_markdown populates segments
# and metadata but never touches `status`/`markdown_path`, so a video whose # and metadata but never touches `status`/`markdown_path`, so a video whose
# .md is already on disk would keep showing a failure after every restart. # .md is already on disk would keep showing a failure after every restart.
#
# It walks and parses EVERY .md in the tree, which on a large library
# blocked uvicorn boot for the whole scan — so it runs on a daemon thread
# and the server answers immediately (SQLite/WAL absorbs the concurrent
# writes; /api/tools/reconcile still re-runs it on demand). healthz
# reports whether the initial pass has finished.
reconcile_done = threading.Event()
def _startup_reconcile() -> None:
try: try:
md_root = Path(cfg.output_dir_resolved) md_root = Path(cfg.output_dir_resolved)
if md_root.exists(): if md_root.exists():
reconcile_markdown(store, md_root, log=pkg_log.info) reconcile_markdown(store, md_root, log=pkg_log.info)
except Exception: except Exception:
pkg_log.exception("startup reconcile failed") pkg_log.exception("startup reconcile failed")
finally:
reconcile_done.set()
threading.Thread(target=_startup_reconcile, name="startup-reconcile", daemon=True).start()
app = FastAPI(title="yt-scraper platform", version="1.0.0") app = FastAPI(title="yt-scraper platform", version="1.0.0")
# StaticFiles/FileResponse send etag + last-modified but no Cache-Control,
# so a browser may heuristically cache an old app.js across updates and
# run yesterday's JS against a new backend. "no-cache" forces
# revalidation on every load (cheap 304s on a local app) while keeping
# the etag benefits.
@app.middleware("http")
async def _revalidate_shell(request, call_next):
response = await call_next(request)
path = request.url.path
if path == "/" or path.startswith("/static/"):
response.headers["Cache-Control"] = "no-cache"
return response
@app.get("/healthz") @app.get("/healthz")
def healthz(): def healthz():
return {"status": "ok"} return {"status": "ok", "reconcile_done": reconcile_done.is_set()}
jobs = JobManager(store, cfg) jobs = JobManager(store, cfg)
app.state.jobs = jobs app.state.jobs = jobs
+10 -50
View File
@@ -9,12 +9,12 @@ from collections import deque
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
from ..config import Config, load_config, parse_languages from ..config import Config, keep_ref, load_config, parse_languages
from ..cookies import resolve_active_path from ..cookies import resolve_active_path
from ..discover import discover_incremental from ..discover import discover_incremental, extract_handle
from ..pipeline import process_video from ..pipeline import process_video
from ..ratelimit import ThrottleGuard, polite_sleep from ..ratelimit import ThrottleGuard, polite_sleep
from ..store import Store, VideoRef from ..store import Store, order_pending
class JobManager: class JobManager:
@@ -269,6 +269,7 @@ class JobManager:
"sleep_interval_requests": cfg.yt_dlp.sleep_subrequests, "sleep_interval_requests": cfg.yt_dlp.sleep_subrequests,
"retries": cfg.yt_dlp.retries, "retries": cfg.yt_dlp.retries,
"socket_timeout": 30.0, "socket_timeout": 30.0,
"js_runtimes": {"node": {}, "deno": {}, "bun": {}, "quickjs": {}},
} }
if cfg.delay.audio_rate_limit: if cfg.delay.audio_rate_limit:
ydl_opts["ratelimit"] = cfg.delay.audio_rate_limit ydl_opts["ratelimit"] = cfg.delay.audio_rate_limit
@@ -453,7 +454,7 @@ class JobManager:
max_window=sync.max_window, max_window=sync.max_window,
overlap=sync.overlap, overlap=sync.overlap,
since=self.store.latest_upload_date(channel_id) if incremental else None, since=self.store.latest_upload_date(channel_id) if incremental else None,
keep=_keep_ref(cfg, include_shorts, no_live), keep=keep_ref(cfg, include_shorts, no_live),
) )
except Exception as exc: except Exception as exc:
self.store.update_job(job_id, status="error", last_error=str(exc), finished=True) self.store.update_job(job_id, status="error", last_error=str(exc), finished=True)
@@ -478,7 +479,7 @@ class JobManager:
from ..discover import deep_channel_avatar from ..discover import deep_channel_avatar
avatar = deep_channel_avatar(channel_url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests) avatar = deep_channel_avatar(channel_url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
if result.full_scan: if result.full_scan:
self.store.upsert_channel(_channel_id, _handle(channel_url), channel_name, len(refs), avatar=avatar) self.store.upsert_channel(_channel_id, extract_handle(channel_url), channel_name, len(refs), avatar=avatar)
else: else:
self.store.update_channel_meta(_channel_id, name=channel_name, avatar=avatar) self.store.update_channel_meta(_channel_id, name=channel_name, avatar=avatar)
if avatar: if avatar:
@@ -489,8 +490,8 @@ class JobManager:
# Process the freshly-seen window first, then the older backlog, so a # Process the freshly-seen window first, then the older backlog, so a
# `limit` still means "the newest N" now that discovery stops early. # `limit` still means "the newest N" now that discovery stops early.
pending = _order_pending(self.store.get_pending(_channel_id), refs) pending = order_pending(self.store.get_pending(_channel_id), refs)
pending = [r for r in pending if _keep_ref(cfg, include_shorts, no_live)(r)] pending = [r for r in pending if keep_ref(cfg, include_shorts, no_live)(r)]
if since: if since:
cutoff = since.replace("-", "") cutoff = since.replace("-", "")
pending = [r for r in pending if not r.upload_date or r.upload_date >= cutoff] pending = [r for r in pending if not r.upload_date or r.upload_date >= cutoff]
@@ -646,12 +647,12 @@ class JobManager:
max_window=sync.max_window, max_window=sync.max_window,
overlap=sync.overlap, overlap=sync.overlap,
since=self.store.latest_upload_date(channel_id) if incremental else None, since=self.store.latest_upload_date(channel_id) if incremental else None,
keep=_keep_ref(self.cfg), keep=keep_ref(self.cfg),
) )
refs = result.refs refs = result.refs
if result.full_scan: if result.full_scan:
self.store.upsert_channel(result.channel_id, _handle(channel_url), result.channel_name, len(refs)) self.store.upsert_channel(result.channel_id, extract_handle(channel_url), result.channel_name, len(refs))
new_videos = self.store.upsert_videos(refs) new_videos = self.store.upsert_videos(refs)
else: else:
self.store.update_channel_meta(result.channel_id, name=result.channel_name) self.store.update_channel_meta(result.channel_id, name=result.channel_name)
@@ -669,41 +670,6 @@ class JobManager:
} }
def _keep_ref(cfg: Config, include_shorts: bool | None = None, no_live: bool | None = None):
"""Predicate matching the shorts/live rules that decide what reaches the DB.
Incremental discovery needs the same filter its stored ids were created
under, otherwise the tail of a window is full of entries that can never be
recognised as known and the window keeps widening for nothing.
"""
shorts = cfg.include_shorts if include_shorts is None else include_shorts
skip_live = (not cfg.include_live) if no_live is None else no_live
def keep(r: Any) -> bool: # VideoRef or VideoRow — both carry `.url`
url = r.url or ""
if not shorts and "/shorts/" in url:
return False
if skip_live and url.startswith("https://www.youtube.com/live/"):
return False
return True
return keep
def _order_pending(rows: list, refs: list[VideoRef]) -> list:
"""Newest-window-first ordering for the pending queue.
`get_pending` is ordered by discovery time, which used to coincide with
newest-first because discovery saw the whole channel at once. With windowed
sync that no longer holds, so put the videos from this run's window at the
front and keep the rest of the backlog behind them.
"""
by_id = {r.video_id: r for r in rows}
ordered = [by_id.pop(x.video_id) for x in refs if x.video_id in by_id]
ordered.extend(by_id.values())
return ordered
def _clone_config(cfg: Config) -> Config: def _clone_config(cfg: Config) -> Config:
import copy import copy
return copy.deepcopy(cfg) return copy.deepcopy(cfg)
@@ -721,12 +687,6 @@ def _resolve_channel_url(store: Store, cfg: Config, channel_id: str | None) -> s
return f"https://www.youtube.com/channel/{channel_id}" return f"https://www.youtube.com/channel/{channel_id}"
def _handle(url: str) -> str:
if "@" in url:
return "@" + url.split("@", 1)[1].split("/", 1)[0]
return ""
def _safe_video_filename(title: str) -> str: def _safe_video_filename(title: str) -> str:
"""Keep the displayed title while making a valid, bounded Windows name.""" """Keep the displayed title while making a valid, bounded Windows name."""
invalid = set(r'\\/:*?"<>|') invalid = set(r'\\/:*?"<>|')
+251 -18
View File
@@ -92,19 +92,37 @@
const size = Number(localStorage.getItem("videos-size")); const size = Number(localStorage.getItem("videos-size"));
if ([10, 25, 50, 100].includes(size)) this.videos.size = size; if ([10, 25, 50, 100].includes(size)) this.videos.size = size;
this.hydrateURL(); this.hydrateURL();
// Every view transition pushes a real history entry, so the browser's
// back/forward buttons have to drive the SPA through popstate.
window.addEventListener("popstate", () => this._onPopState());
// Global shortcuts ("/" jumps to search) ride the same window wiring.
window.addEventListener("keydown", (e) => this._onGlobalKeydown(e));
this.checkHealth(); this.checkHealth();
this.loadDashboard(); this.loadDashboard();
this.loadChannels(); this.loadChannels();
if (this.view === "videos") this.loadVideos(this.videos.page); if (this.view === "videos") this.loadVideos(this.videos.page);
if (this.view === "search" && this.search.q) this.loadSearch(); if (this.view === "search" && this.search.q) this.loadSearch();
if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id, { push: false });
this.loadRetryable(); this.loadRetryable();
this.startLivePolling(); this.startLivePolling();
}, },
hydrateURL() { hydrateURL() {
const p = new URLSearchParams(window.location.search); this._fromParams(new URLSearchParams(window.location.search));
},
// Parse the query string into component state. Shared by the initial
// load and by popstate so browser back/forward restore the exact view
// (including filters). Resets to defaults first because popstate can go
// from a filtered list back to an unfiltered one.
_fromParams(p) {
const views = ["dashboard", "channels", "videos", "detail", "search", "analysis", "scrape", "cookies", "tools", "export"]; const views = ["dashboard", "channels", "videos", "detail", "search", "analysis", "scrape", "cookies", "tools", "export"];
if (views.includes(p.get("view"))) this.view = p.get("view"); const v = p.get("view");
this.view = views.includes(v) ? v : "dashboard";
if (this.view === "detail" && !p.get("video")) this.view = "videos";
this.filters = { channel: "", status: "", from: "", to: "", min_dur: "", q: "", sort: "upload_date" };
this.search.q = "";
this.search.channel = "";
if (this.view === "videos") { if (this.view === "videos") {
["channel", "status", "from", "to", "min_dur", "q", "sort"].forEach(k => { if (p.has(k)) this.filters[k] = p.get(k); }); ["channel", "status", "from", "to", "min_dur", "q", "sort"].forEach(k => { if (p.has(k)) this.filters[k] = p.get(k); });
const page = Number(p.get("page")); const page = Number(p.get("page"));
@@ -116,11 +134,11 @@
this.search.q = p.get("q") || ""; this.search.q = p.get("q") || "";
this.search.channel = p.get("channel") || ""; this.search.channel = p.get("channel") || "";
} }
// A stub is enough for init/popstate to know which video to load.
this.detail.video = this.view === "detail" ? { video_id: p.get("video") } : null;
}, },
syncURL() { _urlParams() {
clearTimeout(this._urlTimer);
this._urlTimer = setTimeout(() => {
const p = new URLSearchParams({ view: this.view }); const p = new URLSearchParams({ view: this.view });
if (this.view === "videos") { if (this.view === "videos") {
Object.entries(this.filters).forEach(([k, v]) => { if (v) p.set(k, v); }); Object.entries(this.filters).forEach(([k, v]) => { if (v) p.set(k, v); });
@@ -128,12 +146,45 @@
} else if (this.view === "search") { } else if (this.view === "search") {
if (this.search.q) p.set("q", this.search.q); if (this.search.q) p.set("q", this.search.q);
if (this.search.channel) p.set("channel", this.search.channel); if (this.search.channel) p.set("channel", this.search.channel);
} else if (this.view === "detail" && this.detail.video) {
p.set("video", this.detail.video.video_id);
} }
const next = "?" + p.toString(); return p;
if (next !== window.location.search) history.replaceState(null, "", next); },
syncURL() {
clearTimeout(this._urlTimer);
this._urlTimer = setTimeout(() => {
const next = "?" + this._urlParams().toString();
if (next !== window.location.search) history.replaceState(history.state, "", next);
}, 120); }, 120);
}, },
// View transitions push a history entry so browser back stays inside
// the app; param tweaks (filters, pagination) only replace via syncURL.
pushURL(extra) {
clearTimeout(this._urlTimer);
const p = this._urlParams();
if (extra && extra.video) p.set("video", extra.video);
const next = "?" + p.toString();
if (next === window.location.search) return;
history.pushState({ view: p.get("view") }, "", next);
this._pushedEntries = (this._pushedEntries || 0) + 1;
},
_onPopState() {
if (this._pushedEntries > 0) this._pushedEntries--;
this._fromParams(new URLSearchParams(window.location.search));
if (this.view === "detail") {
const id = (this.detail.video && this.detail.video.video_id) || "";
if (id && id !== this._detailLoadedId) this.openVideo(id, { push: false });
} else {
this._detailLoadedId = null;
this._armScrollRestore(this.view);
this._loadView(this.view);
}
},
toggleSidebar() { toggleSidebar() {
this.sidebarOpen = !this.sidebarOpen; this.sidebarOpen = !this.sidebarOpen;
localStorage.setItem("sidebar-open", String(this.sidebarOpen)); localStorage.setItem("sidebar-open", String(this.sidebarOpen));
@@ -177,10 +228,18 @@
} }
}, },
setView(v) { setView(v, opts) {
const o = opts || {};
this.view = v; this.view = v;
this.syncURL(); if (o.push === false) this.syncURL();
else this.pushURL();
if (window.innerWidth < 768) this.closeSidebar(); if (window.innerWidth < 768) this.closeSidebar();
if (!o.skipLoad) this._loadView(v);
},
// Data loads that accompany entering a view. Extracted from setView so
// popstate can run them without pushing a new history entry.
_loadView(v) {
// destroy stray canvases when leaving chart-bearing views // destroy stray canvases when leaving chart-bearing views
if (v !== "analysis") this.destroyCharts(["topwords", "timeline"]); if (v !== "analysis") this.destroyCharts(["topwords", "timeline"]);
if (v === "dashboard") this.loadDashboard(); if (v === "dashboard") this.loadDashboard();
@@ -194,6 +253,26 @@
if (v === "tools") { this.loadFolders(); this.loadFormat(); } if (v === "tools") { this.loadFolders(); this.loadFormat(); }
}, },
// One-shot scroll restore, armed ONLY by the back-navigation paths
// (_onPopState and goBack's fallback). Ordinary list interactions
// (pagination, filter changes, openChannel) never set it, so they keep
// landing at the top like a fresh view.
// 'videos': the table is fetched asynchronously, and scrolling before
// the fetch resolves gets clamped to the top of an empty table — so
// the flag is consumed in loadVideos()'s finally, after the rows exist.
// 'search': popstate does not reload results (they stay in memory), so
// restore right after the view switch; nextTick lets Alpine render the
// results and the extra requestAnimationFrame waits for layout, so the
// browser does not clamp the scroll to a not-yet-painted page.
_armScrollRestore(view) {
if (view === "videos") {
this._pendingScrollRestore = "videos";
} else if (view === "search") {
const top = (this._scrollMemory && this._scrollMemory.search) || 0;
this.$nextTick(() => requestAnimationFrame(() => window.scrollTo({ top })));
}
},
// ================================================================= // =================================================================
// dashboard // dashboard
// ================================================================= // =================================================================
@@ -323,14 +402,46 @@
const d = await this.api("/api/videos?" + p.toString()); const d = await this.api("/api/videos?" + p.toString());
this.videos = Object.assign({}, this.videos, { items: d.items, total: d.total, page: d.page, size: d.size }); this.videos = Object.assign({}, this.videos, { items: d.items, total: d.total, page: d.page, size: d.size });
} catch (e) { this.toast("Failed to load videos: " + e.message, "error"); } } catch (e) { this.toast("Failed to load videos: " + e.message, "error"); }
finally { this.loading.videos = false; } finally {
this.loading.videos = false;
// Back-navigation restore. It must happen here, after the fetch has
// resolved — scrolling any earlier is clamped to the top because the
// table is still empty while the request is in flight. The nextTick
// waits for Alpine to drop the transient "loading…" row (it renders
// above the data rows), so the position matches the row the user
// actually left. The flag is cleared on EVERY run, not only when it
// fired, so a stale one can never affect an unrelated loadVideos
// (pagination, filter change).
if (this._pendingScrollRestore === "videos") {
const top = (this._scrollMemory && this._scrollMemory.videos) || 0;
this.$nextTick(() => window.scrollTo({ top }));
}
this._pendingScrollRestore = null;
}
}, },
async openVideo(id) { async openVideo(id, opts) {
const o = opts || {};
// Remember where the user came from so the back button can label
// itself honestly. Skipped when restoring from history (popstate) or
// auto-refreshing after a job, where the origin must not change.
if (o.push !== false) {
if (this.view !== "detail") {
this.detail.fromView = this.view;
// Also remember how far down that view the user had scrolled,
// so returning from the detail can put them back on the same
// row instead of at the top of a reloaded table.
if (!this._scrollMemory) this._scrollMemory = {};
this._scrollMemory[this.view] = window.scrollY;
}
this.view = "detail"; // before pushURL: the entry must say where we went
this.pushURL({ video: id });
}
this.view = "detail"; this.view = "detail";
this._detailLoadedId = id;
this.loading.detail = true; this.loading.detail = true;
this.loading.transcript = true; this.loading.transcript = true;
this.detail = { video: null, transcript: [], hasAudio: false, audioPlaying: false, activeSeg: -1, videoError: false }; this.detail = { video: null, transcript: [], hasAudio: false, audioPlaying: false, activeSeg: -1, videoError: false, fromView: this.detail.fromView };
try { try {
const v = await this.api("/api/videos/" + encodeURIComponent(id)); const v = await this.api("/api/videos/" + encodeURIComponent(id));
// chapters may come embedded or be absent // chapters may come embedded or be absent
@@ -347,6 +458,30 @@
this.checkAudio(id); this.checkAudio(id);
}, },
// The single back affordance for the detail view: prefer real browser
// history (restores the list's filters via the URL), and fall back to
// the origin view when the detail URL was opened directly (refresh or
// shared link) and there is no in-app history to return to.
goBack() {
if ((this._pushedEntries || 0) > 0) { history.back(); return; }
const target = this.detail.fromView && this.detail.fromView !== "detail" ? this.detail.fromView : "videos";
// No in-app history to history.back() into — the popstate handler
// never fires, so arm the same scroll restore before the fallback
// setView() triggers the view's loaders.
this._armScrollRestore(target);
this.setView(target);
},
get backLabel() {
const map = {
search: "Back to search results",
dashboard: "Back to dashboard",
channels: "Back to channels",
videos: "Back to videos",
};
return map[this.detail.fromView] || "Back to videos";
},
async checkAudio(id) { async checkAudio(id) {
try { try {
const r = await fetch("/api/videos/" + encodeURIComponent(id) + "/audio", { method: "HEAD" }); const r = await fetch("/api/videos/" + encodeURIComponent(id) + "/audio", { method: "HEAD" });
@@ -384,7 +519,7 @@
openVideoPlayer() { openVideoPlayer() {
this.detail.videoError = false; this.detail.videoError = false;
this.$nextTick(() => { this.$nextTick(() => {
const player = this.$refs.videoPlayer; const player = this.ref("videoPlayer");
if (player) { player.load(); player.play().catch(() => {}); } if (player) { player.load(); player.play().catch(() => {}); }
}); });
}, },
@@ -426,7 +561,7 @@
}, },
seekAudio(sec) { seekAudio(sec) {
const a = this.$refs.audioPlayer; const a = this.ref("audioPlayer");
if (!a) return false; if (!a) return false;
try { a.currentTime = sec; a.play().catch(() => {}); } catch (_) {} try { a.currentTime = sec; a.play().catch(() => {}); } catch (_) {}
return true; return true;
@@ -434,7 +569,7 @@
seekTo(sec) { seekTo(sec) {
// seek the local audio player if available; otherwise just scroll the transcript // seek the local audio player if available; otherwise just scroll the transcript
if (this.detail.hasAudio && this.$refs.audioPlayer) { if (this.detail.hasAudio && this.ref("audioPlayer")) {
if (this.seekAudio(sec)) return; if (this.seekAudio(sec)) return;
} }
this.seekTranscript(sec); this.seekTranscript(sec);
@@ -477,6 +612,18 @@
}, },
toggleSelectAll() { if (this.allSelected) this.selectNone(); else this.selectAll(); }, toggleSelectAll() { if (this.allSelected) this.selectNone(); else this.selectAll(); },
// Active-filter chips: removing one must be one click, not a hunt
// through the selects — especially the channel filter set by
// openChannel(), which used to feel like a trap.
clearFilter(key) {
this.filters[key] = "";
this.loadVideos(1);
},
clearAllFilters() {
this.filters = { channel: "", status: "", from: "", to: "", min_dur: "", q: "", sort: "upload_date" };
this.loadVideos(1);
},
// ================================================================= // =================================================================
// downloads (.md / thumbnails / audio / single) // downloads (.md / thumbnails / audio / single)
// ================================================================= // =================================================================
@@ -502,6 +649,27 @@
this.downloadFile("/api/videos/" + encodeURIComponent(videoId) + "/markdown"); this.downloadFile("/api/videos/" + encodeURIComponent(videoId) + "/markdown");
}, },
async copyMd(videoId) {
try {
const res = await fetch("/api/videos/" + encodeURIComponent(videoId) + "/markdown");
if (!res.ok) throw new Error("Could not fetch markdown file");
const text = await res.text();
await navigator.clipboard.writeText(text);
this.toast("Contenido .md copiado al portapapeles");
} catch (e) {
this.toast("Error al copiar .md: " + e.message, "error");
}
},
async openMd(videoId) {
try {
await this.api("/api/videos/" + encodeURIComponent(videoId) + "/open-markdown", { method: "POST" });
this.toast("Abriendo archivo .md...");
} catch (e) {
this.toast("Error al abrir .md: " + e.message, "error");
}
},
async processOne(videoId) { async processOne(videoId) {
try { try {
const d = await this.api("/api/scrape/video/" + encodeURIComponent(videoId), { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({}) }); const d = await this.api("/api/scrape/video/" + encodeURIComponent(videoId), { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({}) });
@@ -681,9 +849,51 @@
if (tag === "INPUT" || tag === "TEXTAREA" || (e.target && e.target.isContentEditable)) return; if (tag === "INPUT" || tag === "TEXTAREA" || (e.target && e.target.isContentEditable)) return;
if (this.stats.open) { this.closeStats(); return; } if (this.stats.open) { this.closeStats(); return; }
if (this.clip.open) { this.closeClip(); return; } if (this.clip.open) { this.closeClip(); return; }
// Detail is the deepest layer of the app, so Esc backs out of it —
// same gesture that closes every modal. Guarded against the Esc that
// merely exits fullscreen playback.
if (this.view === "detail" && !document.fullscreenElement) { this.goBack(); return; }
if (this.cookies && this.cookies.drag) { this.cookies.drag = false; return; } if (this.cookies && this.cookies.drag) { this.cookies.drag = false; return; }
}, },
// Alpine 3.17 mounts template x-if content under its own scope, so
// x-ref elements never reach the root component's $refs — and every
// view here lives inside a template x-if. Resolve refs from the DOM
// instead; $refs stays as the fast path for anything ever registered.
ref(name) {
return (this.$refs && this.$refs[name]) || document.querySelector('#root [x-ref="' + name + '"]');
},
// "/" anywhere (outside form fields) jumps to the transcript search
// with the query input focused. Guarded like onGlobalEscape so it
// never fires while typing, composing or under a modal.
_onGlobalKeydown(e) {
if (e.key !== "/") return;
// IME composition and modifier combos belong to the browser/OS,
// not to this shortcut.
if (e.isComposing || e.ctrlKey || e.altKey || e.metaKey || e.shiftKey) return;
// Let editable fields receive a literal "/" instead of stealing it.
const el = document.activeElement;
const tag = el && el.tagName;
if (tag === "INPUT" || tag === "TEXTAREA" || tag === "SELECT" || (el && el.isContentEditable)) return;
// Modals own the keyboard while they are open.
if (this.confirmBox.open || this.stats.open || this.clip.open) return;
// preventDefault so the "/" never lands in the input we focus next.
e.preventDefault();
if (this.view !== "search") this.setView("search");
this._focusSearchInput(10);
},
// The search view is a template x-if whose content mounts a beat after
// the reactive flush, so a single $nextTick can query before the input
// exists. Retry across a few animation frames and focus as soon as it
// renders (no perceptible delay when it is already there).
_focusSearchInput(retries) {
const input = this.ref("searchInput");
if (input) { input.focus(); return; }
if (retries > 0) requestAnimationFrame(() => this._focusSearchInput(retries - 1));
},
// ================================================================= // =================================================================
// channels: pending download // channels: pending download
// ================================================================= // =================================================================
@@ -1007,7 +1217,10 @@
// Scrape tab (where startScrape lives) refreshed nothing at all, and // Scrape tab (where startScrape lives) refreshed nothing at all, and
// the setView cache guard then kept the stale rows on navigation. // the setView cache guard then kept the stale rows on navigation.
this.refreshLiveState({ force: true }); this.refreshLiveState({ force: true });
if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id); // Refresh the open detail view after a job, without pushing a new
// history entry — Back must still return to where the user came
// from, not to a duplicate of the same video.
if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id, { push: false });
this._armAutoHide(); this._armAutoHide();
}); });
es.addEventListener("cancelled", () => { this.scrape.log.push("[cancelled]"); this.closeStream(); this.loadJobs(); this._armAutoHide(); }); es.addEventListener("cancelled", () => { this.scrape.log.push("[cancelled]"); this.closeStream(); this.loadJobs(); this._armAutoHide(); });
@@ -1132,7 +1345,7 @@
if (!files.length) { this.toast("Drop .txt cookie files only", "error"); return; } if (!files.length) { this.toast("Drop .txt cookie files only", "error"); return; }
this.uploadCookies(files); this.uploadCookies(files);
// reset the file input so the same file can be picked again // reset the file input so the same file can be picked again
try { if (this.$refs.cookieFile) this.$refs.cookieFile.value = ""; } catch (_) {} try { const input = this.ref("cookieFile"); if (input) input.value = ""; } catch (_) {}
}, },
async uploadCookies(files) { async uploadCookies(files) {
@@ -1148,6 +1361,24 @@
} catch (e) { this.toast("Upload failed: " + e.message, "error"); } } catch (e) { this.toast("Upload failed: " + e.message, "error"); }
}, },
async importBrowserCookies() {
if (this.loading.cookies) return;
this.loading.cookies = true;
try {
const d = await this.api("/api/cookies/from-browser", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ browser: "brave" }),
});
this.toast("Imported " + (d.cookie_count || "") + " cookies from Brave and activated them");
await this.loadCookies();
} catch (e) {
this.toast("Brave import failed: " + e.message, "error");
} finally {
this.loading.cookies = false;
}
},
async activateCookie(id) { async activateCookie(id) {
try { await this.api("/api/cookies/" + encodeURIComponent(id) + "/activate", { method: "POST" }); this.loadCookies(); } try { await this.api("/api/cookies/" + encodeURIComponent(id) + "/activate", { method: "POST" }); this.loadCookies(); }
catch (e) { this.toast("Activate failed: " + e.message, "error"); } catch (e) { this.toast("Activate failed: " + e.message, "error"); }
@@ -1189,7 +1420,9 @@
// ================================================================= // =================================================================
openChannel(id) { openChannel(id) {
this.filters.channel = id; this.filters.channel = id;
this.setView("videos"); // skipLoad: loadVideos(1) below is the authoritative load — setView
// would trigger a second one with the stale page number.
this.setView("videos", { skipLoad: true });
this.loadVideos(1); this.loadVideos(1);
}, },
+145 -13
View File
@@ -86,19 +86,23 @@
<div class="grid grid-cols-2 md:grid-cols-4 gap-4"> <div class="grid grid-cols-2 md:grid-cols-4 gap-4">
<div class="glass p-4"> <div class="glass p-4">
<div class="text-xs text-zinc-500 uppercase tracking-wider">Channels</div> <div class="text-xs text-zinc-500 uppercase tracking-wider">Channels</div>
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="dash.channels.length"></div> <template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-12" aria-hidden="true"></div></template>
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="dash.channels.length"></div></template>
</div> </div>
<div class="glass p-4"> <div class="glass p-4">
<div class="text-xs text-zinc-500 uppercase tracking-wider">Videos</div> <div class="text-xs text-zinc-500 uppercase tracking-wider">Videos</div>
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.video_count||0),0))"></div> <template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-16" aria-hidden="true"></div></template>
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.video_count||0),0))"></div></template>
</div> </div>
<div class="glass p-4"> <div class="glass p-4">
<div class="text-xs text-zinc-500 uppercase tracking-wider">Total Views</div> <div class="text-xs text-zinc-500 uppercase tracking-wider">Total Views</div>
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.total_views||0),0))"></div> <template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-20" aria-hidden="true"></div></template>
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.total_views||0),0))"></div></template>
</div> </div>
<div class="glass p-4"> <div class="glass p-4">
<div class="text-xs text-zinc-500 uppercase tracking-wider">Total Duration</div> <div class="text-xs text-zinc-500 uppercase tracking-wider">Total Duration</div>
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="humanDur(dash.channels.reduce((s,c)=>s+(c.total_duration||0),0))"></div> <template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-20" aria-hidden="true"></div></template>
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="humanDur(dash.channels.reduce((s,c)=>s+(c.total_duration||0),0))"></div></template>
</div> </div>
</div> </div>
@@ -185,7 +189,31 @@
<table class="tbl"> <table class="tbl">
<thead><tr><th>Name</th><th>Handle</th><th>Videos</th><th>Pending</th><th>Latest video</th><th>Last scraped</th><th class="text-right">Actions</th></tr></thead> <thead><tr><th>Name</th><th>Handle</th><th>Videos</th><th>Pending</th><th>Latest video</th><th>Last scraped</th><th class="text-right">Actions</th></tr></thead>
<tbody> <tbody>
<template x-if="loading.channels"><tr><td colspan="7" class="text-center text-zinc-500 py-8">loading…</td></tr></template> <template x-if="loading.channels">
<template x-for="i in 5" :key="i">
<tr>
<td>
<template x-if="i===1"><span class="sr-only" role="status">Loading channels…</span></template>
<div class="flex items-center gap-2" aria-hidden="true">
<div class="skeleton w-7 h-7 rounded-full"></div>
<div class="skeleton h-4 w-28"></div>
</div>
</td>
<td><div class="skeleton h-3.5 w-20" aria-hidden="true"></div></td>
<td><div class="skeleton h-3.5 w-10" aria-hidden="true"></div></td>
<td><div class="skeleton h-5 w-20 rounded-full" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-16" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-16" aria-hidden="true"></div></td>
<td>
<div class="flex items-center justify-end gap-1" aria-hidden="true">
<div class="skeleton h-5 w-14"></div>
<div class="skeleton h-5 w-12"></div>
<div class="skeleton h-5 w-12"></div>
</div>
</td>
</tr>
</template>
</template>
<template x-if="!loading.channels && channels.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-8">No channels tracked yet.</td></tr></template> <template x-if="!loading.channels && channels.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-8">No channels tracked yet.</td></tr></template>
<template x-for="c in channels.items" :key="c.channel_id"> <template x-for="c in channels.items" :key="c.channel_id">
<tr class="row-hover"> <tr class="row-hover">
@@ -280,6 +308,25 @@
</div> </div>
</div> </div>
<!-- active filters: visible at a glance and one click to remove, so
a channel filter set from the dashboard never feels like a trap -->
<div x-show="filters.channel || filters.status" x-cloak class="flex flex-wrap items-center gap-2">
<span class="text-xs text-zinc-500">Showing</span>
<template x-if="filters.channel">
<button class="filter-chip" @click="clearFilter('channel')" title="Clear this filter and show all channels">
<span class="max-w-56 truncate" x-text="chanName(filters.channel)"></span>
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" aria-hidden="true"><path stroke-linecap="round" d="M6 6l12 12M18 6L6 18"/></svg>
</button>
</template>
<template x-if="filters.status">
<button class="filter-chip" @click="clearFilter('status')" title="Clear this status filter">
<span x-text="filters.status === '__blocked__' ? 'locked videos' : filters.status + ' videos'"></span>
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" aria-hidden="true"><path stroke-linecap="round" d="M6 6l12 12M18 6L6 18"/></svg>
</button>
</template>
<button class="text-xs text-zinc-400 hover:text-white underline underline-offset-2 transition-colors" @click="clearAllFilters()">Clear all</button>
</div>
<!-- bulk action bar --> <!-- bulk action bar -->
<template x-if="videos.selected.length > 0"> <template x-if="videos.selected.length > 0">
<div class="bulk-bar glass p-3 flex flex-wrap items-center gap-2"> <div class="bulk-bar glass p-3 flex flex-wrap items-center gap-2">
@@ -320,8 +367,31 @@
<th>Duration</th><th>Views</th><th>Status</th><th class="text-right">Actions</th> <th>Duration</th><th>Views</th><th>Status</th><th class="text-right">Actions</th>
</tr></thead> </tr></thead>
<tbody> <tbody>
<template x-if="loading.videos"><tr><td colspan="9" class="text-center text-zinc-500 py-10">loading…</td></tr></template> <!-- skeleton rows mirror the real columns; the sr-only status rides row 1 so each table has exactly one live region -->
<template x-if="!loading.videos && videos.items.length===0"><tr><td colspan="9" class="text-center text-zinc-500 py-10"><div>No videos match these filters.</div><button class="btn btn-ghost mt-3" @click.stop="filters={ channel:'', status:'', from:'', to:'', min_dur:'', q:'', sort:'upload_date' }; loadVideos(1)">Clear filters</button></td></tr></template> <template x-if="loading.videos">
<template x-for="i in 8" :key="i">
<tr>
<td>
<template x-if="i===1"><span class="sr-only" role="status">Loading videos…</span></template>
<div class="skeleton w-4 h-4 rounded" aria-hidden="true"></div>
</td>
<td><div class="skeleton w-20 h-12 rounded-lg" aria-hidden="true"></div></td>
<td><div class="skeleton h-4 w-full max-w-md" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-24" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-16" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-12" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-14" aria-hidden="true"></div></td>
<td><div class="skeleton h-5 w-16 rounded-full" aria-hidden="true"></div></td>
<td>
<div class="flex items-center justify-end gap-1" aria-hidden="true">
<div class="skeleton h-5 w-12"></div>
<div class="skeleton h-5 w-10"></div>
</div>
</td>
</tr>
</template>
</template>
<template x-if="!loading.videos && videos.items.length===0"><tr><td colspan="9" class="text-center text-zinc-500 py-10"><div>No videos match these filters.</div><button class="btn btn-ghost mt-3" @click.stop="clearAllFilters()">Clear filters</button></td></tr></template>
<template x-for="v in videos.items" :key="v.video_id"> <template x-for="v in videos.items" :key="v.video_id">
<tr class="row-hover" :class="isSelected(v.video_id) ? 'row-selected' : ''" @click="openVideo(v.video_id)"> <tr class="row-hover" :class="isSelected(v.video_id) ? 'row-selected' : ''" @click="openVideo(v.video_id)">
<td @click.stop><input type="checkbox" class="accent-rose-500" :checked="isSelected(v.video_id)" @change="toggleSelect(v.video_id)" /></td> <td @click.stop><input type="checkbox" class="accent-rose-500" :checked="isSelected(v.video_id)" @change="toggleSelect(v.video_id)" /></td>
@@ -368,7 +438,12 @@
<!-- ===== Detail ===== --> <!-- ===== Detail ===== -->
<template x-if="view==='detail'"> <template x-if="view==='detail'">
<section class="space-y-5" x-transition.opacity> <section class="space-y-5" x-transition.opacity>
<button class="btn btn-ghost" @click="setView('videos')">← Back to videos</button> <div class="flex items-center justify-between gap-3 flex-wrap">
<button class="btn btn-ghost" @click="goBack()" title="Go back (or press Esc)">
<svg class="w-4 h-4 shrink-0" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" aria-hidden="true"><path stroke-linecap="round" stroke-linejoin="round" d="M15 19l-7-7 7-7"/></svg>
<span x-text="backLabel"></span>
</button>
</div>
<template x-if="loading.detail"><div class="glass p-10 text-center text-zinc-500">loading…</div></template> <template x-if="loading.detail"><div class="glass p-10 text-center text-zinc-500">loading…</div></template>
<template x-if="!loading.detail && detail.video"> <template x-if="!loading.detail && detail.video">
<div class="space-y-5"> <div class="space-y-5">
@@ -400,6 +475,14 @@
<div class="flex flex-wrap gap-2 mt-4"> <div class="flex flex-wrap gap-2 mt-4">
<button class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="startDiscovery(detail.video.channel_id)" :disabled="jobActive()" title="Find recent videos from this channel without downloading transcripts">Investigate channel</button> <button class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="startDiscovery(detail.video.channel_id)" :disabled="jobActive()" title="Find recent videos from this channel without downloading transcripts">Investigate channel</button>
<button x-show="!isBlocked(detail.video) || detail.video.status==='done'" class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="detail.video.status==='done' ? downloadMd(detail.video.video_id) : processOne(detail.video.video_id)" x-text="detail.video.status==='done' ? 'Download .md' : 'Process to .md'"></button> <button x-show="!isBlocked(detail.video) || detail.video.status==='done'" class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="detail.video.status==='done' ? downloadMd(detail.video.video_id) : processOne(detail.video.video_id)" x-text="detail.video.status==='done' ? 'Download .md' : 'Process to .md'"></button>
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="copyMd(detail.video.video_id)" x-show="detail.video && detail.video.status==='done'" title="Copiar todo el contenido del .md al portapapeles">
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><rect x="9" y="9" width="13" height="13" rx="2" ry="2"/><path d="M5 15H4a2 2 0 0 1-2-2V4a2 2 0 0 1 2-2h9a2 2 0 0 1 2 2v1"/></svg>
<span>Copiar .md</span>
</button>
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="openMd(detail.video.video_id)" x-show="detail.video && detail.video.status==='done'" title="Abrir archivo .md en tu aplicación predeterminada (Obsidian, VS Code, etc.)">
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"/><polyline points="15 3 21 3 21 9"/><line x1="10" y1="14" x2="21" y2="3"/></svg>
<span>Abrir .md</span>
</button>
<button x-show="isBlocked(detail.video) && detail.video.status!=='done'" class="btn btn-ghost !py-1.5 !px-3 text-xs opacity-60" @click="processOne(detail.video.video_id)" :title="blockTitle(detail.video) + ' Click anyway if you have since bought the membership and refreshed your cookies.'">Try anyway</button> <button x-show="isBlocked(detail.video) && detail.video.status!=='done'" class="btn btn-ghost !py-1.5 !px-3 text-xs opacity-60" @click="processOne(detail.video.video_id)" :title="blockTitle(detail.video) + ' Click anyway if you have since bought the membership and refreshed your cookies.'">Try anyway</button>
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="downloadAudioOne(detail.video.video_id)" x-show="detail.video.status==='done'" x-text="detail.hasAudio ? 'Re-download audio' : 'Download audio'"> <button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="downloadAudioOne(detail.video.video_id)" x-show="detail.video.status==='done'" x-text="detail.hasAudio ? 'Re-download audio' : 'Download audio'">
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><path stroke-linecap="round" stroke-linejoin="round" d="M9 18V6l10-2v12M9 18a3 3 0 11-6 0 3 3 0 016 0zm10-2a3 3 0 11-6 0 3 3 0 016 0z"/></svg> <svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><path stroke-linecap="round" stroke-linejoin="round" d="M9 18V6l10-2v12M9 18a3 3 0 11-6 0 3 3 0 016 0zm10-2a3 3 0 11-6 0 3 3 0 016 0z"/></svg>
@@ -527,7 +610,7 @@
<section class="space-y-5" x-transition.opacity> <section class="space-y-5" x-transition.opacity>
<header><h1 class="text-2xl font-bold text-white">Search transcripts</h1><p class="text-sm text-zinc-500">Find moments across all videos</p></header> <header><h1 class="text-2xl font-bold text-white">Search transcripts</h1><p class="text-sm text-zinc-500">Find moments across all videos</p></header>
<form class="glass p-4 flex flex-col md:flex-row gap-2" @submit.prevent="loadSearch()"> <form class="glass p-4 flex flex-col md:flex-row gap-2" @submit.prevent="loadSearch()">
<input class="field flex-1" type="text" placeholder="search query… (Press Enter)" x-model="search.q" aria-label="Search query" /> <input x-ref="searchInput" class="field flex-1" type="text" placeholder="search query… (Press Enter or /)" x-model="search.q" aria-label="Search query" />
<select class="field md:w-56" x-model="search.channel" @change="if(search.ran) loadSearch()" aria-label="Filter by channel"><option value="">All channels</option><template x-for="c in channels.items" :key="c.channel_id"><option :value="c.channel_id" x-text="c.name"></option></template></select> <select class="field md:w-56" x-model="search.channel" @change="if(search.ran) loadSearch()" aria-label="Filter by channel"><option value="">All channels</option><template x-for="c in channels.items" :key="c.channel_id"><option :value="c.channel_id" x-text="c.name"></option></template></select>
<button class="btn accent-grad btn-primary" :disabled="loading.search || !search.q.trim()" @click="loadSearch()">Search</button> <button class="btn accent-grad btn-primary" :disabled="loading.search || !search.q.trim()" @click="loadSearch()">Search</button>
<button class="btn btn-ghost" type="button" @click="resetSearch()" x-show="search.ran || search.q || search.channel" title="Clear search">Clear</button> <button class="btn btn-ghost" type="button" @click="resetSearch()" x-show="search.ran || search.q || search.channel" title="Clear search">Clear</button>
@@ -540,7 +623,7 @@
</div> </div>
</template> </template>
<template x-if="!loading.search && !search.ran"> <template x-if="!loading.search && !search.ran">
<div class="glass p-10 text-center text-zinc-500">Type a query and press <kbd class="kbd">Enter</kbd> to search transcripts across all channels.<br><span class="text-xs text-zinc-500">Tip: try a phrase in its original language for best matches.</span></div> <div class="glass p-10 text-center text-zinc-500">Type a query and press <kbd class="kbd">Enter</kbd> to search transcripts across all channels. Press / anywhere to jump to this search.<br><span class="text-xs text-zinc-500">Tip: try a phrase in its original language for best matches.</span></div>
</template> </template>
<div class="space-y-3"> <div class="space-y-3">
<template x-for="r in search.items" :key="r.video_id+r.start_sec"> <template x-for="r in search.items" :key="r.video_id+r.start_sec">
@@ -675,7 +758,23 @@
<table class="tbl"> <table class="tbl">
<thead><tr><th>Channel</th><th>Status</th><th>Progress</th><th>Started</th><th>Finished</th><th></th></tr></thead> <thead><tr><th>Channel</th><th>Status</th><th>Progress</th><th>Started</th><th>Finished</th><th></th></tr></thead>
<tbody> <tbody>
<template x-if="loading.jobs"><tr><td colspan="6" class="text-center text-zinc-500 py-6">loading…</td></tr></template> <template x-if="loading.jobs">
<template x-for="i in 4" :key="i">
<tr>
<td>
<template x-if="i===1"><span class="sr-only" role="status">Loading jobs…</span></template>
<div class="skeleton h-3.5 w-28" aria-hidden="true"></div>
</td>
<td><div class="skeleton h-5 w-16 rounded-full" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-12" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-24" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-24" aria-hidden="true"></div></td>
<td>
<div class="flex justify-end" aria-hidden="true"><div class="skeleton h-5 w-12"></div></div>
</td>
</tr>
</template>
</template>
<template x-if="!loading.jobs && scrape.jobs.length===0"><tr><td colspan="6" class="text-center text-zinc-500 py-6">No jobs yet.</td></tr></template> <template x-if="!loading.jobs && scrape.jobs.length===0"><tr><td colspan="6" class="text-center text-zinc-500 py-6">No jobs yet.</td></tr></template>
<template x-for="j in scrape.jobs" :key="j.id"> <template x-for="j in scrape.jobs" :key="j.id">
<tr> <tr>
@@ -698,8 +797,19 @@
<section class="space-y-5" x-transition.opacity> <section class="space-y-5" x-transition.opacity>
<header><h1 class="text-2xl font-bold text-white">Cookie vault</h1><p class="text-sm text-zinc-500">Upload YouTube cookie files to authenticate</p></header> <header><h1 class="text-2xl font-bold text-white">Cookie vault</h1><p class="text-sm text-zinc-500">Upload YouTube cookie files to authenticate</p></header>
<div class="glass px-4 py-3 flex items-center justify-between gap-3">
<div>
<div class="text-sm text-zinc-200 font-medium">Import from Brave</div>
<div class="text-xs text-zinc-500">Pulls your logged-in YouTube session straight from the browser's cookie store — no extensions or manual exports. <span class="text-amber-400/90">Close Brave first</span>: it keeps its cookie database locked while running.</div>
</div>
<button class="btn !py-1.5 !px-3 text-xs shrink-0" :disabled="loading.cookies" @click="importBrowserCookies()">
<span x-show="!loading.cookies">Import from Brave</span>
<span x-show="loading.cookies">Importing…</span>
</button>
</div>
<div class="dropzone p-8 text-center transition-all" :class="cookies.drag ? 'drag' : ''" <div class="dropzone p-8 text-center transition-all" :class="cookies.drag ? 'drag' : ''"
@click="$refs.cookieFile.click()" @click="ref('cookieFile') && ref('cookieFile').click()"
@dragover.prevent="cookies.drag=true" @dragover.prevent="cookies.drag=true"
@dragleave.prevent="cookies.drag=false" @dragleave.prevent="cookies.drag=false"
@drop.prevent="handleDrop($event)"> @drop.prevent="handleDrop($event)">
@@ -718,7 +828,29 @@
<table class="tbl"> <table class="tbl">
<thead><tr><th>Label</th><th>File</th><th>Expires</th><th>Session</th><th>Active</th><th>Test</th><th></th></tr></thead> <thead><tr><th>Label</th><th>File</th><th>Expires</th><th>Session</th><th>Active</th><th>Test</th><th></th></tr></thead>
<tbody> <tbody>
<template x-if="loading.cookies"><tr><td colspan="7" class="text-center text-zinc-500 py-6">loading…</td></tr></template> <template x-if="loading.cookies">
<template x-for="i in 4" :key="i">
<tr>
<td>
<template x-if="i===1"><span class="sr-only" role="status">Loading cookies…</span></template>
<div class="skeleton h-4 w-24" aria-hidden="true"></div>
</td>
<td><div class="skeleton h-3 w-40" aria-hidden="true"></div></td>
<td><div class="skeleton h-3 w-28" aria-hidden="true"></div></td>
<td><div class="skeleton h-5 w-16 rounded-full" aria-hidden="true"></div></td>
<td><div class="skeleton w-4 h-4 rounded-full" aria-hidden="true"></div></td>
<td>
<div class="flex items-center gap-2" aria-hidden="true">
<div class="skeleton h-5 w-10"></div>
<div class="skeleton h-3 w-16"></div>
</div>
</td>
<td>
<div class="flex justify-end" aria-hidden="true"><div class="skeleton h-5 w-14"></div></div>
</td>
</tr>
</template>
</template>
<template x-if="!loading.cookies && cookies.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-6">Vault is empty.</td></tr></template> <template x-if="!loading.cookies && cookies.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-6">Vault is empty.</td></tr></template>
<template x-for="c in cookies.items" :key="c.id"> <template x-for="c in cookies.items" :key="c.id">
<tr> <tr>
+30
View File
@@ -107,6 +107,22 @@ select.field { appearance: none; background-image: linear-gradient(45deg, transp
border: 1px solid transparent; border: 1px solid transparent;
} }
/* ---- active-filter chips (videos) ---- */
/* Removable summary of the filters constraining the list: rose tint marks
them as the accent-colored "current context", the ✕ is the affordance. */
.filter-chip {
display: inline-flex; align-items: center; gap: 0.4rem;
padding: 0.25rem 0.65rem; border-radius: 999px;
background: rgba(244, 63, 94, 0.10);
border: 1px solid rgba(244, 63, 94, 0.35);
color: #fecdd3; font-size: 0.75rem; font-weight: 600;
cursor: pointer; transition: all 0.15s ease;
}
.filter-chip:hover { background: rgba(244, 63, 94, 0.20); color: #fff; border-color: rgba(244, 63, 94, 0.55); }
.filter-chip:active { transform: translateY(1px); }
.filter-chip svg { width: 0.75rem; height: 0.75rem; opacity: 0.7; flex-shrink: 0; }
.filter-chip:hover svg { opacity: 1; }
/* ---- table ---- */ /* ---- table ---- */
.tbl { width: 100%; border-collapse: separate; border-spacing: 0; } .tbl { width: 100%; border-collapse: separate; border-spacing: 0; }
.tbl th { text-align: left; font-size: 0.7rem; text-transform: uppercase; letter-spacing: 0.05em; color: #71717a; font-weight: 600; padding: 0.6rem 0.75rem; border-bottom: 1px solid #27272a; } .tbl th { text-align: left; font-size: 0.7rem; text-transform: uppercase; letter-spacing: 0.05em; color: #71717a; font-weight: 600; padding: 0.6rem 0.75rem; border-bottom: 1px solid #27272a; }
@@ -143,6 +159,20 @@ select.field { appearance: none; background-image: linear-gradient(45deg, transp
.spin { animation: spin 0.9s linear infinite; } .spin { animation: spin 0.9s linear infinite; }
@keyframes spin { to { transform: rotate(360deg); } } @keyframes spin { to { transform: rotate(360deg); } }
/* ---- skeleton loaders ---- */
/* One class only: size/shape is varied per element with Tailwind utilities. */
.skeleton {
background: linear-gradient(90deg, #18181b 25%, #27272a 50%, #18181b 75%);
background-size: 200% 100%;
animation: skeleton-shimmer 1.4s ease infinite;
border-radius: 0.375rem;
}
@keyframes skeleton-shimmer {
0% { background-position: 200% 0; }
100% { background-position: -200% 0; }
}
@media (prefers-reduced-motion: reduce) { .skeleton { animation: none; } }
/* ---- wordcloud ---- */ /* ---- wordcloud ---- */
.cloud-word { display: inline-block; margin: 0.35rem 0.4rem; line-height: 1; cursor: default; transition: opacity 0.15s ease; } .cloud-word { display: inline-block; margin: 0.35rem 0.4rem; line-height: 1; cursor: default; transition: opacity 0.15s ease; }
.cloud-word:hover { opacity: 0.75; } .cloud-word:hover { opacity: 0.75; }
+152
View File
@@ -0,0 +1,152 @@
from __future__ import annotations
from pathlib import Path
import pytest
from yt_scraper.cookies import (
BrowserCookieLockedError,
auto_import_dir,
delete,
import_from_browser,
is_expired,
parse_netscape,
resolve_active_path,
)
from yt_scraper.store import CookieRow, Store
SAMPLE_NETSCAPE = """# Netscape HTTP Cookie File
.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSID\tsample_sid_token
.youtube.com\tTRUE\t/\tTRUE\t2000000000\tLOGIN_INFO\tsample_login_info
"""
def test_parse_netscape():
ok, info = parse_netscape(SAMPLE_NETSCAPE)
assert ok is True
assert info["count"] == 2
assert info["has_session"] is True
assert info["expires_at"] is not None
def test_parse_netscape_httponly_lines_are_data():
# SID/HSID are HttpOnly; Netscape exports prefix those lines and a
# comment-skipping parser would silently strip the login out.
text = (
"# Netscape HTTP Cookie File\n"
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSID\ttok\n"
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tHSID\ttok\n"
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSSID\ttok\n"
"# a real comment\n"
)
ok, info = parse_netscape(text)
assert ok is True
assert info["count"] == 3
assert info["has_session"] is True
def test_partial_session_export_is_not_a_session():
# A lone __Secure-3PSID (partial extension export) is anonymous to
# YouTube; it must not be reported as a usable session.
text = (
"# Netscape HTTP Cookie File\n"
".youtube.com\tTRUE\t/\tTRUE\t2000000000\t__Secure-3PSID\ttok\n"
".youtube.com\tTRUE\t/\tTRUE\t2000000000\t__Secure-3PAPISID\ttok\n"
)
ok, info = parse_netscape(text)
assert ok is True
assert info["has_session"] is False
def _browser_cookie(name, value, *, domain=".youtube.com", httponly=False):
import http.cookiejar
return http.cookiejar.Cookie(
version=0, name=name, value=value, port=None, port_specified=False,
domain=domain, domain_specified=True, domain_initial_dot=domain.startswith("."),
path="/", path_specified=True, secure=True, expires=2000000000,
discard=False, comment=None, comment_url=None,
rest={"HttpOnly": None} if httponly else {},
)
def test_import_from_browser_writes_vault_file(tmp_path: Path, monkeypatch):
store = Store(tmp_path / "state.db")
cookie_dir = tmp_path / "cookies"
def fake_extract(browser, profile=None, logger=None, **kw):
return iter([
_browser_cookie("SID", "sid_tok", httponly=True),
_browser_cookie("HSID", "hsid_tok", httponly=True),
_browser_cookie("SSID", "ssid_tok", httponly=True),
_browser_cookie("__Secure-3PSID", "psid_tok"),
_browser_cookie("PREF", "pref_tok", domain=".google.com"), # not youtube -> dropped
])
import yt_dlp.cookies as ydl_cookies
monkeypatch.setattr(ydl_cookies, "extract_cookies_from_browser", fake_extract)
cid = import_from_browser(store, browser="brave", cookie_dir=cookie_dir)
row = store.get_cookie(cid)
assert row is not None
assert row.cookie_count == 4
assert row.has_session is True
# The written file must round-trip as a valid Netscape file whose
# HttpOnly session cookies survive the vault's own parser.
from yt_scraper.cookies import parse_netscape_file
path = cookie_dir / row.filename
ok, info = parse_netscape_file(path)
assert ok is True
assert info["count"] == 4
assert info["has_session"] is True
def test_import_from_browser_maps_locked_db(tmp_path: Path, monkeypatch):
store = Store(tmp_path / "state.db")
def fake_extract(browser, profile=None, logger=None, **kw):
raise RuntimeError("Could not copy Chrome cookie database. See https://github.com/yt-dlp/yt-dlp/issues/7271 for more info")
import yt_dlp.cookies as ydl_cookies
monkeypatch.setattr(ydl_cookies, "extract_cookies_from_browser", fake_extract)
with pytest.raises(BrowserCookieLockedError) as exc:
import_from_browser(store, browser="brave", cookie_dir=tmp_path / "cookies")
assert "close brave" in str(exc.value).lower()
def test_auto_import_and_prune_dead_cookies(tmp_path: Path):
db_path = tmp_path / "state.db"
store = Store(db_path)
cookie_dir = tmp_path / "cookies"
cookie_dir.mkdir()
# Place two cookie files
f1 = cookie_dir / "c1.txt"
f1.write_text(SAMPLE_NETSCAPE, encoding="utf-8")
f2 = cookie_dir / "c2.txt"
f2.write_text(SAMPLE_NETSCAPE, encoding="utf-8")
# auto_import imports both and activates the first
n = auto_import_dir(store, dir_path=cookie_dir)
assert n == 2
assert len(store.list_cookies()) == 2
active = store.get_active_cookie()
assert active is not None
assert active.filename in ("c1.txt", "c2.txt")
active_path = resolve_active_path(store, cookie_dir=cookie_dir)
assert active_path is not None
assert Path(active_path).exists()
# Now simulate user deleting the active cookie file from disk
Path(active_path).unlink()
# resolve_active_path should fall back to the surviving cookie file
fallback_path = resolve_active_path(store, cookie_dir=cookie_dir)
assert fallback_path is not None
assert Path(fallback_path).exists()
# auto_import_dir should clean up the deleted cookie row from DB
auto_import_dir(store, dir_path=cookie_dir)
assert len(store.list_cookies()) == 1
+68
View File
@@ -0,0 +1,68 @@
from __future__ import annotations
from pathlib import Path
from fastapi import FastAPI
from fastapi.testclient import TestClient
import pytest
from yt_scraper.config import Config
from yt_scraper.store import Store, VideoRef
from yt_scraper.webapp.api import build_router
from yt_scraper.webapp.jobs import JobManager
@pytest.fixture
def api_client(tmp_path: Path, monkeypatch):
db_path = tmp_path / "state.db"
md_root = tmp_path / "markdown"
md_root.mkdir()
store = Store(db_path)
cfg = Config(database_path=str(db_path), output_dir=str(md_root))
jobs = JobManager(store, cfg)
router = build_router(store, cfg, jobs)
app = FastAPI()
app.include_router(router)
client = TestClient(app)
return client, store, cfg, tmp_path
def test_get_and_open_markdown(api_client, monkeypatch):
client, store, cfg, tmp_path = api_client
store.upsert_channel("UC1", "@test", "Test Channel", 1)
store.upsert_videos([
VideoRef("vid1", "UC1", "Test Video", "https://www.youtube.com/watch?v=vid1")
])
# 1. Not generated yet -> 404
r_get = client.get("/api/videos/vid1/markdown")
assert r_get.status_code == 404
r_open = client.post("/api/videos/vid1/open-markdown")
assert r_open.status_code == 404
# 2. Create markdown file and mark done
md_dir = tmp_path / "markdown" / "Test Channel"
md_dir.mkdir(parents=True)
md_file = md_dir / "2026-08-23_test-video.md"
md_content = "# Test Video\n\nContent here"
md_file.write_text(md_content, encoding="utf-8", newline="\n")
rel_path = "markdown/Test Channel/2026-08-23_test-video.md"
store.mark_done("vid1", rel_path, "es", "auto", False)
# 3. GET markdown returns text
r_get = client.get("/api/videos/vid1/markdown")
assert r_get.status_code == 200
assert r_get.text.replace("\r\n", "\n") == md_content
# 4. POST open-markdown calls _open_in_os
opened_paths = []
from yt_scraper.webapp import api as api_mod
monkeypatch.setattr(api_mod, "_open_in_os", lambda p: opened_paths.append(str(p)))
r_open = client.post("/api/videos/vid1/open-markdown")
assert r_open.status_code == 200
assert len(opened_paths) == 1
assert str(md_file.resolve()) in [str(Path(p).resolve()) for p in opened_paths]
+4 -2
View File
@@ -136,14 +136,16 @@ def test_cookie_vault(store):
sample = ( sample = (
"# Netscape HTTP Cookie File\n" "# Netscape HTTP Cookie File\n"
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSID\tabc\n" ".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSID\tabc\n"
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSAPISID\tdef\n" ".youtube.com\tTRUE\t/\tTRUE\t1800287621\tHSID\tdef\n"
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSSID\tghi\n"
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSAPISID\tjkl\n"
".youtube.com\tTRUE\t/\tFALSE\t1800287621\tPREF\tf1=2\n" ".youtube.com\tTRUE\t/\tFALSE\t1800287621\tPREF\tf1=2\n"
) )
cid = cookies.import_text(store, sample, label="test", cookie_dir=str(Path(store.db_path).parent / "ck")) cid = cookies.import_text(store, sample, label="test", cookie_dir=str(Path(store.db_path).parent / "ck"))
vault = cookies.list_vault(store) vault = cookies.list_vault(store)
assert len(vault) == 1 assert len(vault) == 1
assert vault[0].has_session is True assert vault[0].has_session is True
assert vault[0].cookie_count == 3 assert vault[0].cookie_count == 5
# set active + resolve # set active + resolve
cookies.set_active(store, cid) cookies.set_active(store, cid)
path = cookies.resolve_active_path(store, cookie_dir=str(Path(store.db_path).parent / "ck")) path = cookies.resolve_active_path(store, cookie_dir=str(Path(store.db_path).parent / "ck"))
+88
View File
@@ -0,0 +1,88 @@
from __future__ import annotations
import io
import json
from pathlib import Path
from yt_scraper.extract import extract_via_watch_page
WATCH_PAGE = (
"var ytInitialPlayerResponse = "
+ json.dumps({
"playabilityStatus": {"status": "OK"},
"videoDetails": {
"title": "Members only test",
"author": "Canal",
"lengthSeconds": "42",
"viewCount": "7",
"shortDescription": "desc",
"keywords": ["a", "b"],
},
"microformat": {"playerMicroformatRenderer": {"publishDate": "2026-09-01"}},
"captions": {"playerCaptionsTracklistRenderer": {"captionTracks": [
{"languageCode": "es", "kind": "asr", "baseUrl": "https://captions.test/es"},
]}},
})
+ ";"
)
JSON3 = json.dumps({
"events": [
{"tStartMs": 0, "dDurationMs": 1000, "segs": [{"utf8": "hola "}, {"utf8": "mundo"}]},
{"tStartMs": 1000, "dDurationMs": 500, "segs": [{"utf8": "segundo"}]},
]
})
class _FakeResponse(io.BytesIO):
def __enter__(self):
return self
def __exit__(self, *a):
return False
def test_watch_page_fallback_builds_segments(tmp_path: Path, monkeypatch):
from yt_scraper import extract as ex
def fake_urlopen(cookies_file, url, **kw):
if url.startswith("https://www.youtube.com/"):
return _FakeResponse(WATCH_PAGE.encode("utf-8"))
assert "fmt=json3" in url
return _FakeResponse(JSON3.encode("utf-8"))
monkeypatch.setattr(ex, "_session_urlopen", fake_urlopen)
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
data = extract_via_watch_page(
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
)
assert data is not None
assert data.subtitle is not None and data.subtitle.lang == "es"
# parse_auto_dump merges caption events with no gap between them.
assert len(data.segments) == 1
assert data.segments[0].text == "hola mundo segundo"
assert data.info["title"] == "Members only test"
assert data.info["upload_date"] == "20260901"
assert data.info["duration"] == 42
assert data.skip_reason is None
def test_watch_page_fallback_without_tracks_returns_none(tmp_path: Path, monkeypatch):
from yt_scraper import extract as ex
page = WATCH_PAGE.replace('"captionTracks": [{', '"captionTracks": [{')
page = "var ytInitialPlayerResponse = " + json.dumps({
"playabilityStatus": {"status": "OK"},
"videoDetails": {"title": "x"},
}) + ";"
monkeypatch.setattr(ex, "_session_urlopen",
lambda *a, **kw: _FakeResponse(page.encode("utf-8")))
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
assert extract_via_watch_page(
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
) is None