diff --git a/src/yt_scraper/cli.py b/src/yt_scraper/cli.py index 219c475..7eb395f 100644 --- a/src/yt_scraper/cli.py +++ b/src/yt_scraper/cli.py @@ -13,11 +13,10 @@ from rich.progress import Progress, SpinnerColumn, TextColumn, BarColumn, TaskPr from rich.table import Table from .config import Config, load_config, parse_languages -from .store import Store, VideoRef, VideoRow -from .discover import discover_channel, discover_incremental -from .chapters import align_chapters, chapters_from_info, Chapter, Section -from .parse import Segment -from .render import build_filename_stem, render_markdown +from .store import Store, order_pending +from .discover import discover_channel, discover_incremental, extract_handle +from .segments import seconds_to_ts +from .render import safe_filename from .ratelimit import ThrottleGuard, configure_global_pacer, polite_sleep from .pipeline import process_video from .cookies import auto_import_dir, resolve_active_path @@ -157,7 +156,7 @@ def _run_scrape(obj, limit, since, languages, no_auto, no_shorts, include_shorts channel_id, channel_name, refs = result.channel_id, result.channel_name, result.refs if result.full_scan: - store.upsert_channel(channel_id, _extract_handle(cfg.channel_url), channel_name, len(refs)) + store.upsert_channel(channel_id, extract_handle(cfg.channel_url), channel_name, len(refs)) else: store.update_channel_meta(channel_id, name=channel_name) console.print( @@ -190,7 +189,7 @@ def _run_scrape(obj, limit, since, languages, no_auto, no_shorts, include_shorts else: # Discovery only saw the newest slice; keep the older backlog reachable # but put this run's videos first so --limit still means "los mas nuevos". - pending = _order_pending(pending, refs) + pending = order_pending(pending, refs) if limit: pending = pending[:limit] if not pending: @@ -244,7 +243,7 @@ def _known_channel_for(store: Store, channel_url: str) -> str | None: """ if not channel_url: return None - handle = _extract_handle(channel_url).lstrip("@").lower() + handle = extract_handle(channel_url).lstrip("@").lower() tail = channel_url.rstrip("/").split("/")[-1] for ch in store.list_channels(): cid = ch.get("channel_id") or "" @@ -256,14 +255,6 @@ def _known_channel_for(store: Store, channel_url: str) -> str | None: return None -def _order_pending(rows: list, refs: list) -> list: - """Videos from this run's window first, then the rest of the backlog.""" - by_id = {r.video_id: r for r in rows} - ordered = [by_id.pop(r.video_id) for r in refs if r.video_id in by_id] - ordered.extend(by_id.values()) - return ordered - - def _print_dry_run(refs): table = Table(show_lines=False) table.add_column("Fecha", style="dim") @@ -388,7 +379,13 @@ def export_cmd(obj, fmt, out_dir): def audio_cmd(obj, limit, since, force): """Descarga audio MP3 (requiere ffmpeg).""" if not shutil.which("ffmpeg"): - console.print("[red]ffmpeg no encontrado.[/red] Instala: winget install ffmpeg") + if sys.platform == "win32": + hint = "winget install ffmpeg (o choco install ffmpeg)" + elif sys.platform == "darwin": + hint = "brew install ffmpeg" + else: + hint = "apt install ffmpeg (Debian/Ubuntu) / dnf install ffmpeg (Fedora) / pacman -S ffmpeg (Arch)" + console.print(f"[red]ffmpeg no encontrado.[/red] Instala: {hint}") sys.exit(1) store: Store = obj.store cfg: Config = obj.cfg @@ -408,6 +405,7 @@ def audio_cmd(obj, limit, since, force): "outtmpl": str(out_dir / "%(title)s.%(ext)s"), "postprocessors": [{"key": "FFmpegExtractAudio", "preferredcodec": "mp3", "preferredquality": "128"}], "quiet": True, "no_warnings": True, "noprogress": True, + "js_runtimes": {"node": {}, "deno": {}, "bun": {}, "quickjs": {}}, } if cookie_path: ydl_opts["cookiefile"] = cookie_path @@ -415,7 +413,7 @@ def audio_cmd(obj, limit, since, force): ydl_opts["cookiesfrombrowser"] = (obj.cookies_from_browser,) with yt_dlp.YoutubeDL(ydl_opts) as ydl: for v in videos: - target = out_dir / f"{_safe_filename(v.title or v.video_id)}.mp3" + target = out_dir / f"{safe_filename(v.title or v.video_id)}.mp3" if target.exists() and not force: continue try: @@ -454,7 +452,7 @@ def channels_add(obj, url): cfg.channel_url = url console.print("[cyan]Resolviendo canal...[/cyan]") channel_id, name, avatar, refs = discover_channel(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests) - obj.store.upsert_channel(channel_id, _extract_handle(url), name, len(refs), avatar=avatar) + obj.store.upsert_channel(channel_id, extract_handle(url), name, len(refs), avatar=avatar) obj.store.upsert_videos(refs) console.print(f"[green]Added:[/green] {name} ({channel_id}) — {len(refs)} videos") @@ -549,6 +547,7 @@ def re_render_cmd(obj, backfill): """Regenerar Markdown desde segmentos almacenados.""" store: Store = obj.store cfg: Config = obj.cfg + from .pipeline import re_render_videos from .segments import backfill_from_markdown if backfill: md_root = Path(cfg.output_dir_resolved) @@ -559,26 +558,13 @@ def re_render_cmd(obj, backfill): if not videos: console.print("[yellow]No hay videos con segments_json. Usa --backfill.[/yellow]") return - import json - from .render import render_markdown console.print(f"[cyan]Re-renderizando {len(videos)} videos...[/cyan]") - for v in videos: - segs = [Segment(start=s["start"], end=s["end"], text=s["text"]) for s in json.loads(v.segments_json)] - chapters = [Chapter(title=c["title"], start_time=c["start"], end_time=c.get("end", c["start"])) for c in json.loads(v.chapters_json or "[]")] - sections = align_chapters(segs, chapters) - context = { - "video_id": v.video_id, "title": v.title or v.video_id, "channel_name": "", - "channel_id": v.channel_id, "channel_url": "", "upload_date": v.upload_date or "", - "duration": v.duration or 0, "url": v.url, "transcript_lang": v.transcript_lang or "", - "transcript_src": v.transcript_src or "", "view_count": v.view_count, "like_count": v.like_count, - "tags": _parse_tags(v.tags), "thumbnail": v.thumbnail or "", "description": v.description or "", - "sections": sections, - } - stem = build_filename_stem(v.upload_date, v.title or v.video_id, cfg.filename_template) - ch = store.get_channel(v.channel_id) - out_subdir = Path(cfg.output_dir_resolved) / _safe_dirname((ch or {}).get("name") or "unknown") - render_markdown(cfg.template_path_resolved, out_subdir, stem, context) - console.print(f"[green]Re-render completo.[/green]") + # Delegate to the pipeline implementation instead of a local loop: it + # retires the .md a re-render supersedes (renamed titles used to leave + # orphans), re-points markdown_path via mark_done and fills channel_name + # from the channels table. + n = sum(re_render_videos(store, cfg, cid) for cid in targets) + console.print(f"[green]Re-render completo:[/green] {n} videos") # --------------------------------------------------------------------------- helpers @@ -612,12 +598,6 @@ def _title_for(store: Store, video_id: str) -> str | None: return v.title if v else None -def _extract_handle(url: str) -> str: - if "@" in url: - return "@" + url.split("@", 1)[1].split("/", 1)[0] - return "" - - def _fmt_date(d: str | None) -> str: if not d: return "" @@ -634,33 +614,6 @@ def _fmt_duration(seconds: int | None) -> str: return f"{h}:{m:02d}:{s:02d}" if h else f"{m}:{s:02d}" -def seconds_to_ts(sec: float) -> str: - total = int(sec) - h, rem = divmod(total, 3600) - m, s = divmod(rem, 60) - return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}" - - -def _safe_dirname(name: str) -> str: - safe = "".join(c for c in name if c not in r'\/:*?"<>|') - return safe.strip().strip(".") or "unknown" - - -def _safe_filename(name: str) -> str: - safe = "".join(c for c in name if c not in r'\/:*?"<>|') - return safe.strip().strip(".") or "untitled" - - -def _parse_tags(tags_json: str | None) -> list[str]: - if not tags_json: - return [] - import json - try: - return json.loads(tags_json) - except (json.JSONDecodeError, TypeError): - return [] - - def _parse_interval(s: str) -> float: s = s.strip().lower() if s.endswith("s"): diff --git a/src/yt_scraper/config.py b/src/yt_scraper/config.py index bf4d945..058deb2 100644 --- a/src/yt_scraper/config.py +++ b/src/yt_scraper/config.py @@ -105,6 +105,30 @@ class Config: LANGUAGE_MODES = ("manual", "auto", "any") +def keep_ref(cfg: Config, include_shorts: bool | None = None, no_live: bool | None = None): + """Predicate matching the shorts/live rules that decide what reaches the DB. + + Lives next to the `Config` flags it reads so the policy has one home. + Incremental discovery needs the same filter its stored ids were created + under, otherwise the tail of a window is full of entries that can never be + recognised as known and the window keeps widening for nothing. The optional + overrides serve the job runner, whose per-run opts may disagree with the + config the store was populated under. + """ + shorts = cfg.include_shorts if include_shorts is None else include_shorts + skip_live = (not cfg.include_live) if no_live is None else no_live + + def keep(r) -> bool: # VideoRef or VideoRow — both carry `.url` + url = r.url or "" + if not shorts and "/shorts/" in url: + return False + if skip_live and url.startswith("https://www.youtube.com/live/"): + return False + return True + + return keep + + def parse_languages(raw: Any, prefer_manual: bool) -> dict[str, str]: """Normalise legacy list / new dict / None into ``{lang: mode}``.""" default = "manual" if prefer_manual else "auto" diff --git a/src/yt_scraper/cookies.py b/src/yt_scraper/cookies.py index 0d43bca..3d97b78 100644 --- a/src/yt_scraper/cookies.py +++ b/src/yt_scraper/cookies.py @@ -1,6 +1,7 @@ from __future__ import annotations import logging +import os import uuid from datetime import datetime, timezone from pathlib import Path @@ -10,7 +11,12 @@ from .store import CookieRow, Store log = logging.getLogger(__name__) +# A logged-in YouTube session always sets this core set. A lone +# `__Secure-3PSID` (partial extension export) is NOT a session — YouTube +# treats the request as anonymous and members-only content stays locked, +# which is how an "active membership" cookie once failed invisibly. SESSION_COOKIE_NAMES = {"SID", "SAPISID", "__Secure-3PSID", "SSID", "LOGIN_INFO", "HSID", "APISID"} +FULL_SESSION_NAMES = {"SID", "HSID", "SSID"} _DEFAULT_DIR = Path("cookies") @@ -28,10 +34,16 @@ def parse_netscape(text: str) -> tuple[bool, dict]: """ lines = text.splitlines() expiries: list[int] = [] + session_expiries: list[int] = [] names: set[str] = set() count = 0 for line in lines: line = line.rstrip("\n") + # "#HttpOnly_" is a data prefix, not a comment — and the session + # cookies themselves (SID, HSID, ...) are HttpOnly, so skipping + # these lines would silently strip the login out of an export. + if line.startswith("#HttpOnly_"): + line = line[len("#HttpOnly_"):] if not line.strip() or line.startswith("#"): continue parts = line.split("\t") @@ -48,14 +60,20 @@ def parse_netscape(text: str) -> tuple[bool, dict]: count += 1 if expiry: expiries.append(expiry) + if name in SESSION_COOKIE_NAMES: + session_expiries.append(expiry) if count == 0: return False, {"count": 0, "has_session": False, "expires_at": None, "names": set()} - has_session = bool(names & SESSION_COOKIE_NAMES) + has_session = FULL_SESSION_NAMES <= names or "LOGIN_INFO" in names + # What the user needs to know is when the LOGIN dies, not when the + # earliest throwaway cookie (YSC and friends live hours) lapses — so + # report the session cookies' own expiry, falling back to the file max. + relevant = session_expiries or expiries expires_at = None - if expiries: - earliest = min(expiries) - if earliest > 0: - expires_at = datetime.fromtimestamp(earliest, tz=timezone.utc).isoformat(timespec="seconds") + if relevant: + latest = max(relevant) + if latest > 0: + expires_at = datetime.fromtimestamp(latest, tz=timezone.utc).isoformat(timespec="seconds") return True, {"count": count, "has_session": has_session, "expires_at": expires_at, "names": names} @@ -99,10 +117,15 @@ def import_text(store: Store, text: str, label: str, cookie_dir: str | Path | No def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int: """Import any loose .txt Netscape files in dir that aren't tracked yet. Returns count.""" cdir = cookies_dir(dir_path) - tracked = {c.filename for c in store.list_cookies()} + tracked = store.list_cookies() + for c in tracked: + if not (cdir / c.filename).exists(): + store.delete_cookie(c.id) + + tracked_filenames = {c.filename for c in store.list_cookies()} n = 0 for f in sorted(cdir.glob("*.txt")): - if f.name in tracked: + if f.name in tracked_filenames: continue ok, info = parse_netscape_file(f) if not ok: @@ -118,14 +141,210 @@ def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int: cookie_count=info["count"], ) n += 1 - # activate first cookie if none active - if not store.get_active_cookie(): + # activate first cookie if none active or active file missing + active = store.get_active_cookie() + if not active or not (cdir / active.filename).exists(): cookies = store.list_cookies() if cookies: store.set_active_cookie(cookies[0].id) return n +class BrowserCookieLockedError(RuntimeError): + """The browser's cookie database could not be read — it is running. + + Chromium opens its Cookies file without sharing read access, so the + browser (including its background/tray processes) must be fully closed + for the extraction to succeed. + """ + + +# Where the Brave executable lives on a default install, in preference +# order. The %VAR% placeholders only expand on Windows (elsewhere they stay +# literal and simply never match), so one flat cross-platform list works. +_BRAVE_EXE_CANDIDATES = ( + # Windows + r"%ProgramFiles%\BraveSoftware\Brave-Browser\Application\brave.exe", + r"%LocalAppData%\BraveSoftware\Brave-Browser\Application\brave.exe", + # macOS (system-wide Applications and per-user ~/Applications) + "/Applications/Brave Browser.app/Contents/MacOS/Brave Browser", + "~/Applications/Brave Browser.app/Contents/MacOS/Brave Browser", + # Linux (official deb/rpm packages, distro builds, snap, manual installs) + "/usr/bin/brave-browser", + "/usr/bin/brave", + "/opt/brave.com/brave/brave-browser", + "/snap/bin/brave", + "/usr/local/bin/brave-browser", +) + +# Default profile ("User Data") directories per OS — same flat-list trick: +# only the one for the current OS exists, the rest never match. +_BRAVE_USER_DATA_CANDIDATES = ( + r"%LocalAppData%\BraveSoftware\Brave-Browser\User Data", # Windows + "~/Library/Application Support/BraveSoftware/Brave-Browser", # macOS + "~/.config/BraveSoftware/Brave-Browser", # Linux +) + + +def _expand_path(candidate: str) -> str: + """Expand %VAR% (Windows) and ~ (macOS/Linux) in a path candidate.""" + return os.path.expanduser(os.path.expandvars(candidate)) + + +def _find_brave() -> tuple[str | None, str | None]: + """(exe_path, user_data_dir) for a default Brave install on Windows/macOS/Linux.""" + exe = next( + (p for p in (_expand_path(c) for c in _BRAVE_EXE_CANDIDATES) if os.path.isfile(p)), + None, + ) + ud = next( + (p for p in (_expand_path(c) for c in _BRAVE_USER_DATA_CANDIDATES) if os.path.isdir(p)), + None, + ) + return exe, ud + + +def _cdp_cookie(cookie: dict): + """DevTools cookie dict -> http.cookiejar.Cookie.""" + import http.cookiejar + domain = cookie.get("domain") or ".youtube.com" + http_only = bool(cookie.get("httpOnly")) + return http.cookiejar.Cookie( + version=0, name=cookie["name"], value=cookie.get("value") or "", + port=None, port_specified=False, + domain=domain, domain_specified=True, domain_initial_dot=domain.startswith("."), + path=cookie.get("path") or "/", path_specified=True, + secure=bool(cookie.get("secure")), + expires=int(cookie.get("expires")) if cookie.get("expires") else None, + discard=False, comment=None, comment_url=None, + rest={"HttpOnly": None} if http_only else {}, + ) + + +def _extract_brave_cdp(timeout: float = 45.0) -> list: + """Launch Brave headless on its REAL profile and read decrypted cookies + over DevTools. + + Chromium 127+ encrypts new cookies "app-bound" (v20): only the browser + itself can decrypt them, which is why yt-dlp's file-based extraction + dies with "Failed to decrypt with DPAPI". Launching the browser with + its user-data-dir passed EXPLICITLY on the command line keeps remote + debugging allowed (Chromium 136+ blocks it for the implicit default + dir), and the browser hands its own cookies over in plaintext. The + browser must still be closed — the profile is single-writer. + """ + import json + import socket + import subprocess + import time + import urllib.request + + exe, user_data = _find_brave() + if not exe or not user_data: + raise RuntimeError( + "Brave installation not found (looked in the default install " + "locations for Windows, macOS and Linux)" + ) + + with socket.socket() as s: + s.bind(("127.0.0.1", 0)) + port = s.getsockname()[1] + + proc = subprocess.Popen( + [ + exe, "--headless=new", f"--remote-debugging-port={port}", + f"--user-data-dir={user_data}", "--no-first-run", + "--no-default-browser-check", "about:blank", + ], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + try: + deadline = time.monotonic() + timeout + version = None + while time.monotonic() < deadline: + try: + version = json.load(urllib.request.urlopen( + f"http://127.0.0.1:{port}/json/version", timeout=2)) + break + except Exception: + time.sleep(0.5) + if version is None: + # Most common cause: Brave is already running and the new + # process just delegated to it without opening a debug port. + raise BrowserCookieLockedError( + "could not attach to Brave — close Brave completely " + "(including background processes) and try again" + ) + + from websockets.sync.client import connect + with connect(version["webSocketDebuggerUrl"], open_timeout=10) as ws: + ws.send(json.dumps({"id": 1, "method": "Storage.getCookies"})) + reply = json.loads(ws.recv()) + return [_cdp_cookie(c) for c in reply.get("result", {}).get("cookies", [])] + finally: + proc.kill() + + + +def import_from_browser( + store: Store, + browser: str = "brave", + profile: str | None = None, + label: str | None = None, + cookie_dir: str | Path | None = None, +) -> str: + """Pull youtube.com cookies straight from a local browser profile. + + Tries yt-dlp's file-based extraction first; on Chromium 127+ profiles + whose cookies are app-bound (v20 — "Failed to decrypt with DPAPI"), + falls back to launching the browser itself headless and reading the + decrypted cookies over DevTools. Only youtube.com cookies are kept and + stored in the vault as a normal Netscape file, so activation, expiry + tracking and every extraction path work unchanged. + """ + from yt_dlp.cookies import extract_cookies_from_browser + + cookies = None + try: + cookies = extract_cookies_from_browser(browser, profile or None) + except Exception as exc: + msg = str(exc) + if "Could not copy Chrome cookie database" in msg or "database is locked" in msg: + raise BrowserCookieLockedError( + f"could not read {browser}'s cookie database — close {browser} completely " + "(including background processes) and try again" + ) from exc + # App-bound (v20) cookies: only the browser can decrypt them. + if browser == "brave" and ("decrypt with DPAPI" in msg or "decrypt" in msg.lower()): + log.info("Brave cookies are app-bound; falling back to DevTools extraction") + cookies = _extract_brave_cdp() + else: + raise + + lines = [] + for c in cookies: + if "youtube.com" not in (c.domain or ""): + continue + # Netscape format; the #HttpOnly_ prefix is stripped again on parse. + prefix = "#HttpOnly_" if c.has_nonstandard_attr("httponly") else "" + domain = c.domain or ".youtube.com" + include_subdomains = "TRUE" if domain.startswith(".") else "FALSE" + expiry = int(c.expires) if c.expires else 0 + lines.append( + f"{prefix}{domain}\t{include_subdomains}\t{c.path or '/'}\t" + f"{'TRUE' if c.secure else 'FALSE'}\t{expiry}\t{c.name}\t{c.value}" + ) + if not lines: + raise ValueError(f"no youtube.com cookies found in {browser} (not logged in?)") + + text = ( + "# Netscape HTTP Cookie File\n" + f"# Extracted from {browser} profile {profile or 'default'}\n" + + "\n".join(lines) + "\n" + ) + return import_text(store, text, label=label or f"{browser}", cookie_dir=cookie_dir) + + def set_active(store: Store, cookie_id: str) -> None: store.set_active_cookie(cookie_id) @@ -143,12 +362,18 @@ def delete(store: Store, cookie_id: str, cookie_dir: str | Path | None = None) - def resolve_active_path(store: Store, cookie_dir: str | Path | None = None) -> str | None: - row = store.get_active_cookie() - if not row: - return None cdir = cookies_dir(cookie_dir) - path = cdir / row.filename - return str(path) if path.exists() else None + row = store.get_active_cookie() + if row: + path = cdir / row.filename + if path.exists(): + return str(path) + for c in store.list_cookies(): + path = cdir / c.filename + if path.exists(): + store.set_active_cookie(c.id) + return str(path) + return None def is_expired(row: CookieRow) -> bool: diff --git a/src/yt_scraper/discover.py b/src/yt_scraper/discover.py index 8a40851..d82f653 100644 --- a/src/yt_scraper/discover.py +++ b/src/yt_scraper/discover.py @@ -211,6 +211,17 @@ def discover_incremental( size = min(size * 2, max_window) +def extract_handle(url: str) -> str: + """@handle embedded in a channel URL, or "" for /channel/ URLs. + + Every caller that records a channel (CLI, webapp add-channel, jobs, watch) + normalises the handle the same way; this is that one shared definition. + """ + if "@" in url: + return "@" + url.split("@", 1)[1].split("/", 1)[0] + return "" + + def _pick_channel_avatar(info: dict[str, Any]) -> str | None: """Best-effort channel avatar URL from a yt-dlp channel info dict. diff --git a/src/yt_scraper/extract.py b/src/yt_scraper/extract.py index bf58f79..0bd8729 100644 --- a/src/yt_scraper/extract.py +++ b/src/yt_scraper/extract.py @@ -1,7 +1,12 @@ from __future__ import annotations +import json import logging +import os +import re +import urllib.request from dataclasses import dataclass +from pathlib import Path from typing import Any, Mapping import yt_dlp @@ -100,8 +105,20 @@ def extract_video( # One video extraction is two requests: the watch page and the InnerTube # player call. yt-dlp spaces them itself; the pacer needs to know they exist. GLOBAL_PACER.wait(cost=2) - with yt_dlp.YoutubeDL(ydl_opts) as ydl: - info = ydl.extract_info(video_url, download=False) + try: + with yt_dlp.YoutubeDL(ydl_opts) as ydl: + info = ydl.extract_info(video_url, download=False) + except Exception: + # Logged-in sessions on current YouTube increasingly end here + # ("The page needs to be reloaded" / format-availability failures). + # The session itself is usually fine — the watch page still hands + # metadata and caption tracks to a plain cookie'd GET — so try that + # before giving up. Without cookies there is nothing to fall back to. + if cookies_file: + fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual) + if fallback is not None: + return fallback + raise pick = pick_subtitle(info, languages_dict, prefer_manual) segments: list[Segment] = [] @@ -128,6 +145,131 @@ def extract_video( ) +_WATCH_PAGE_UA = ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36" +) + +# Precompiled: the fallback can fire on every video of a throttled batch, and +# re-compiling per call showed up under those runs. +_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;") + +# One opener per (cookie file, mtime): the jar parse is per-call work that is +# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under +# the same path still gets a fresh jar; only the newest entry is kept. +_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {} + + +def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0): + path = str(Path(cookies_file).resolve()) + try: + mtime = os.path.getmtime(path) + except OSError: + mtime = -1.0 + key = (path, mtime) + opener = _OPENER_CACHE.get(key) + if opener is None: + jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file) + jar.load(ignore_discard=True, ignore_expires=True) + opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar)) + opener.addheaders = [ + ("User-Agent", _WATCH_PAGE_UA), + ("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"), + ] + _OPENER_CACHE.clear() + _OPENER_CACHE[key] = opener + return opener.open(url, timeout=timeout) + + +def extract_via_watch_page( + video_url: str, + cookies_file: str, + languages: Mapping[str, str], + prefer_manual: bool = True, +) -> VideoData | None: + """Session-cookie fallback: scrape the watch page directly. + + yt-dlp's InnerTube clients reject logged-in sessions that lack a PO + token (playability "The page needs to be reloaded") or return no + formats/captions, which kills cookie-authenticated videos — members + being the case this exists for. The plain watch page served to the + logged-in browser still carries `ytInitialPlayerResponse` with + metadata and caption tracks, so GET it with the vault cookie and + reuse the normal subtitle picker. Returns None when the page holds + no caption tracks at all, so callers keep their own error semantics. + """ + GLOBAL_PACER.wait() + html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace") + m = _INITIAL_PLAYER_RE.search(html) + if not m: + log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url) + return None + try: + pr = json.loads(m.group(1)) + except json.JSONDecodeError: + log.warning("watch-page fallback: unparseable player response for %s", video_url) + return None + + status = (pr.get("playabilityStatus") or {}).get("status") + if status != "OK": + reason = (pr.get("playabilityStatus") or {}).get("reason") or status + raise RuntimeError(f"watch-page fallback: video not playable ({reason})") + + details = pr.get("videoDetails") or {} + micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {} + tracks = ( + (pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {} + ).get("captionTracks") or [] + if not tracks: + return None + + # Reuse pick_subtitle by shaping the tracks as an info dict. + info: dict[str, Any] = { + "title": details.get("title"), + "channel": details.get("author"), + "duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None, + "view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None, + "description": details.get("shortDescription") or "", + "tags": details.get("keywords") or [], + "thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"), + "upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None, + # The watch page carries no availability signal; leaving it unset + # keeps the (more informed) discovery value in the store. + "availability": None, + "subtitles": {}, + "automatic_captions": {}, + } + for t in tracks: + base = t.get("baseUrl") or "" + if not base: + continue + entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}] + if t.get("kind") == "asr": + info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry) + else: + info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry) + + pick = pick_subtitle(info, languages, prefer_manual) + segments: list[Segment] = [] + skip_reason: str | None = None + if pick: + try: + GLOBAL_PACER.wait() + raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace") + segments = parse_auto_dump(raw) + if not segments: + skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)" + except Exception as exc: # pylint: disable=broad-except + skip_reason = f"caption download failed via watch-page fallback: {exc}" + else: + skip_reason = describe_missing_subtitle(info, languages) + + return VideoData( + info=info, segments=segments, subtitle=pick, + has_chapters=bool(info.get("chapters")), skip_reason=skip_reason, + ) + + def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str: """Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'. diff --git a/src/yt_scraper/monitor.py b/src/yt_scraper/monitor.py index 597e8ee..84b1cbc 100644 --- a/src/yt_scraper/monitor.py +++ b/src/yt_scraper/monitor.py @@ -4,9 +4,9 @@ import logging import time from typing import Callable -from .config import Config +from .config import Config, keep_ref from .cookies import resolve_active_path -from .discover import discover_incremental +from .discover import discover_incremental, extract_handle from .pipeline import process_video from .ratelimit import ThrottleGuard, polite_sleep from .store import Store @@ -77,13 +77,13 @@ def _run_once( max_window=cfg.sync.max_window, overlap=cfg.sync.overlap, since=store.latest_upload_date(channel_id) if known else None, - keep=_keep_ref(cfg), + keep=keep_ref(cfg), ) channel_name, refs = result.channel_name, result.refs target_channel = channel_id or result.channel_id if result.full_scan: store.upsert_channel( - target_channel, _extract_handle(cfg.channel_url), channel_name, len(refs), avatar=result.avatar + target_channel, extract_handle(cfg.channel_url), channel_name, len(refs), avatar=result.avatar ) else: store.update_channel_meta(target_channel, name=channel_name, avatar=result.avatar) @@ -133,23 +133,3 @@ def _run_once( _emit(f"watch: throttled, waiting {wait:.1f}s") time.sleep(wait) polite_sleep(cfg.delay.min_seconds, cfg.delay.max_seconds) - - -def _keep_ref(cfg: Config): - """Same shorts/live rules the store was populated under — see jobs._keep_ref.""" - - def keep(r) -> bool: - url = r.url or "" - if not cfg.include_shorts and "/shorts/" in url: - return False - if not cfg.include_live and url.startswith("https://www.youtube.com/live/"): - return False - return True - - return keep - - -def _extract_handle(url: str) -> str: - if "@" in url: - return "@" + url.split("@", 1)[1].split("/", 1)[0] - return "" diff --git a/src/yt_scraper/pipeline.py b/src/yt_scraper/pipeline.py index 89efc77..1e3e976 100644 --- a/src/yt_scraper/pipeline.py +++ b/src/yt_scraper/pipeline.py @@ -9,7 +9,7 @@ from .chapters import align_chapters, chapters_from_info from .config import Config from .extract import extract_video from .ratelimit import is_rate_limited -from .render import build_filename_stem, render_markdown +from .render import build_filename_stem, render_markdown, safe_dirname from .store import Store, VideoRow from ._yt_http import yt_get @@ -44,7 +44,7 @@ def _render_and_retire( template=cfg.filename_template, video_id=video_id, ) - out_subdir = Path(cfg.output_dir_resolved) / _safe_dirname(channel_name) + out_subdir = Path(cfg.output_dir_resolved) / safe_dirname(channel_name) md_path = render_markdown(cfg.template_path_resolved, out_subdir, stem, context) out_root = Path(cfg.output_dir_resolved).parent @@ -122,14 +122,16 @@ def process_video( return "error" info = data.info - store.set_availability(row.video_id, info.get("availability")) # Metadata is persisted BEFORE the no-transcript exit. The extraction already # cost its requests and the info dict is in hand; discarding it because the # separate caption fetch failed means a retry re-spends them for data we # already had. Measured after a throttling incident: five rows left with # upload_date, view_count, description and thumbnail all NULL. - _store_metadata(store, row, info) + # Agrupadas en una transaccion: una conexion/commit en vez de tres. + with store.transaction(): + store.set_availability(row.video_id, info.get("availability")) + _store_metadata(store, row, info) if not data.segments: reason = data.skip_reason or "no transcript" @@ -148,21 +150,23 @@ def process_video( chapters = chapters_from_info(data.info) sections = align_chapters(data.segments, chapters) - # persist segments + rich metadata to DB (for search, stats, webapp) - store.store_segments(row.video_id, data.segments) - seg_json = json.dumps( - [{"start": s.start, "end": s.end, "text": s.text} for s in data.segments], - ensure_ascii=False, - ) - ch_json = json.dumps( - [{"title": c.title, "start": c.start_time, "end": c.end_time} for c in chapters], - ensure_ascii=False, - ) - store.update_video_metadata( - row.video_id, - chapters_json=ch_json, - segments_json=seg_json, - ) + # persist segments + rich metadata to DB (for search, stats, webapp); + # una transaccion: delete+inserts+update atomicos y un solo commit + with store.transaction(): + store.store_segments(row.video_id, data.segments) + seg_json = json.dumps( + [{"start": s.start, "end": s.end, "text": s.text} for s in data.segments], + ensure_ascii=False, + ) + ch_json = json.dumps( + [{"title": c.title, "start": c.start_time, "end": c.end_time} for c in chapters], + ensure_ascii=False, + ) + store.update_video_metadata( + row.video_id, + chapters_json=ch_json, + segments_json=seg_json, + ) context = { "video_id": row.video_id, @@ -205,11 +209,6 @@ def _normalize_date(d: str | None) -> str: return d -def _safe_dirname(name: str) -> str: - safe = "".join(c for c in name if c not in r'\/:*?"<>|') - return safe.strip().strip(".") or "unknown" - - def thumbnail_url_for(video_row) -> str: """Thumbnail URL for a video: stored URL, else the canonical YouTube one derived from its id.""" if video_row and getattr(video_row, "thumbnail", None): @@ -259,7 +258,6 @@ def re_render_videos(store: Store, cfg: Config, channel_id: str | None = None) - import json from .chapters import align_chapters, Chapter from .parse import Segment - from .render import build_filename_stem, render_markdown videos = [v for v in store.get_all(channel_id) if v.status == "done" and v.segments_json] n = 0 diff --git a/src/yt_scraper/ratelimit.py b/src/yt_scraper/ratelimit.py index e118d90..01518fd 100644 --- a/src/yt_scraper/ratelimit.py +++ b/src/yt_scraper/ratelimit.py @@ -191,6 +191,7 @@ def ydl_throttle_opts( *, extractor_retries: int = 3, socket_timeout: float = 30.0, + js_runtimes: dict[str, dict] | None = None, ) -> dict[str, object]: """The politeness half of every `ydl_opts` dict in this project. @@ -213,6 +214,7 @@ def ydl_throttle_opts( # this only covers transient 5xx and network errors. "extractor_retries": int(extractor_retries), "socket_timeout": float(socket_timeout), + "js_runtimes": js_runtimes if js_runtimes is not None else {"node": {}, "deno": {}, "bun": {}, "quickjs": {}}, } diff --git a/src/yt_scraper/render.py b/src/yt_scraper/render.py index 52b825c..0bab96e 100644 --- a/src/yt_scraper/render.py +++ b/src/yt_scraper/render.py @@ -32,16 +32,26 @@ def to_json(value) -> str: return json.dumps(value, ensure_ascii=False) +# One Environment per template directory, cached for the process lifetime: +# building it (loader + filters) per rendered note was the dominant cost of +# batch renders. Jinja's own per-env template cache keeps the compiled +# Template, and its default auto_reload still picks up on-disk edits. +_ENV_CACHE: dict[Path, Environment] = {} + + def _make_env(template_dir: Path) -> Environment: - env = Environment( - loader=FileSystemLoader(str(template_dir)), - autoescape=select_autoescape(disabled_extensions=("j2", "txt")), - trim_blocks=True, - lstrip_blocks=True, - ) - env.filters["format_timestamp"] = format_timestamp - env.filters["quote_yaml"] = quote_yaml - env.filters["to_json"] = to_json + env = _ENV_CACHE.get(template_dir) + if env is None: + env = Environment( + loader=FileSystemLoader(str(template_dir)), + autoescape=select_autoescape(disabled_extensions=("j2", "txt")), + trim_blocks=True, + lstrip_blocks=True, + ) + env.filters["format_timestamp"] = format_timestamp + env.filters["quote_yaml"] = quote_yaml + env.filters["to_json"] = to_json + _ENV_CACHE[template_dir] = env return env @@ -63,6 +73,24 @@ def render_markdown( return out_file +def safe_dirname(name: str | None) -> str: + """Channel name -> markdown subdirectory name, shared by every .md writer. + + Strips the characters Windows forbids in a path segment, then leading and + trailing spaces/dots (also illegal there), falling back to "unknown" so the + output tree never grows a nameless root. Callers with a bare filename want + `safe_filename` instead ("untitled" fallback). + """ + safe = "".join(c for c in (name or "") if c not in r'\/:*?"<>|') + return safe.strip().strip(".") or "unknown" + + +def safe_filename(name: str) -> str: + """Free-text (title) -> file-legal stem; "untitled" when nothing survives.""" + safe = "".join(c for c in name if c not in r'\/:*?"<>|') + return safe.strip().strip(".") or "untitled" + + def build_filename_stem( upload_date: str | None, title: str, diff --git a/src/yt_scraper/segments.py b/src/yt_scraper/segments.py index 1449955..f0832bf 100644 --- a/src/yt_scraper/segments.py +++ b/src/yt_scraper/segments.py @@ -8,7 +8,7 @@ from pathlib import Path from typing import Callable from .parse import Segment -from .store import SearchHit, Store +from .store import Store log = logging.getLogger(__name__) @@ -91,14 +91,6 @@ def _strip_quotes(value: str) -> str: return v -def store_segments(store: Store, video_id: str, segments: list[Segment]) -> None: - store.store_segments(video_id, segments) - - -def search(store: Store, query: str, channel_id: str | None = None, limit: int = 50) -> list[SearchHit]: - return store.search_segments(query, channel_id=channel_id, limit=limit) - - def backfill_from_markdown( store: Store, md_root: Path, diff --git a/src/yt_scraper/store.py b/src/yt_scraper/store.py index 4c28bf0..8716a8d 100644 --- a/src/yt_scraper/store.py +++ b/src/yt_scraper/store.py @@ -2,6 +2,7 @@ from __future__ import annotations import json import sqlite3 +import threading from contextlib import contextmanager from dataclasses import dataclass from datetime import datetime, timezone @@ -393,10 +394,30 @@ class JobRow: last_error: str | None +def order_pending(rows: list[VideoRow], refs: list[VideoRef]) -> list[VideoRow]: + """Newest-window-first ordering for the pending queue. + + `get_pending` is ordered by discovery time, which used to coincide with + newest-first because discovery saw the whole channel at once. With windowed + sync that no longer holds, so put the videos from this run's window (the + `refs` discovery just returned, newest first) at the front and keep the rest + of the backlog behind them. Shared by the CLI scrape and the webapp's + channel jobs, which must agree on what `--limit` / `limit` mean. + """ + by_id = {r.video_id: r for r in rows} + ordered = [by_id.pop(x.video_id) for x in refs if x.video_id in by_id] + ordered.extend(by_id.values()) + return ordered + + class Store: def __init__(self, db_path: str | Path): self.db_path = Path(db_path) self.db_path.parent.mkdir(parents=True, exist_ok=True) + # Pila de transacciones ambientales, por hilo: permite agrupar varias + # llamadas a Store en una sola conexion/commit sin pasar la conexion + # como parametro a cada metodo. + self._local = threading.local() self._init_schema() def _connect(self) -> sqlite3.Connection: @@ -419,13 +440,53 @@ class Store: if col not in ch_existing: conn.execute(f"ALTER TABLE channels ADD COLUMN {col} {coltype}") conn.executescript(_EXTRA_SCHEMA) - # Unconditional, not "only when the column was just added": a row can - # also arrive unranked afterwards, and an unranked row is displayed - # in the wrong place rather than merely in an arbitrary one. - _rank_unranked(conn) + # Unconditional in spirit, not in cost: a row can also arrive + # unranked afterwards, and an unranked row is displayed in the + # wrong place rather than merely in an arbitrary one. But + # _rank_unranked scans the whole videos table, and Store is built + # on every webapp import — so gate it behind the same predicate it + # matches on, which is a single indexed probe. + if conn.execute( + "SELECT 1 FROM videos WHERE channel_seq IS NULL LIMIT 1" + ).fetchone() is not None: + _rank_unranked(conn) + + @contextmanager + def transaction(self): + """Agrupa varias escrituras de Store en una sola conexion y un commit. + + `process_video` costaba ~6 connect/commit por video (un fsync cada + uno en WAL). Dentro de este bloque, toda llamada a Store hecha desde + el MISMO hilo se une a la conexion ambiental; un error revierte el + grupo entero, que es la atomicidad por video que se quiere de todos + modos. + """ + conn = self._connect() + stack: list[sqlite3.Connection] = getattr(self._local, "stack", None) or [] + self._local.stack = stack + stack.append(conn) + try: + yield conn + conn.commit() + except BaseException: + conn.rollback() + raise + finally: + stack.remove(conn) + conn.close() @contextmanager def _cursor(self) -> Iterator[sqlite3.Cursor]: + # Dentro de transaction(): misma conexion y sin commit intermedio — + # el bloque externo decide cuando el trabajo se vuelve durable. + stack: list[sqlite3.Connection] = getattr(self._local, "stack", None) or [] + if stack: + cur = stack[-1].cursor() + try: + yield cur + finally: + cur.close() + return conn = self._connect() try: yield conn.cursor() @@ -576,34 +637,40 @@ class Store: for rank, (_, r) in enumerate(ordered): seqs[r.video_id] = base + width - rank - for r in refs: - cur.execute( - """INSERT INTO videos - (video_id, channel_id, title, url, upload_date, - upload_date_approx, duration, availability, - channel_seq, status, discovered_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?) - ON CONFLICT(video_id) DO UPDATE SET - title = COALESCE(excluded.title, videos.title), - upload_date = CASE - WHEN excluded.upload_date IS NULL THEN videos.upload_date - WHEN videos.upload_date IS NULL THEN excluded.upload_date - WHEN COALESCE(videos.upload_date_approx, 0) = 1 - THEN excluded.upload_date - ELSE videos.upload_date END, - upload_date_approx = CASE - WHEN videos.upload_date IS NOT NULL - AND COALESCE(videos.upload_date_approx, 0) = 0 - THEN videos.upload_date_approx - ELSE COALESCE(excluded.upload_date_approx, - videos.upload_date_approx) END, - duration = COALESCE(excluded.duration, videos.duration), - availability = COALESCE(excluded.availability, videos.availability), - channel_seq = COALESCE(excluded.channel_seq, videos.channel_seq)""", + # One executemany instead of a prepared-statement round per ref: + # discovery hands over hundreds of rows at once and the statement + # text is identical for all of them. + cur.executemany( + """INSERT INTO videos + (video_id, channel_id, title, url, upload_date, + upload_date_approx, duration, availability, + channel_seq, status, discovered_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'pending', ?) + ON CONFLICT(video_id) DO UPDATE SET + title = COALESCE(excluded.title, videos.title), + upload_date = CASE + WHEN excluded.upload_date IS NULL THEN videos.upload_date + WHEN videos.upload_date IS NULL THEN excluded.upload_date + WHEN COALESCE(videos.upload_date_approx, 0) = 1 + THEN excluded.upload_date + ELSE videos.upload_date END, + upload_date_approx = CASE + WHEN videos.upload_date IS NOT NULL + AND COALESCE(videos.upload_date_approx, 0) = 0 + THEN videos.upload_date_approx + ELSE COALESCE(excluded.upload_date_approx, + videos.upload_date_approx) END, + duration = COALESCE(excluded.duration, videos.duration), + availability = COALESCE(excluded.availability, videos.availability), + channel_seq = COALESCE(excluded.channel_seq, videos.channel_seq)""", + [ (r.video_id, r.channel_id, r.title, r.url, r.upload_date, - int(r.date_approx or 0), r.duration, - r.availability, seqs.get(r.video_id), now), - ) + int(r.date_approx or 0), r.duration, r.availability, + seqs.get(r.video_id), now) + for r in refs + ], + ) + for r in refs: if r.video_id not in existing: inserted += 1 existing.add(r.video_id) @@ -1038,34 +1105,44 @@ class Store: def dashboard(self) -> dict: with self._cursor() as cur: channels = [dict(r) for r in cur.execute("SELECT * FROM channels ORDER BY name").fetchall()] + # One GROUP BY instead of one aggregate query per channel (N+1). + # Channels with zero videos are absent from the grouping and get + # the same zeros/NULLs the per-channel query used to return. + agg = { + r["channel_id"]: r + for r in cur.execute( + "SELECT channel_id, COUNT(*) AS n, COALESCE(SUM(duration),0) AS dur, " + "MIN(upload_date) AS mind, MAX(upload_date) AS maxd, " + "COALESCE(SUM(view_count),0) AS views, COALESCE(SUM(like_count),0) AS likes " + "FROM videos GROUP BY channel_id" + ).fetchall() + } for ch in channels: - cid = ch["channel_id"] - row = cur.execute( - "SELECT COUNT(*) AS n, COALESCE(SUM(duration),0) AS dur, MIN(upload_date) AS mind, MAX(upload_date) AS maxd, COALESCE(SUM(view_count),0) AS views, COALESCE(SUM(like_count),0) AS likes FROM videos WHERE channel_id = ?", - (cid,), - ).fetchone() - ch["video_count_db"] = row["n"] - ch["total_duration"] = row["dur"] - ch["date_min"] = row["mind"] - ch["date_max"] = row["maxd"] - ch["total_views"] = row["views"] - ch["total_likes"] = row["likes"] + row = agg.get(ch["channel_id"]) + ch["video_count_db"] = row["n"] if row else 0 + ch["total_duration"] = row["dur"] if row else 0 + ch["date_min"] = row["mind"] if row else None + ch["date_max"] = row["maxd"] if row else None + ch["total_views"] = row["views"] if row else 0 + ch["total_likes"] = row["likes"] if row else 0 status_breakdown = { r["status"]: r["n"] for r in cur.execute("SELECT status, COUNT(*) AS n FROM videos GROUP BY status").fetchall() } - # top tags (tags is JSON array text) - tag_rows = cur.execute("SELECT tags FROM videos WHERE tags IS NOT NULL AND tags != '[]'").fetchall() - tag_counts: dict[str, int] = {} - for tr in tag_rows: - try: - for t in json.loads(tr["tags"]): - t = (t or "").strip().lower() - if t: - tag_counts[t] = tag_counts.get(t, 0) + 1 - except (json.JSONDecodeError, TypeError): - continue - top_tags = sorted(tag_counts.items(), key=lambda x: x[1], reverse=True)[:20] + # top tags (tags is JSON array text). Counted in SQL via json_each + # instead of loading every tags string into Python; json_valid + # skips rows the old json.loads/except path also skipped, and the + # TRIM/LOWER/empty-filter mirrors the per-tag normalisation. + tag_counts = { + r["tag"]: r["n"] + for r in cur.execute( + "SELECT LOWER(TRIM(je.value)) AS tag, COUNT(*) AS n " + "FROM videos v, json_each(v.tags) je " + "WHERE json_valid(v.tags) AND TRIM(je.value) <> '' " + "GROUP BY tag ORDER BY n DESC LIMIT 20" + ).fetchall() + } + top_tags = sorted(tag_counts.items(), key=lambda x: x[1], reverse=True) return { "channels": channels, "status_breakdown": status_breakdown, diff --git a/src/yt_scraper/webapp/api.py b/src/yt_scraper/webapp/api.py index 171e015..4056470 100644 --- a/src/yt_scraper/webapp/api.py +++ b/src/yt_scraper/webapp/api.py @@ -1,6 +1,7 @@ from __future__ import annotations import json +from concurrent.futures import ThreadPoolExecutor from pathlib import Path from typing import Any @@ -12,7 +13,9 @@ from .. import analysis as analysis_mod from .. import cookies as cookies_mod from .. import export as export_mod from ..config import Config +from ..discover import extract_handle from ..ratelimit import polite_sleep +from ..render import safe_dirname from .. import store as store_mod from ..store import Store @@ -21,6 +24,11 @@ from ..store import Store #: so this only bounds the eager burst that follows "Add channel". THUMBNAIL_AUTO_LIMIT = 60 +#: Thumbnail fetches go to the i.ytimg.com CDN (not youtube.com), so the +#: Pacer does not apply to them; a small pool turns ~60 sequential round +#: trips into ~10 batches without hammering the CDN. +THUMBNAIL_WORKERS = 6 + def build_router(store: Store, cfg: Config, jobs) -> APIRouter: r = APIRouter(prefix="/api") @@ -50,7 +58,7 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: avatar_cached = False if not avatar: avatar = deep_channel_avatar(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests) - store.upsert_channel(cid, _handle(url), name, len(refs), avatar=avatar) + store.upsert_channel(cid, extract_handle(url), name, len(refs), avatar=avatar) store.upsert_videos(refs) if avatar: avatars_dir = Path(cfg.output_dir_resolved).parent / "avatars" @@ -229,6 +237,31 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: cookies_mod.set_active(store, imported[0]) return {"imported": imported} + @r.post("/cookies/from-browser") + def cookies_from_browser(payload: dict): + """Extract youtube.com cookies from a local browser (default: Brave). + + Requires the browser to be fully closed; Chromium keeps its cookie + database locked while running. + """ + payload = payload or {} + browser = (payload.get("browser") or "brave").strip().lower() + profile = payload.get("profile") or None + try: + cid = cookies_mod.import_from_browser(store, browser=browser, profile=profile) + except cookies_mod.BrowserCookieLockedError as exc: + raise HTTPException(409, str(exc)) + except Exception as exc: + raise HTTPException(400, str(exc)) + # The user imported it precisely to use it. + cookies_mod.set_active(store, cid) + row = store.get_cookie(cid) + return { + "imported": [cid], "active": cid, "browser": browser, + "cookie_count": row.cookie_count if row else None, + "has_session": row.has_session if row else None, + } + @r.post("/cookies/{cookie_id}/activate") def cookies_activate(cookie_id: str): cookies_mod.set_active(store, cookie_id) @@ -289,8 +322,11 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: "database": _folder_info(Path(cfg.database_path_resolved).parent), }} + # `def`, not `async def`: same reason as add_channel — the body blocks on + # SQLite reads and a subprocess/os.startfile call, which would hold the + # event loop if this were async. @r.post("/folders/open") - async def open_folder(payload: dict): + def open_folder(payload: dict): kind = (payload or {}).get("kind") channel_id = (payload or {}).get("channel_id") data_root = Path(cfg.output_dir_resolved).parent @@ -299,7 +335,7 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: if channel_id: ch = store.get_channel(channel_id) or {} if ch.get("name"): - target = Path(cfg.output_dir_resolved) / _safe_dir(ch["name"]) + target = Path(cfg.output_dir_resolved) / safe_dirname(ch["name"]) elif kind == "exports": target = data_root / "exports" elif kind == "audio": @@ -375,6 +411,17 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: raise HTTPException(404, "markdown file missing on disk") return FileResponse(str(p), filename=p.name, media_type="text/markdown") + @r.post("/videos/{video_id}/open-markdown") + def open_video_markdown(video_id: str): + v = store.get_video(video_id) + if not v or not v.markdown_path: + raise HTTPException(404, "markdown not generated yet") + p = Path(cfg.output_dir_resolved).parent / v.markdown_path + if not p.exists(): + raise HTTPException(404, "markdown file missing on disk") + _open_in_os(p) + return {"opened": str(p)} + @r.api_route("/videos/{video_id}/audio", methods=["GET", "HEAD"]) def video_audio(video_id: str, request: Request): # audio is stored as data/audio/.mp3 (see jobs._run_audio outtmpl) @@ -423,10 +470,13 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: video_ids = video_ids[:limit] out_dir = Path(cfg.output_dir_resolved).parent / "thumbnails" out_dir.mkdir(parents=True, exist_ok=True) - n = 0 - for vid in video_ids: - if cache_thumbnail(store, vid, out_dir): - n += 1 + # CDN fetches, one per video, each independent: run them on a small + # thread pool instead of sequentially. cache_thumbnail owns its own + # SQLite connection and writes one file per id, so this is safe; + # failures still count as False exactly as before. + with ThreadPoolExecutor(max_workers=THUMBNAIL_WORKERS) as pool: + results = list(pool.map(lambda vid: cache_thumbnail(store, vid, out_dir), video_ids)) + n = sum(1 for ok in results if ok) return {"downloaded": n, "dir": str(out_dir), "skipped": skipped} @r.get("/thumbnails/{video_id}") @@ -455,8 +505,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: "permanent": counts.get("permanent", 0), } + # `def`, not `async def`: the UPDATE below blocks on SQLite; run it on the + # threadpool like the other blocking handlers. @r.post("/videos/reset") - async def reset_videos(payload: dict): + def reset_videos(payload: dict): """Send `error` / `no_subtitles` videos back to pending so they can be retried. `no_subtitles` is resettable on purpose: it is recorded whenever the @@ -620,8 +672,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: "top_tags": d["top_tags"], "totals": totals} # -------------------------------------------------- audio (job) + # `def`, not `async def`: no await here, and the SQLite reads below would + # block the event loop. @r.post("/tools/audio") - async def tools_audio(payload: dict): + def tools_audio(payload: dict): video_ids = (payload or {}).get("video_ids") or [] channel_id = (payload or {}).get("channel_id") if not video_ids and not channel_id: @@ -632,8 +686,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: job_id = jobs.enqueue(channel_id, opts) return {"job_id": job_id} + # `def`, not `async def`: same as tools_audio — SQLite reads + filesystem + # stats only, no await. @r.post("/tools/video") - async def tools_video(payload: dict): + def tools_video(payload: dict): video_ids = (payload or {}).get("video_ids") or [] if len(video_ids) != 1: raise HTTPException(400, "exactly one video_id required") @@ -652,11 +708,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter: def _video_dict(v) -> dict: - import json as _json tags = [] if v.tags: try: - tags = _json.loads(v.tags) + tags = json.loads(v.tags) except Exception: tags = [] return { @@ -710,22 +765,11 @@ def _loads(s): return {} -def _handle(url: str) -> str: - if "@" in url: - return "@" + url.split("@", 1)[1].split("/", 1)[0] - return "" - - def _folder_info(path: Path) -> dict: path = Path(path) return {"path": str(path.resolve()), "exists": path.exists()} -def _safe_dir(name: str) -> str: - safe = "".join(c for c in (name or "") if c not in r'\/:*?"<>|') - return (safe.strip().strip(".") or "unknown") - - def _open_in_os(path: Path) -> None: import os import sys diff --git a/src/yt_scraper/webapp/app.py b/src/yt_scraper/webapp/app.py index 354dbe8..b239419 100644 --- a/src/yt_scraper/webapp/app.py +++ b/src/yt_scraper/webapp/app.py @@ -1,6 +1,7 @@ from __future__ import annotations import logging +import threading from pathlib import Path from fastapi import FastAPI @@ -47,18 +48,44 @@ def create_app(cfg: Config | None = None, db_path: str | Path | None = None) -> # Reconcile, not just backfill: backfill_from_markdown populates segments # and metadata but never touches `status`/`markdown_path`, so a video whose # .md is already on disk would keep showing a failure after every restart. - try: - md_root = Path(cfg.output_dir_resolved) - if md_root.exists(): - reconcile_markdown(store, md_root, log=pkg_log.info) - except Exception: - pkg_log.exception("startup reconcile failed") + # + # It walks and parses EVERY .md in the tree, which on a large library + # blocked uvicorn boot for the whole scan — so it runs on a daemon thread + # and the server answers immediately (SQLite/WAL absorbs the concurrent + # writes; /api/tools/reconcile still re-runs it on demand). healthz + # reports whether the initial pass has finished. + reconcile_done = threading.Event() + + def _startup_reconcile() -> None: + try: + md_root = Path(cfg.output_dir_resolved) + if md_root.exists(): + reconcile_markdown(store, md_root, log=pkg_log.info) + except Exception: + pkg_log.exception("startup reconcile failed") + finally: + reconcile_done.set() + + threading.Thread(target=_startup_reconcile, name="startup-reconcile", daemon=True).start() app = FastAPI(title="yt-scraper platform", version="1.0.0") + # StaticFiles/FileResponse send etag + last-modified but no Cache-Control, + # so a browser may heuristically cache an old app.js across updates and + # run yesterday's JS against a new backend. "no-cache" forces + # revalidation on every load (cheap 304s on a local app) while keeping + # the etag benefits. + @app.middleware("http") + async def _revalidate_shell(request, call_next): + response = await call_next(request) + path = request.url.path + if path == "/" or path.startswith("/static/"): + response.headers["Cache-Control"] = "no-cache" + return response + @app.get("/healthz") def healthz(): - return {"status": "ok"} + return {"status": "ok", "reconcile_done": reconcile_done.is_set()} jobs = JobManager(store, cfg) app.state.jobs = jobs diff --git a/src/yt_scraper/webapp/jobs.py b/src/yt_scraper/webapp/jobs.py index 2d837ff..f654430 100644 --- a/src/yt_scraper/webapp/jobs.py +++ b/src/yt_scraper/webapp/jobs.py @@ -9,12 +9,12 @@ from collections import deque from pathlib import Path from typing import Any -from ..config import Config, load_config, parse_languages +from ..config import Config, keep_ref, load_config, parse_languages from ..cookies import resolve_active_path -from ..discover import discover_incremental +from ..discover import discover_incremental, extract_handle from ..pipeline import process_video from ..ratelimit import ThrottleGuard, polite_sleep -from ..store import Store, VideoRef +from ..store import Store, order_pending class JobManager: @@ -269,6 +269,7 @@ class JobManager: "sleep_interval_requests": cfg.yt_dlp.sleep_subrequests, "retries": cfg.yt_dlp.retries, "socket_timeout": 30.0, + "js_runtimes": {"node": {}, "deno": {}, "bun": {}, "quickjs": {}}, } if cfg.delay.audio_rate_limit: ydl_opts["ratelimit"] = cfg.delay.audio_rate_limit @@ -453,7 +454,7 @@ class JobManager: max_window=sync.max_window, overlap=sync.overlap, since=self.store.latest_upload_date(channel_id) if incremental else None, - keep=_keep_ref(cfg, include_shorts, no_live), + keep=keep_ref(cfg, include_shorts, no_live), ) except Exception as exc: self.store.update_job(job_id, status="error", last_error=str(exc), finished=True) @@ -478,7 +479,7 @@ class JobManager: from ..discover import deep_channel_avatar avatar = deep_channel_avatar(channel_url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests) if result.full_scan: - self.store.upsert_channel(_channel_id, _handle(channel_url), channel_name, len(refs), avatar=avatar) + self.store.upsert_channel(_channel_id, extract_handle(channel_url), channel_name, len(refs), avatar=avatar) else: self.store.update_channel_meta(_channel_id, name=channel_name, avatar=avatar) if avatar: @@ -489,8 +490,8 @@ class JobManager: # Process the freshly-seen window first, then the older backlog, so a # `limit` still means "the newest N" now that discovery stops early. - pending = _order_pending(self.store.get_pending(_channel_id), refs) - pending = [r for r in pending if _keep_ref(cfg, include_shorts, no_live)(r)] + pending = order_pending(self.store.get_pending(_channel_id), refs) + pending = [r for r in pending if keep_ref(cfg, include_shorts, no_live)(r)] if since: cutoff = since.replace("-", "") pending = [r for r in pending if not r.upload_date or r.upload_date >= cutoff] @@ -646,12 +647,12 @@ class JobManager: max_window=sync.max_window, overlap=sync.overlap, since=self.store.latest_upload_date(channel_id) if incremental else None, - keep=_keep_ref(self.cfg), + keep=keep_ref(self.cfg), ) refs = result.refs if result.full_scan: - self.store.upsert_channel(result.channel_id, _handle(channel_url), result.channel_name, len(refs)) + self.store.upsert_channel(result.channel_id, extract_handle(channel_url), result.channel_name, len(refs)) new_videos = self.store.upsert_videos(refs) else: self.store.update_channel_meta(result.channel_id, name=result.channel_name) @@ -669,41 +670,6 @@ class JobManager: } -def _keep_ref(cfg: Config, include_shorts: bool | None = None, no_live: bool | None = None): - """Predicate matching the shorts/live rules that decide what reaches the DB. - - Incremental discovery needs the same filter its stored ids were created - under, otherwise the tail of a window is full of entries that can never be - recognised as known and the window keeps widening for nothing. - """ - shorts = cfg.include_shorts if include_shorts is None else include_shorts - skip_live = (not cfg.include_live) if no_live is None else no_live - - def keep(r: Any) -> bool: # VideoRef or VideoRow — both carry `.url` - url = r.url or "" - if not shorts and "/shorts/" in url: - return False - if skip_live and url.startswith("https://www.youtube.com/live/"): - return False - return True - - return keep - - -def _order_pending(rows: list, refs: list[VideoRef]) -> list: - """Newest-window-first ordering for the pending queue. - - `get_pending` is ordered by discovery time, which used to coincide with - newest-first because discovery saw the whole channel at once. With windowed - sync that no longer holds, so put the videos from this run's window at the - front and keep the rest of the backlog behind them. - """ - by_id = {r.video_id: r for r in rows} - ordered = [by_id.pop(x.video_id) for x in refs if x.video_id in by_id] - ordered.extend(by_id.values()) - return ordered - - def _clone_config(cfg: Config) -> Config: import copy return copy.deepcopy(cfg) @@ -721,12 +687,6 @@ def _resolve_channel_url(store: Store, cfg: Config, channel_id: str | None) -> s return f"https://www.youtube.com/channel/{channel_id}" -def _handle(url: str) -> str: - if "@" in url: - return "@" + url.split("@", 1)[1].split("/", 1)[0] - return "" - - def _safe_video_filename(title: str) -> str: """Keep the displayed title while making a valid, bounded Windows name.""" invalid = set(r'\\/:*?"<>|') diff --git a/src/yt_scraper/webapp/static/app.js b/src/yt_scraper/webapp/static/app.js index 1a2542a..72f0449 100644 --- a/src/yt_scraper/webapp/static/app.js +++ b/src/yt_scraper/webapp/static/app.js @@ -92,19 +92,37 @@ const size = Number(localStorage.getItem("videos-size")); if ([10, 25, 50, 100].includes(size)) this.videos.size = size; this.hydrateURL(); + // Every view transition pushes a real history entry, so the browser's + // back/forward buttons have to drive the SPA through popstate. + window.addEventListener("popstate", () => this._onPopState()); + // Global shortcuts ("/" jumps to search) ride the same window wiring. + window.addEventListener("keydown", (e) => this._onGlobalKeydown(e)); this.checkHealth(); this.loadDashboard(); this.loadChannels(); if (this.view === "videos") this.loadVideos(this.videos.page); if (this.view === "search" && this.search.q) this.loadSearch(); + if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id, { push: false }); this.loadRetryable(); this.startLivePolling(); }, hydrateURL() { - const p = new URLSearchParams(window.location.search); + this._fromParams(new URLSearchParams(window.location.search)); + }, + + // Parse the query string into component state. Shared by the initial + // load and by popstate so browser back/forward restore the exact view + // (including filters). Resets to defaults first because popstate can go + // from a filtered list back to an unfiltered one. + _fromParams(p) { const views = ["dashboard", "channels", "videos", "detail", "search", "analysis", "scrape", "cookies", "tools", "export"]; - if (views.includes(p.get("view"))) this.view = p.get("view"); + const v = p.get("view"); + this.view = views.includes(v) ? v : "dashboard"; + if (this.view === "detail" && !p.get("video")) this.view = "videos"; + this.filters = { channel: "", status: "", from: "", to: "", min_dur: "", q: "", sort: "upload_date" }; + this.search.q = ""; + this.search.channel = ""; if (this.view === "videos") { ["channel", "status", "from", "to", "min_dur", "q", "sort"].forEach(k => { if (p.has(k)) this.filters[k] = p.get(k); }); const page = Number(p.get("page")); @@ -116,24 +134,57 @@ this.search.q = p.get("q") || ""; this.search.channel = p.get("channel") || ""; } + // A stub is enough for init/popstate to know which video to load. + this.detail.video = this.view === "detail" ? { video_id: p.get("video") } : null; + }, + + _urlParams() { + const p = new URLSearchParams({ view: this.view }); + if (this.view === "videos") { + Object.entries(this.filters).forEach(([k, v]) => { if (v) p.set(k, v); }); + p.set("page", this.videos.page); p.set("size", this.videos.size); + } else if (this.view === "search") { + if (this.search.q) p.set("q", this.search.q); + if (this.search.channel) p.set("channel", this.search.channel); + } else if (this.view === "detail" && this.detail.video) { + p.set("video", this.detail.video.video_id); + } + return p; }, syncURL() { clearTimeout(this._urlTimer); this._urlTimer = setTimeout(() => { - const p = new URLSearchParams({ view: this.view }); - if (this.view === "videos") { - Object.entries(this.filters).forEach(([k, v]) => { if (v) p.set(k, v); }); - p.set("page", this.videos.page); p.set("size", this.videos.size); - } else if (this.view === "search") { - if (this.search.q) p.set("q", this.search.q); - if (this.search.channel) p.set("channel", this.search.channel); - } - const next = "?" + p.toString(); - if (next !== window.location.search) history.replaceState(null, "", next); + const next = "?" + this._urlParams().toString(); + if (next !== window.location.search) history.replaceState(history.state, "", next); }, 120); }, + // View transitions push a history entry so browser back stays inside + // the app; param tweaks (filters, pagination) only replace via syncURL. + pushURL(extra) { + clearTimeout(this._urlTimer); + const p = this._urlParams(); + if (extra && extra.video) p.set("video", extra.video); + const next = "?" + p.toString(); + if (next === window.location.search) return; + history.pushState({ view: p.get("view") }, "", next); + this._pushedEntries = (this._pushedEntries || 0) + 1; + }, + + _onPopState() { + if (this._pushedEntries > 0) this._pushedEntries--; + this._fromParams(new URLSearchParams(window.location.search)); + if (this.view === "detail") { + const id = (this.detail.video && this.detail.video.video_id) || ""; + if (id && id !== this._detailLoadedId) this.openVideo(id, { push: false }); + } else { + this._detailLoadedId = null; + this._armScrollRestore(this.view); + this._loadView(this.view); + } + }, + toggleSidebar() { this.sidebarOpen = !this.sidebarOpen; localStorage.setItem("sidebar-open", String(this.sidebarOpen)); @@ -177,10 +228,18 @@ } }, - setView(v) { + setView(v, opts) { + const o = opts || {}; this.view = v; - this.syncURL(); + if (o.push === false) this.syncURL(); + else this.pushURL(); if (window.innerWidth < 768) this.closeSidebar(); + if (!o.skipLoad) this._loadView(v); + }, + + // Data loads that accompany entering a view. Extracted from setView so + // popstate can run them without pushing a new history entry. + _loadView(v) { // destroy stray canvases when leaving chart-bearing views if (v !== "analysis") this.destroyCharts(["topwords", "timeline"]); if (v === "dashboard") this.loadDashboard(); @@ -194,6 +253,26 @@ if (v === "tools") { this.loadFolders(); this.loadFormat(); } }, + // One-shot scroll restore, armed ONLY by the back-navigation paths + // (_onPopState and goBack's fallback). Ordinary list interactions + // (pagination, filter changes, openChannel) never set it, so they keep + // landing at the top like a fresh view. + // 'videos': the table is fetched asynchronously, and scrolling before + // the fetch resolves gets clamped to the top of an empty table — so + // the flag is consumed in loadVideos()'s finally, after the rows exist. + // 'search': popstate does not reload results (they stay in memory), so + // restore right after the view switch; nextTick lets Alpine render the + // results and the extra requestAnimationFrame waits for layout, so the + // browser does not clamp the scroll to a not-yet-painted page. + _armScrollRestore(view) { + if (view === "videos") { + this._pendingScrollRestore = "videos"; + } else if (view === "search") { + const top = (this._scrollMemory && this._scrollMemory.search) || 0; + this.$nextTick(() => requestAnimationFrame(() => window.scrollTo({ top }))); + } + }, + // ================================================================= // dashboard // ================================================================= @@ -323,14 +402,46 @@ const d = await this.api("/api/videos?" + p.toString()); this.videos = Object.assign({}, this.videos, { items: d.items, total: d.total, page: d.page, size: d.size }); } catch (e) { this.toast("Failed to load videos: " + e.message, "error"); } - finally { this.loading.videos = false; } + finally { + this.loading.videos = false; + // Back-navigation restore. It must happen here, after the fetch has + // resolved — scrolling any earlier is clamped to the top because the + // table is still empty while the request is in flight. The nextTick + // waits for Alpine to drop the transient "loading…" row (it renders + // above the data rows), so the position matches the row the user + // actually left. The flag is cleared on EVERY run, not only when it + // fired, so a stale one can never affect an unrelated loadVideos + // (pagination, filter change). + if (this._pendingScrollRestore === "videos") { + const top = (this._scrollMemory && this._scrollMemory.videos) || 0; + this.$nextTick(() => window.scrollTo({ top })); + } + this._pendingScrollRestore = null; + } }, - async openVideo(id) { + async openVideo(id, opts) { + const o = opts || {}; + // Remember where the user came from so the back button can label + // itself honestly. Skipped when restoring from history (popstate) or + // auto-refreshing after a job, where the origin must not change. + if (o.push !== false) { + if (this.view !== "detail") { + this.detail.fromView = this.view; + // Also remember how far down that view the user had scrolled, + // so returning from the detail can put them back on the same + // row instead of at the top of a reloaded table. + if (!this._scrollMemory) this._scrollMemory = {}; + this._scrollMemory[this.view] = window.scrollY; + } + this.view = "detail"; // before pushURL: the entry must say where we went + this.pushURL({ video: id }); + } this.view = "detail"; + this._detailLoadedId = id; this.loading.detail = true; this.loading.transcript = true; - this.detail = { video: null, transcript: [], hasAudio: false, audioPlaying: false, activeSeg: -1, videoError: false }; + this.detail = { video: null, transcript: [], hasAudio: false, audioPlaying: false, activeSeg: -1, videoError: false, fromView: this.detail.fromView }; try { const v = await this.api("/api/videos/" + encodeURIComponent(id)); // chapters may come embedded or be absent @@ -347,6 +458,30 @@ this.checkAudio(id); }, + // The single back affordance for the detail view: prefer real browser + // history (restores the list's filters via the URL), and fall back to + // the origin view when the detail URL was opened directly (refresh or + // shared link) and there is no in-app history to return to. + goBack() { + if ((this._pushedEntries || 0) > 0) { history.back(); return; } + const target = this.detail.fromView && this.detail.fromView !== "detail" ? this.detail.fromView : "videos"; + // No in-app history to history.back() into — the popstate handler + // never fires, so arm the same scroll restore before the fallback + // setView() triggers the view's loaders. + this._armScrollRestore(target); + this.setView(target); + }, + + get backLabel() { + const map = { + search: "Back to search results", + dashboard: "Back to dashboard", + channels: "Back to channels", + videos: "Back to videos", + }; + return map[this.detail.fromView] || "Back to videos"; + }, + async checkAudio(id) { try { const r = await fetch("/api/videos/" + encodeURIComponent(id) + "/audio", { method: "HEAD" }); @@ -384,7 +519,7 @@ openVideoPlayer() { this.detail.videoError = false; this.$nextTick(() => { - const player = this.$refs.videoPlayer; + const player = this.ref("videoPlayer"); if (player) { player.load(); player.play().catch(() => {}); } }); }, @@ -426,7 +561,7 @@ }, seekAudio(sec) { - const a = this.$refs.audioPlayer; + const a = this.ref("audioPlayer"); if (!a) return false; try { a.currentTime = sec; a.play().catch(() => {}); } catch (_) {} return true; @@ -434,7 +569,7 @@ seekTo(sec) { // seek the local audio player if available; otherwise just scroll the transcript - if (this.detail.hasAudio && this.$refs.audioPlayer) { + if (this.detail.hasAudio && this.ref("audioPlayer")) { if (this.seekAudio(sec)) return; } this.seekTranscript(sec); @@ -477,6 +612,18 @@ }, toggleSelectAll() { if (this.allSelected) this.selectNone(); else this.selectAll(); }, + // Active-filter chips: removing one must be one click, not a hunt + // through the selects — especially the channel filter set by + // openChannel(), which used to feel like a trap. + clearFilter(key) { + this.filters[key] = ""; + this.loadVideos(1); + }, + clearAllFilters() { + this.filters = { channel: "", status: "", from: "", to: "", min_dur: "", q: "", sort: "upload_date" }; + this.loadVideos(1); + }, + // ================================================================= // downloads (.md / thumbnails / audio / single) // ================================================================= @@ -502,6 +649,27 @@ this.downloadFile("/api/videos/" + encodeURIComponent(videoId) + "/markdown"); }, + async copyMd(videoId) { + try { + const res = await fetch("/api/videos/" + encodeURIComponent(videoId) + "/markdown"); + if (!res.ok) throw new Error("Could not fetch markdown file"); + const text = await res.text(); + await navigator.clipboard.writeText(text); + this.toast("Contenido .md copiado al portapapeles"); + } catch (e) { + this.toast("Error al copiar .md: " + e.message, "error"); + } + }, + + async openMd(videoId) { + try { + await this.api("/api/videos/" + encodeURIComponent(videoId) + "/open-markdown", { method: "POST" }); + this.toast("Abriendo archivo .md..."); + } catch (e) { + this.toast("Error al abrir .md: " + e.message, "error"); + } + }, + async processOne(videoId) { try { const d = await this.api("/api/scrape/video/" + encodeURIComponent(videoId), { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({}) }); @@ -681,9 +849,51 @@ if (tag === "INPUT" || tag === "TEXTAREA" || (e.target && e.target.isContentEditable)) return; if (this.stats.open) { this.closeStats(); return; } if (this.clip.open) { this.closeClip(); return; } + // Detail is the deepest layer of the app, so Esc backs out of it — + // same gesture that closes every modal. Guarded against the Esc that + // merely exits fullscreen playback. + if (this.view === "detail" && !document.fullscreenElement) { this.goBack(); return; } if (this.cookies && this.cookies.drag) { this.cookies.drag = false; return; } }, + // Alpine 3.17 mounts template x-if content under its own scope, so + // x-ref elements never reach the root component's $refs — and every + // view here lives inside a template x-if. Resolve refs from the DOM + // instead; $refs stays as the fast path for anything ever registered. + ref(name) { + return (this.$refs && this.$refs[name]) || document.querySelector('#root [x-ref="' + name + '"]'); + }, + + // "/" anywhere (outside form fields) jumps to the transcript search + // with the query input focused. Guarded like onGlobalEscape so it + // never fires while typing, composing or under a modal. + _onGlobalKeydown(e) { + if (e.key !== "/") return; + // IME composition and modifier combos belong to the browser/OS, + // not to this shortcut. + if (e.isComposing || e.ctrlKey || e.altKey || e.metaKey || e.shiftKey) return; + // Let editable fields receive a literal "/" instead of stealing it. + const el = document.activeElement; + const tag = el && el.tagName; + if (tag === "INPUT" || tag === "TEXTAREA" || tag === "SELECT" || (el && el.isContentEditable)) return; + // Modals own the keyboard while they are open. + if (this.confirmBox.open || this.stats.open || this.clip.open) return; + // preventDefault so the "/" never lands in the input we focus next. + e.preventDefault(); + if (this.view !== "search") this.setView("search"); + this._focusSearchInput(10); + }, + + // The search view is a template x-if whose content mounts a beat after + // the reactive flush, so a single $nextTick can query before the input + // exists. Retry across a few animation frames and focus as soon as it + // renders (no perceptible delay when it is already there). + _focusSearchInput(retries) { + const input = this.ref("searchInput"); + if (input) { input.focus(); return; } + if (retries > 0) requestAnimationFrame(() => this._focusSearchInput(retries - 1)); + }, + // ================================================================= // channels: pending download // ================================================================= @@ -1007,7 +1217,10 @@ // Scrape tab (where startScrape lives) refreshed nothing at all, and // the setView cache guard then kept the stale rows on navigation. this.refreshLiveState({ force: true }); - if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id); + // Refresh the open detail view after a job, without pushing a new + // history entry — Back must still return to where the user came + // from, not to a duplicate of the same video. + if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id, { push: false }); this._armAutoHide(); }); es.addEventListener("cancelled", () => { this.scrape.log.push("[cancelled]"); this.closeStream(); this.loadJobs(); this._armAutoHide(); }); @@ -1132,7 +1345,7 @@ if (!files.length) { this.toast("Drop .txt cookie files only", "error"); return; } this.uploadCookies(files); // reset the file input so the same file can be picked again - try { if (this.$refs.cookieFile) this.$refs.cookieFile.value = ""; } catch (_) {} + try { const input = this.ref("cookieFile"); if (input) input.value = ""; } catch (_) {} }, async uploadCookies(files) { @@ -1148,6 +1361,24 @@ } catch (e) { this.toast("Upload failed: " + e.message, "error"); } }, + async importBrowserCookies() { + if (this.loading.cookies) return; + this.loading.cookies = true; + try { + const d = await this.api("/api/cookies/from-browser", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ browser: "brave" }), + }); + this.toast("Imported " + (d.cookie_count || "") + " cookies from Brave and activated them"); + await this.loadCookies(); + } catch (e) { + this.toast("Brave import failed: " + e.message, "error"); + } finally { + this.loading.cookies = false; + } + }, + async activateCookie(id) { try { await this.api("/api/cookies/" + encodeURIComponent(id) + "/activate", { method: "POST" }); this.loadCookies(); } catch (e) { this.toast("Activate failed: " + e.message, "error"); } @@ -1189,7 +1420,9 @@ // ================================================================= openChannel(id) { this.filters.channel = id; - this.setView("videos"); + // skipLoad: loadVideos(1) below is the authoritative load — setView + // would trigger a second one with the stale page number. + this.setView("videos", { skipLoad: true }); this.loadVideos(1); }, diff --git a/src/yt_scraper/webapp/static/index.html b/src/yt_scraper/webapp/static/index.html index bcb8435..6d9272a 100644 --- a/src/yt_scraper/webapp/static/index.html +++ b/src/yt_scraper/webapp/static/index.html @@ -86,19 +86,23 @@
Channels
-
+ +
Videos
-
+ +
Total Views
-
+ +
Total Duration
-
+ +
@@ -185,7 +189,31 @@ - +
NameHandleVideosPendingLatest videoLast scrapedActions