feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo
Extraccion autenticada: - import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20 - extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla con sesion logueada (members-only); regex y opener cacheados - js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts - rutas de Brave multiplataforma (Windows/macOS/Linux) Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL, memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics Rendimiento: - entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota) - _rank_unranked con guarda (antes full-scan en cada arranque/import) - upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL) - thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool) - reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md), healthz expone reconcile_done - Store.transaction(): escrituras por video agrupadas (~6 commits -> 3) Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render, order_pending->store, keep_ref->config, seconds_to_ts solo en segments); re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done); fuera wrappers muertos de segments.py
This commit is contained in:
+24
-71
@@ -13,11 +13,10 @@ from rich.progress import Progress, SpinnerColumn, TextColumn, BarColumn, TaskPr
|
|||||||
from rich.table import Table
|
from rich.table import Table
|
||||||
|
|
||||||
from .config import Config, load_config, parse_languages
|
from .config import Config, load_config, parse_languages
|
||||||
from .store import Store, VideoRef, VideoRow
|
from .store import Store, order_pending
|
||||||
from .discover import discover_channel, discover_incremental
|
from .discover import discover_channel, discover_incremental, extract_handle
|
||||||
from .chapters import align_chapters, chapters_from_info, Chapter, Section
|
from .segments import seconds_to_ts
|
||||||
from .parse import Segment
|
from .render import safe_filename
|
||||||
from .render import build_filename_stem, render_markdown
|
|
||||||
from .ratelimit import ThrottleGuard, configure_global_pacer, polite_sleep
|
from .ratelimit import ThrottleGuard, configure_global_pacer, polite_sleep
|
||||||
from .pipeline import process_video
|
from .pipeline import process_video
|
||||||
from .cookies import auto_import_dir, resolve_active_path
|
from .cookies import auto_import_dir, resolve_active_path
|
||||||
@@ -157,7 +156,7 @@ def _run_scrape(obj, limit, since, languages, no_auto, no_shorts, include_shorts
|
|||||||
|
|
||||||
channel_id, channel_name, refs = result.channel_id, result.channel_name, result.refs
|
channel_id, channel_name, refs = result.channel_id, result.channel_name, result.refs
|
||||||
if result.full_scan:
|
if result.full_scan:
|
||||||
store.upsert_channel(channel_id, _extract_handle(cfg.channel_url), channel_name, len(refs))
|
store.upsert_channel(channel_id, extract_handle(cfg.channel_url), channel_name, len(refs))
|
||||||
else:
|
else:
|
||||||
store.update_channel_meta(channel_id, name=channel_name)
|
store.update_channel_meta(channel_id, name=channel_name)
|
||||||
console.print(
|
console.print(
|
||||||
@@ -190,7 +189,7 @@ def _run_scrape(obj, limit, since, languages, no_auto, no_shorts, include_shorts
|
|||||||
else:
|
else:
|
||||||
# Discovery only saw the newest slice; keep the older backlog reachable
|
# Discovery only saw the newest slice; keep the older backlog reachable
|
||||||
# but put this run's videos first so --limit still means "los mas nuevos".
|
# but put this run's videos first so --limit still means "los mas nuevos".
|
||||||
pending = _order_pending(pending, refs)
|
pending = order_pending(pending, refs)
|
||||||
if limit:
|
if limit:
|
||||||
pending = pending[:limit]
|
pending = pending[:limit]
|
||||||
if not pending:
|
if not pending:
|
||||||
@@ -244,7 +243,7 @@ def _known_channel_for(store: Store, channel_url: str) -> str | None:
|
|||||||
"""
|
"""
|
||||||
if not channel_url:
|
if not channel_url:
|
||||||
return None
|
return None
|
||||||
handle = _extract_handle(channel_url).lstrip("@").lower()
|
handle = extract_handle(channel_url).lstrip("@").lower()
|
||||||
tail = channel_url.rstrip("/").split("/")[-1]
|
tail = channel_url.rstrip("/").split("/")[-1]
|
||||||
for ch in store.list_channels():
|
for ch in store.list_channels():
|
||||||
cid = ch.get("channel_id") or ""
|
cid = ch.get("channel_id") or ""
|
||||||
@@ -256,14 +255,6 @@ def _known_channel_for(store: Store, channel_url: str) -> str | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _order_pending(rows: list, refs: list) -> list:
|
|
||||||
"""Videos from this run's window first, then the rest of the backlog."""
|
|
||||||
by_id = {r.video_id: r for r in rows}
|
|
||||||
ordered = [by_id.pop(r.video_id) for r in refs if r.video_id in by_id]
|
|
||||||
ordered.extend(by_id.values())
|
|
||||||
return ordered
|
|
||||||
|
|
||||||
|
|
||||||
def _print_dry_run(refs):
|
def _print_dry_run(refs):
|
||||||
table = Table(show_lines=False)
|
table = Table(show_lines=False)
|
||||||
table.add_column("Fecha", style="dim")
|
table.add_column("Fecha", style="dim")
|
||||||
@@ -388,7 +379,13 @@ def export_cmd(obj, fmt, out_dir):
|
|||||||
def audio_cmd(obj, limit, since, force):
|
def audio_cmd(obj, limit, since, force):
|
||||||
"""Descarga audio MP3 (requiere ffmpeg)."""
|
"""Descarga audio MP3 (requiere ffmpeg)."""
|
||||||
if not shutil.which("ffmpeg"):
|
if not shutil.which("ffmpeg"):
|
||||||
console.print("[red]ffmpeg no encontrado.[/red] Instala: winget install ffmpeg")
|
if sys.platform == "win32":
|
||||||
|
hint = "winget install ffmpeg (o choco install ffmpeg)"
|
||||||
|
elif sys.platform == "darwin":
|
||||||
|
hint = "brew install ffmpeg"
|
||||||
|
else:
|
||||||
|
hint = "apt install ffmpeg (Debian/Ubuntu) / dnf install ffmpeg (Fedora) / pacman -S ffmpeg (Arch)"
|
||||||
|
console.print(f"[red]ffmpeg no encontrado.[/red] Instala: {hint}")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
store: Store = obj.store
|
store: Store = obj.store
|
||||||
cfg: Config = obj.cfg
|
cfg: Config = obj.cfg
|
||||||
@@ -408,6 +405,7 @@ def audio_cmd(obj, limit, since, force):
|
|||||||
"outtmpl": str(out_dir / "%(title)s.%(ext)s"),
|
"outtmpl": str(out_dir / "%(title)s.%(ext)s"),
|
||||||
"postprocessors": [{"key": "FFmpegExtractAudio", "preferredcodec": "mp3", "preferredquality": "128"}],
|
"postprocessors": [{"key": "FFmpegExtractAudio", "preferredcodec": "mp3", "preferredquality": "128"}],
|
||||||
"quiet": True, "no_warnings": True, "noprogress": True,
|
"quiet": True, "no_warnings": True, "noprogress": True,
|
||||||
|
"js_runtimes": {"node": {}, "deno": {}, "bun": {}, "quickjs": {}},
|
||||||
}
|
}
|
||||||
if cookie_path:
|
if cookie_path:
|
||||||
ydl_opts["cookiefile"] = cookie_path
|
ydl_opts["cookiefile"] = cookie_path
|
||||||
@@ -415,7 +413,7 @@ def audio_cmd(obj, limit, since, force):
|
|||||||
ydl_opts["cookiesfrombrowser"] = (obj.cookies_from_browser,)
|
ydl_opts["cookiesfrombrowser"] = (obj.cookies_from_browser,)
|
||||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||||
for v in videos:
|
for v in videos:
|
||||||
target = out_dir / f"{_safe_filename(v.title or v.video_id)}.mp3"
|
target = out_dir / f"{safe_filename(v.title or v.video_id)}.mp3"
|
||||||
if target.exists() and not force:
|
if target.exists() and not force:
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
@@ -454,7 +452,7 @@ def channels_add(obj, url):
|
|||||||
cfg.channel_url = url
|
cfg.channel_url = url
|
||||||
console.print("[cyan]Resolviendo canal...[/cyan]")
|
console.print("[cyan]Resolviendo canal...[/cyan]")
|
||||||
channel_id, name, avatar, refs = discover_channel(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
|
channel_id, name, avatar, refs = discover_channel(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
|
||||||
obj.store.upsert_channel(channel_id, _extract_handle(url), name, len(refs), avatar=avatar)
|
obj.store.upsert_channel(channel_id, extract_handle(url), name, len(refs), avatar=avatar)
|
||||||
obj.store.upsert_videos(refs)
|
obj.store.upsert_videos(refs)
|
||||||
console.print(f"[green]Added:[/green] {name} ({channel_id}) — {len(refs)} videos")
|
console.print(f"[green]Added:[/green] {name} ({channel_id}) — {len(refs)} videos")
|
||||||
|
|
||||||
@@ -549,6 +547,7 @@ def re_render_cmd(obj, backfill):
|
|||||||
"""Regenerar Markdown desde segmentos almacenados."""
|
"""Regenerar Markdown desde segmentos almacenados."""
|
||||||
store: Store = obj.store
|
store: Store = obj.store
|
||||||
cfg: Config = obj.cfg
|
cfg: Config = obj.cfg
|
||||||
|
from .pipeline import re_render_videos
|
||||||
from .segments import backfill_from_markdown
|
from .segments import backfill_from_markdown
|
||||||
if backfill:
|
if backfill:
|
||||||
md_root = Path(cfg.output_dir_resolved)
|
md_root = Path(cfg.output_dir_resolved)
|
||||||
@@ -559,26 +558,13 @@ def re_render_cmd(obj, backfill):
|
|||||||
if not videos:
|
if not videos:
|
||||||
console.print("[yellow]No hay videos con segments_json. Usa --backfill.[/yellow]")
|
console.print("[yellow]No hay videos con segments_json. Usa --backfill.[/yellow]")
|
||||||
return
|
return
|
||||||
import json
|
|
||||||
from .render import render_markdown
|
|
||||||
console.print(f"[cyan]Re-renderizando {len(videos)} videos...[/cyan]")
|
console.print(f"[cyan]Re-renderizando {len(videos)} videos...[/cyan]")
|
||||||
for v in videos:
|
# Delegate to the pipeline implementation instead of a local loop: it
|
||||||
segs = [Segment(start=s["start"], end=s["end"], text=s["text"]) for s in json.loads(v.segments_json)]
|
# retires the .md a re-render supersedes (renamed titles used to leave
|
||||||
chapters = [Chapter(title=c["title"], start_time=c["start"], end_time=c.get("end", c["start"])) for c in json.loads(v.chapters_json or "[]")]
|
# orphans), re-points markdown_path via mark_done and fills channel_name
|
||||||
sections = align_chapters(segs, chapters)
|
# from the channels table.
|
||||||
context = {
|
n = sum(re_render_videos(store, cfg, cid) for cid in targets)
|
||||||
"video_id": v.video_id, "title": v.title or v.video_id, "channel_name": "",
|
console.print(f"[green]Re-render completo:[/green] {n} videos")
|
||||||
"channel_id": v.channel_id, "channel_url": "", "upload_date": v.upload_date or "",
|
|
||||||
"duration": v.duration or 0, "url": v.url, "transcript_lang": v.transcript_lang or "",
|
|
||||||
"transcript_src": v.transcript_src or "", "view_count": v.view_count, "like_count": v.like_count,
|
|
||||||
"tags": _parse_tags(v.tags), "thumbnail": v.thumbnail or "", "description": v.description or "",
|
|
||||||
"sections": sections,
|
|
||||||
}
|
|
||||||
stem = build_filename_stem(v.upload_date, v.title or v.video_id, cfg.filename_template)
|
|
||||||
ch = store.get_channel(v.channel_id)
|
|
||||||
out_subdir = Path(cfg.output_dir_resolved) / _safe_dirname((ch or {}).get("name") or "unknown")
|
|
||||||
render_markdown(cfg.template_path_resolved, out_subdir, stem, context)
|
|
||||||
console.print(f"[green]Re-render completo.[/green]")
|
|
||||||
|
|
||||||
|
|
||||||
# --------------------------------------------------------------------------- helpers
|
# --------------------------------------------------------------------------- helpers
|
||||||
@@ -612,12 +598,6 @@ def _title_for(store: Store, video_id: str) -> str | None:
|
|||||||
return v.title if v else None
|
return v.title if v else None
|
||||||
|
|
||||||
|
|
||||||
def _extract_handle(url: str) -> str:
|
|
||||||
if "@" in url:
|
|
||||||
return "@" + url.split("@", 1)[1].split("/", 1)[0]
|
|
||||||
return ""
|
|
||||||
|
|
||||||
|
|
||||||
def _fmt_date(d: str | None) -> str:
|
def _fmt_date(d: str | None) -> str:
|
||||||
if not d:
|
if not d:
|
||||||
return ""
|
return ""
|
||||||
@@ -634,33 +614,6 @@ def _fmt_duration(seconds: int | None) -> str:
|
|||||||
return f"{h}:{m:02d}:{s:02d}" if h else f"{m}:{s:02d}"
|
return f"{h}:{m:02d}:{s:02d}" if h else f"{m}:{s:02d}"
|
||||||
|
|
||||||
|
|
||||||
def seconds_to_ts(sec: float) -> str:
|
|
||||||
total = int(sec)
|
|
||||||
h, rem = divmod(total, 3600)
|
|
||||||
m, s = divmod(rem, 60)
|
|
||||||
return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}"
|
|
||||||
|
|
||||||
|
|
||||||
def _safe_dirname(name: str) -> str:
|
|
||||||
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
|
|
||||||
return safe.strip().strip(".") or "unknown"
|
|
||||||
|
|
||||||
|
|
||||||
def _safe_filename(name: str) -> str:
|
|
||||||
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
|
|
||||||
return safe.strip().strip(".") or "untitled"
|
|
||||||
|
|
||||||
|
|
||||||
def _parse_tags(tags_json: str | None) -> list[str]:
|
|
||||||
if not tags_json:
|
|
||||||
return []
|
|
||||||
import json
|
|
||||||
try:
|
|
||||||
return json.loads(tags_json)
|
|
||||||
except (json.JSONDecodeError, TypeError):
|
|
||||||
return []
|
|
||||||
|
|
||||||
|
|
||||||
def _parse_interval(s: str) -> float:
|
def _parse_interval(s: str) -> float:
|
||||||
s = s.strip().lower()
|
s = s.strip().lower()
|
||||||
if s.endswith("s"):
|
if s.endswith("s"):
|
||||||
|
|||||||
@@ -105,6 +105,30 @@ class Config:
|
|||||||
LANGUAGE_MODES = ("manual", "auto", "any")
|
LANGUAGE_MODES = ("manual", "auto", "any")
|
||||||
|
|
||||||
|
|
||||||
|
def keep_ref(cfg: Config, include_shorts: bool | None = None, no_live: bool | None = None):
|
||||||
|
"""Predicate matching the shorts/live rules that decide what reaches the DB.
|
||||||
|
|
||||||
|
Lives next to the `Config` flags it reads so the policy has one home.
|
||||||
|
Incremental discovery needs the same filter its stored ids were created
|
||||||
|
under, otherwise the tail of a window is full of entries that can never be
|
||||||
|
recognised as known and the window keeps widening for nothing. The optional
|
||||||
|
overrides serve the job runner, whose per-run opts may disagree with the
|
||||||
|
config the store was populated under.
|
||||||
|
"""
|
||||||
|
shorts = cfg.include_shorts if include_shorts is None else include_shorts
|
||||||
|
skip_live = (not cfg.include_live) if no_live is None else no_live
|
||||||
|
|
||||||
|
def keep(r) -> bool: # VideoRef or VideoRow — both carry `.url`
|
||||||
|
url = r.url or ""
|
||||||
|
if not shorts and "/shorts/" in url:
|
||||||
|
return False
|
||||||
|
if skip_live and url.startswith("https://www.youtube.com/live/"):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
return keep
|
||||||
|
|
||||||
|
|
||||||
def parse_languages(raw: Any, prefer_manual: bool) -> dict[str, str]:
|
def parse_languages(raw: Any, prefer_manual: bool) -> dict[str, str]:
|
||||||
"""Normalise legacy list / new dict / None into ``{lang: mode}``."""
|
"""Normalise legacy list / new dict / None into ``{lang: mode}``."""
|
||||||
default = "manual" if prefer_manual else "auto"
|
default = "manual" if prefer_manual else "auto"
|
||||||
|
|||||||
+238
-13
@@ -1,6 +1,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
import uuid
|
import uuid
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -10,7 +11,12 @@ from .store import CookieRow, Store
|
|||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# A logged-in YouTube session always sets this core set. A lone
|
||||||
|
# `__Secure-3PSID` (partial extension export) is NOT a session — YouTube
|
||||||
|
# treats the request as anonymous and members-only content stays locked,
|
||||||
|
# which is how an "active membership" cookie once failed invisibly.
|
||||||
SESSION_COOKIE_NAMES = {"SID", "SAPISID", "__Secure-3PSID", "SSID", "LOGIN_INFO", "HSID", "APISID"}
|
SESSION_COOKIE_NAMES = {"SID", "SAPISID", "__Secure-3PSID", "SSID", "LOGIN_INFO", "HSID", "APISID"}
|
||||||
|
FULL_SESSION_NAMES = {"SID", "HSID", "SSID"}
|
||||||
|
|
||||||
_DEFAULT_DIR = Path("cookies")
|
_DEFAULT_DIR = Path("cookies")
|
||||||
|
|
||||||
@@ -28,10 +34,16 @@ def parse_netscape(text: str) -> tuple[bool, dict]:
|
|||||||
"""
|
"""
|
||||||
lines = text.splitlines()
|
lines = text.splitlines()
|
||||||
expiries: list[int] = []
|
expiries: list[int] = []
|
||||||
|
session_expiries: list[int] = []
|
||||||
names: set[str] = set()
|
names: set[str] = set()
|
||||||
count = 0
|
count = 0
|
||||||
for line in lines:
|
for line in lines:
|
||||||
line = line.rstrip("\n")
|
line = line.rstrip("\n")
|
||||||
|
# "#HttpOnly_" is a data prefix, not a comment — and the session
|
||||||
|
# cookies themselves (SID, HSID, ...) are HttpOnly, so skipping
|
||||||
|
# these lines would silently strip the login out of an export.
|
||||||
|
if line.startswith("#HttpOnly_"):
|
||||||
|
line = line[len("#HttpOnly_"):]
|
||||||
if not line.strip() or line.startswith("#"):
|
if not line.strip() or line.startswith("#"):
|
||||||
continue
|
continue
|
||||||
parts = line.split("\t")
|
parts = line.split("\t")
|
||||||
@@ -48,14 +60,20 @@ def parse_netscape(text: str) -> tuple[bool, dict]:
|
|||||||
count += 1
|
count += 1
|
||||||
if expiry:
|
if expiry:
|
||||||
expiries.append(expiry)
|
expiries.append(expiry)
|
||||||
|
if name in SESSION_COOKIE_NAMES:
|
||||||
|
session_expiries.append(expiry)
|
||||||
if count == 0:
|
if count == 0:
|
||||||
return False, {"count": 0, "has_session": False, "expires_at": None, "names": set()}
|
return False, {"count": 0, "has_session": False, "expires_at": None, "names": set()}
|
||||||
has_session = bool(names & SESSION_COOKIE_NAMES)
|
has_session = FULL_SESSION_NAMES <= names or "LOGIN_INFO" in names
|
||||||
|
# What the user needs to know is when the LOGIN dies, not when the
|
||||||
|
# earliest throwaway cookie (YSC and friends live hours) lapses — so
|
||||||
|
# report the session cookies' own expiry, falling back to the file max.
|
||||||
|
relevant = session_expiries or expiries
|
||||||
expires_at = None
|
expires_at = None
|
||||||
if expiries:
|
if relevant:
|
||||||
earliest = min(expiries)
|
latest = max(relevant)
|
||||||
if earliest > 0:
|
if latest > 0:
|
||||||
expires_at = datetime.fromtimestamp(earliest, tz=timezone.utc).isoformat(timespec="seconds")
|
expires_at = datetime.fromtimestamp(latest, tz=timezone.utc).isoformat(timespec="seconds")
|
||||||
return True, {"count": count, "has_session": has_session, "expires_at": expires_at, "names": names}
|
return True, {"count": count, "has_session": has_session, "expires_at": expires_at, "names": names}
|
||||||
|
|
||||||
|
|
||||||
@@ -99,10 +117,15 @@ def import_text(store: Store, text: str, label: str, cookie_dir: str | Path | No
|
|||||||
def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int:
|
def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int:
|
||||||
"""Import any loose .txt Netscape files in dir that aren't tracked yet. Returns count."""
|
"""Import any loose .txt Netscape files in dir that aren't tracked yet. Returns count."""
|
||||||
cdir = cookies_dir(dir_path)
|
cdir = cookies_dir(dir_path)
|
||||||
tracked = {c.filename for c in store.list_cookies()}
|
tracked = store.list_cookies()
|
||||||
|
for c in tracked:
|
||||||
|
if not (cdir / c.filename).exists():
|
||||||
|
store.delete_cookie(c.id)
|
||||||
|
|
||||||
|
tracked_filenames = {c.filename for c in store.list_cookies()}
|
||||||
n = 0
|
n = 0
|
||||||
for f in sorted(cdir.glob("*.txt")):
|
for f in sorted(cdir.glob("*.txt")):
|
||||||
if f.name in tracked:
|
if f.name in tracked_filenames:
|
||||||
continue
|
continue
|
||||||
ok, info = parse_netscape_file(f)
|
ok, info = parse_netscape_file(f)
|
||||||
if not ok:
|
if not ok:
|
||||||
@@ -118,14 +141,210 @@ def auto_import_dir(store: Store, dir_path: str | Path | None = None) -> int:
|
|||||||
cookie_count=info["count"],
|
cookie_count=info["count"],
|
||||||
)
|
)
|
||||||
n += 1
|
n += 1
|
||||||
# activate first cookie if none active
|
# activate first cookie if none active or active file missing
|
||||||
if not store.get_active_cookie():
|
active = store.get_active_cookie()
|
||||||
|
if not active or not (cdir / active.filename).exists():
|
||||||
cookies = store.list_cookies()
|
cookies = store.list_cookies()
|
||||||
if cookies:
|
if cookies:
|
||||||
store.set_active_cookie(cookies[0].id)
|
store.set_active_cookie(cookies[0].id)
|
||||||
return n
|
return n
|
||||||
|
|
||||||
|
|
||||||
|
class BrowserCookieLockedError(RuntimeError):
|
||||||
|
"""The browser's cookie database could not be read — it is running.
|
||||||
|
|
||||||
|
Chromium opens its Cookies file without sharing read access, so the
|
||||||
|
browser (including its background/tray processes) must be fully closed
|
||||||
|
for the extraction to succeed.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
# Where the Brave executable lives on a default install, in preference
|
||||||
|
# order. The %VAR% placeholders only expand on Windows (elsewhere they stay
|
||||||
|
# literal and simply never match), so one flat cross-platform list works.
|
||||||
|
_BRAVE_EXE_CANDIDATES = (
|
||||||
|
# Windows
|
||||||
|
r"%ProgramFiles%\BraveSoftware\Brave-Browser\Application\brave.exe",
|
||||||
|
r"%LocalAppData%\BraveSoftware\Brave-Browser\Application\brave.exe",
|
||||||
|
# macOS (system-wide Applications and per-user ~/Applications)
|
||||||
|
"/Applications/Brave Browser.app/Contents/MacOS/Brave Browser",
|
||||||
|
"~/Applications/Brave Browser.app/Contents/MacOS/Brave Browser",
|
||||||
|
# Linux (official deb/rpm packages, distro builds, snap, manual installs)
|
||||||
|
"/usr/bin/brave-browser",
|
||||||
|
"/usr/bin/brave",
|
||||||
|
"/opt/brave.com/brave/brave-browser",
|
||||||
|
"/snap/bin/brave",
|
||||||
|
"/usr/local/bin/brave-browser",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Default profile ("User Data") directories per OS — same flat-list trick:
|
||||||
|
# only the one for the current OS exists, the rest never match.
|
||||||
|
_BRAVE_USER_DATA_CANDIDATES = (
|
||||||
|
r"%LocalAppData%\BraveSoftware\Brave-Browser\User Data", # Windows
|
||||||
|
"~/Library/Application Support/BraveSoftware/Brave-Browser", # macOS
|
||||||
|
"~/.config/BraveSoftware/Brave-Browser", # Linux
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _expand_path(candidate: str) -> str:
|
||||||
|
"""Expand %VAR% (Windows) and ~ (macOS/Linux) in a path candidate."""
|
||||||
|
return os.path.expanduser(os.path.expandvars(candidate))
|
||||||
|
|
||||||
|
|
||||||
|
def _find_brave() -> tuple[str | None, str | None]:
|
||||||
|
"""(exe_path, user_data_dir) for a default Brave install on Windows/macOS/Linux."""
|
||||||
|
exe = next(
|
||||||
|
(p for p in (_expand_path(c) for c in _BRAVE_EXE_CANDIDATES) if os.path.isfile(p)),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
ud = next(
|
||||||
|
(p for p in (_expand_path(c) for c in _BRAVE_USER_DATA_CANDIDATES) if os.path.isdir(p)),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
return exe, ud
|
||||||
|
|
||||||
|
|
||||||
|
def _cdp_cookie(cookie: dict):
|
||||||
|
"""DevTools cookie dict -> http.cookiejar.Cookie."""
|
||||||
|
import http.cookiejar
|
||||||
|
domain = cookie.get("domain") or ".youtube.com"
|
||||||
|
http_only = bool(cookie.get("httpOnly"))
|
||||||
|
return http.cookiejar.Cookie(
|
||||||
|
version=0, name=cookie["name"], value=cookie.get("value") or "",
|
||||||
|
port=None, port_specified=False,
|
||||||
|
domain=domain, domain_specified=True, domain_initial_dot=domain.startswith("."),
|
||||||
|
path=cookie.get("path") or "/", path_specified=True,
|
||||||
|
secure=bool(cookie.get("secure")),
|
||||||
|
expires=int(cookie.get("expires")) if cookie.get("expires") else None,
|
||||||
|
discard=False, comment=None, comment_url=None,
|
||||||
|
rest={"HttpOnly": None} if http_only else {},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_brave_cdp(timeout: float = 45.0) -> list:
|
||||||
|
"""Launch Brave headless on its REAL profile and read decrypted cookies
|
||||||
|
over DevTools.
|
||||||
|
|
||||||
|
Chromium 127+ encrypts new cookies "app-bound" (v20): only the browser
|
||||||
|
itself can decrypt them, which is why yt-dlp's file-based extraction
|
||||||
|
dies with "Failed to decrypt with DPAPI". Launching the browser with
|
||||||
|
its user-data-dir passed EXPLICITLY on the command line keeps remote
|
||||||
|
debugging allowed (Chromium 136+ blocks it for the implicit default
|
||||||
|
dir), and the browser hands its own cookies over in plaintext. The
|
||||||
|
browser must still be closed — the profile is single-writer.
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import socket
|
||||||
|
import subprocess
|
||||||
|
import time
|
||||||
|
import urllib.request
|
||||||
|
|
||||||
|
exe, user_data = _find_brave()
|
||||||
|
if not exe or not user_data:
|
||||||
|
raise RuntimeError(
|
||||||
|
"Brave installation not found (looked in the default install "
|
||||||
|
"locations for Windows, macOS and Linux)"
|
||||||
|
)
|
||||||
|
|
||||||
|
with socket.socket() as s:
|
||||||
|
s.bind(("127.0.0.1", 0))
|
||||||
|
port = s.getsockname()[1]
|
||||||
|
|
||||||
|
proc = subprocess.Popen(
|
||||||
|
[
|
||||||
|
exe, "--headless=new", f"--remote-debugging-port={port}",
|
||||||
|
f"--user-data-dir={user_data}", "--no-first-run",
|
||||||
|
"--no-default-browser-check", "about:blank",
|
||||||
|
],
|
||||||
|
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
deadline = time.monotonic() + timeout
|
||||||
|
version = None
|
||||||
|
while time.monotonic() < deadline:
|
||||||
|
try:
|
||||||
|
version = json.load(urllib.request.urlopen(
|
||||||
|
f"http://127.0.0.1:{port}/json/version", timeout=2))
|
||||||
|
break
|
||||||
|
except Exception:
|
||||||
|
time.sleep(0.5)
|
||||||
|
if version is None:
|
||||||
|
# Most common cause: Brave is already running and the new
|
||||||
|
# process just delegated to it without opening a debug port.
|
||||||
|
raise BrowserCookieLockedError(
|
||||||
|
"could not attach to Brave — close Brave completely "
|
||||||
|
"(including background processes) and try again"
|
||||||
|
)
|
||||||
|
|
||||||
|
from websockets.sync.client import connect
|
||||||
|
with connect(version["webSocketDebuggerUrl"], open_timeout=10) as ws:
|
||||||
|
ws.send(json.dumps({"id": 1, "method": "Storage.getCookies"}))
|
||||||
|
reply = json.loads(ws.recv())
|
||||||
|
return [_cdp_cookie(c) for c in reply.get("result", {}).get("cookies", [])]
|
||||||
|
finally:
|
||||||
|
proc.kill()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
def import_from_browser(
|
||||||
|
store: Store,
|
||||||
|
browser: str = "brave",
|
||||||
|
profile: str | None = None,
|
||||||
|
label: str | None = None,
|
||||||
|
cookie_dir: str | Path | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Pull youtube.com cookies straight from a local browser profile.
|
||||||
|
|
||||||
|
Tries yt-dlp's file-based extraction first; on Chromium 127+ profiles
|
||||||
|
whose cookies are app-bound (v20 — "Failed to decrypt with DPAPI"),
|
||||||
|
falls back to launching the browser itself headless and reading the
|
||||||
|
decrypted cookies over DevTools. Only youtube.com cookies are kept and
|
||||||
|
stored in the vault as a normal Netscape file, so activation, expiry
|
||||||
|
tracking and every extraction path work unchanged.
|
||||||
|
"""
|
||||||
|
from yt_dlp.cookies import extract_cookies_from_browser
|
||||||
|
|
||||||
|
cookies = None
|
||||||
|
try:
|
||||||
|
cookies = extract_cookies_from_browser(browser, profile or None)
|
||||||
|
except Exception as exc:
|
||||||
|
msg = str(exc)
|
||||||
|
if "Could not copy Chrome cookie database" in msg or "database is locked" in msg:
|
||||||
|
raise BrowserCookieLockedError(
|
||||||
|
f"could not read {browser}'s cookie database — close {browser} completely "
|
||||||
|
"(including background processes) and try again"
|
||||||
|
) from exc
|
||||||
|
# App-bound (v20) cookies: only the browser can decrypt them.
|
||||||
|
if browser == "brave" and ("decrypt with DPAPI" in msg or "decrypt" in msg.lower()):
|
||||||
|
log.info("Brave cookies are app-bound; falling back to DevTools extraction")
|
||||||
|
cookies = _extract_brave_cdp()
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
|
||||||
|
lines = []
|
||||||
|
for c in cookies:
|
||||||
|
if "youtube.com" not in (c.domain or ""):
|
||||||
|
continue
|
||||||
|
# Netscape format; the #HttpOnly_ prefix is stripped again on parse.
|
||||||
|
prefix = "#HttpOnly_" if c.has_nonstandard_attr("httponly") else ""
|
||||||
|
domain = c.domain or ".youtube.com"
|
||||||
|
include_subdomains = "TRUE" if domain.startswith(".") else "FALSE"
|
||||||
|
expiry = int(c.expires) if c.expires else 0
|
||||||
|
lines.append(
|
||||||
|
f"{prefix}{domain}\t{include_subdomains}\t{c.path or '/'}\t"
|
||||||
|
f"{'TRUE' if c.secure else 'FALSE'}\t{expiry}\t{c.name}\t{c.value}"
|
||||||
|
)
|
||||||
|
if not lines:
|
||||||
|
raise ValueError(f"no youtube.com cookies found in {browser} (not logged in?)")
|
||||||
|
|
||||||
|
text = (
|
||||||
|
"# Netscape HTTP Cookie File\n"
|
||||||
|
f"# Extracted from {browser} profile {profile or 'default'}\n"
|
||||||
|
+ "\n".join(lines) + "\n"
|
||||||
|
)
|
||||||
|
return import_text(store, text, label=label or f"{browser}", cookie_dir=cookie_dir)
|
||||||
|
|
||||||
|
|
||||||
def set_active(store: Store, cookie_id: str) -> None:
|
def set_active(store: Store, cookie_id: str) -> None:
|
||||||
store.set_active_cookie(cookie_id)
|
store.set_active_cookie(cookie_id)
|
||||||
|
|
||||||
@@ -143,12 +362,18 @@ def delete(store: Store, cookie_id: str, cookie_dir: str | Path | None = None) -
|
|||||||
|
|
||||||
|
|
||||||
def resolve_active_path(store: Store, cookie_dir: str | Path | None = None) -> str | None:
|
def resolve_active_path(store: Store, cookie_dir: str | Path | None = None) -> str | None:
|
||||||
row = store.get_active_cookie()
|
|
||||||
if not row:
|
|
||||||
return None
|
|
||||||
cdir = cookies_dir(cookie_dir)
|
cdir = cookies_dir(cookie_dir)
|
||||||
|
row = store.get_active_cookie()
|
||||||
|
if row:
|
||||||
path = cdir / row.filename
|
path = cdir / row.filename
|
||||||
return str(path) if path.exists() else None
|
if path.exists():
|
||||||
|
return str(path)
|
||||||
|
for c in store.list_cookies():
|
||||||
|
path = cdir / c.filename
|
||||||
|
if path.exists():
|
||||||
|
store.set_active_cookie(c.id)
|
||||||
|
return str(path)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def is_expired(row: CookieRow) -> bool:
|
def is_expired(row: CookieRow) -> bool:
|
||||||
|
|||||||
@@ -211,6 +211,17 @@ def discover_incremental(
|
|||||||
size = min(size * 2, max_window)
|
size = min(size * 2, max_window)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_handle(url: str) -> str:
|
||||||
|
"""@handle embedded in a channel URL, or "" for /channel/<id> URLs.
|
||||||
|
|
||||||
|
Every caller that records a channel (CLI, webapp add-channel, jobs, watch)
|
||||||
|
normalises the handle the same way; this is that one shared definition.
|
||||||
|
"""
|
||||||
|
if "@" in url:
|
||||||
|
return "@" + url.split("@", 1)[1].split("/", 1)[0]
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
def _pick_channel_avatar(info: dict[str, Any]) -> str | None:
|
def _pick_channel_avatar(info: dict[str, Any]) -> str | None:
|
||||||
"""Best-effort channel avatar URL from a yt-dlp channel info dict.
|
"""Best-effort channel avatar URL from a yt-dlp channel info dict.
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,12 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import urllib.request
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
from typing import Any, Mapping
|
from typing import Any, Mapping
|
||||||
|
|
||||||
import yt_dlp
|
import yt_dlp
|
||||||
@@ -100,8 +105,20 @@ def extract_video(
|
|||||||
# One video extraction is two requests: the watch page and the InnerTube
|
# One video extraction is two requests: the watch page and the InnerTube
|
||||||
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
|
# player call. yt-dlp spaces them itself; the pacer needs to know they exist.
|
||||||
GLOBAL_PACER.wait(cost=2)
|
GLOBAL_PACER.wait(cost=2)
|
||||||
|
try:
|
||||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||||
info = ydl.extract_info(video_url, download=False)
|
info = ydl.extract_info(video_url, download=False)
|
||||||
|
except Exception:
|
||||||
|
# Logged-in sessions on current YouTube increasingly end here
|
||||||
|
# ("The page needs to be reloaded" / format-availability failures).
|
||||||
|
# The session itself is usually fine — the watch page still hands
|
||||||
|
# metadata and caption tracks to a plain cookie'd GET — so try that
|
||||||
|
# before giving up. Without cookies there is nothing to fall back to.
|
||||||
|
if cookies_file:
|
||||||
|
fallback = extract_via_watch_page(video_url, cookies_file, languages_dict, prefer_manual)
|
||||||
|
if fallback is not None:
|
||||||
|
return fallback
|
||||||
|
raise
|
||||||
|
|
||||||
pick = pick_subtitle(info, languages_dict, prefer_manual)
|
pick = pick_subtitle(info, languages_dict, prefer_manual)
|
||||||
segments: list[Segment] = []
|
segments: list[Segment] = []
|
||||||
@@ -128,6 +145,131 @@ def extract_video(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
_WATCH_PAGE_UA = (
|
||||||
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||||
|
"(KHTML, like Gecko) Chrome/152.0.7977.64 Safari/537.36"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Precompiled: the fallback can fire on every video of a throttled batch, and
|
||||||
|
# re-compiling per call showed up under those runs.
|
||||||
|
_INITIAL_PLAYER_RE = re.compile(r"ytInitialPlayerResponse\s*=\s*(\{.+?\})\s*;")
|
||||||
|
|
||||||
|
# One opener per (cookie file, mtime): the jar parse is per-call work that is
|
||||||
|
# pure waste inside a batch. Keyed on mtime so a re-imported cookie file under
|
||||||
|
# the same path still gets a fresh jar; only the newest entry is kept.
|
||||||
|
_OPENER_CACHE: dict[tuple[str, float], urllib.request.OpenerDirector] = {}
|
||||||
|
|
||||||
|
|
||||||
|
def _session_urlopen(cookies_file: str, url: str, *, timeout: float = 20.0):
|
||||||
|
path = str(Path(cookies_file).resolve())
|
||||||
|
try:
|
||||||
|
mtime = os.path.getmtime(path)
|
||||||
|
except OSError:
|
||||||
|
mtime = -1.0
|
||||||
|
key = (path, mtime)
|
||||||
|
opener = _OPENER_CACHE.get(key)
|
||||||
|
if opener is None:
|
||||||
|
jar = yt_dlp.cookies.YoutubeDLCookieJar(cookies_file)
|
||||||
|
jar.load(ignore_discard=True, ignore_expires=True)
|
||||||
|
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
|
||||||
|
opener.addheaders = [
|
||||||
|
("User-Agent", _WATCH_PAGE_UA),
|
||||||
|
("Accept-Language", "es-ES,es;q=0.9,en;q=0.8"),
|
||||||
|
]
|
||||||
|
_OPENER_CACHE.clear()
|
||||||
|
_OPENER_CACHE[key] = opener
|
||||||
|
return opener.open(url, timeout=timeout)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_via_watch_page(
|
||||||
|
video_url: str,
|
||||||
|
cookies_file: str,
|
||||||
|
languages: Mapping[str, str],
|
||||||
|
prefer_manual: bool = True,
|
||||||
|
) -> VideoData | None:
|
||||||
|
"""Session-cookie fallback: scrape the watch page directly.
|
||||||
|
|
||||||
|
yt-dlp's InnerTube clients reject logged-in sessions that lack a PO
|
||||||
|
token (playability "The page needs to be reloaded") or return no
|
||||||
|
formats/captions, which kills cookie-authenticated videos — members
|
||||||
|
being the case this exists for. The plain watch page served to the
|
||||||
|
logged-in browser still carries `ytInitialPlayerResponse` with
|
||||||
|
metadata and caption tracks, so GET it with the vault cookie and
|
||||||
|
reuse the normal subtitle picker. Returns None when the page holds
|
||||||
|
no caption tracks at all, so callers keep their own error semantics.
|
||||||
|
"""
|
||||||
|
GLOBAL_PACER.wait()
|
||||||
|
html = _session_urlopen(cookies_file, video_url).read().decode("utf-8", "replace")
|
||||||
|
m = _INITIAL_PLAYER_RE.search(html)
|
||||||
|
if not m:
|
||||||
|
log.warning("watch-page fallback: no ytInitialPlayerResponse for %s", video_url)
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
pr = json.loads(m.group(1))
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
log.warning("watch-page fallback: unparseable player response for %s", video_url)
|
||||||
|
return None
|
||||||
|
|
||||||
|
status = (pr.get("playabilityStatus") or {}).get("status")
|
||||||
|
if status != "OK":
|
||||||
|
reason = (pr.get("playabilityStatus") or {}).get("reason") or status
|
||||||
|
raise RuntimeError(f"watch-page fallback: video not playable ({reason})")
|
||||||
|
|
||||||
|
details = pr.get("videoDetails") or {}
|
||||||
|
micro = (pr.get("microformat") or {}).get("playerMicroformatRenderer") or {}
|
||||||
|
tracks = (
|
||||||
|
(pr.get("captions") or {}).get("playerCaptionsTracklistRenderer") or {}
|
||||||
|
).get("captionTracks") or []
|
||||||
|
if not tracks:
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Reuse pick_subtitle by shaping the tracks as an info dict.
|
||||||
|
info: dict[str, Any] = {
|
||||||
|
"title": details.get("title"),
|
||||||
|
"channel": details.get("author"),
|
||||||
|
"duration": int(details["lengthSeconds"]) if str(details.get("lengthSeconds", "")).isdigit() else None,
|
||||||
|
"view_count": int(details["viewCount"]) if str(details.get("viewCount", "")).isdigit() else None,
|
||||||
|
"description": details.get("shortDescription") or "",
|
||||||
|
"tags": details.get("keywords") or [],
|
||||||
|
"thumbnail": (details.get("thumbnail") or {}).get("thumbnails", [{}])[-1].get("url"),
|
||||||
|
"upload_date": (micro.get("publishDate") or micro.get("uploadDate") or "").replace("-", "") or None,
|
||||||
|
# The watch page carries no availability signal; leaving it unset
|
||||||
|
# keeps the (more informed) discovery value in the store.
|
||||||
|
"availability": None,
|
||||||
|
"subtitles": {},
|
||||||
|
"automatic_captions": {},
|
||||||
|
}
|
||||||
|
for t in tracks:
|
||||||
|
base = t.get("baseUrl") or ""
|
||||||
|
if not base:
|
||||||
|
continue
|
||||||
|
entry = [{"ext": "json3", "url": base + ("&" if "?" in base else "?") + "fmt=json3"}]
|
||||||
|
if t.get("kind") == "asr":
|
||||||
|
info["automatic_captions"].setdefault(t.get("languageCode", ""), []).extend(entry)
|
||||||
|
else:
|
||||||
|
info["subtitles"].setdefault(t.get("languageCode", ""), []).extend(entry)
|
||||||
|
|
||||||
|
pick = pick_subtitle(info, languages, prefer_manual)
|
||||||
|
segments: list[Segment] = []
|
||||||
|
skip_reason: str | None = None
|
||||||
|
if pick:
|
||||||
|
try:
|
||||||
|
GLOBAL_PACER.wait()
|
||||||
|
raw = _session_urlopen(cookies_file, pick.url).read().decode("utf-8", "replace")
|
||||||
|
segments = parse_auto_dump(raw)
|
||||||
|
if not segments:
|
||||||
|
skip_reason = "subtitle downloaded but parsed empty (watch-page fallback)"
|
||||||
|
except Exception as exc: # pylint: disable=broad-except
|
||||||
|
skip_reason = f"caption download failed via watch-page fallback: {exc}"
|
||||||
|
else:
|
||||||
|
skip_reason = describe_missing_subtitle(info, languages)
|
||||||
|
|
||||||
|
return VideoData(
|
||||||
|
info=info, segments=segments, subtitle=pick,
|
||||||
|
has_chapters=bool(info.get("chapters")), skip_reason=skip_reason,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
|
def describe_missing_subtitle(info: dict[str, Any], languages: Mapping[str, str]) -> str:
|
||||||
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
|
"""Explain why no track matched, distinguishing 'none exist' from 'policy rejected them'.
|
||||||
|
|
||||||
|
|||||||
@@ -4,9 +4,9 @@ import logging
|
|||||||
import time
|
import time
|
||||||
from typing import Callable
|
from typing import Callable
|
||||||
|
|
||||||
from .config import Config
|
from .config import Config, keep_ref
|
||||||
from .cookies import resolve_active_path
|
from .cookies import resolve_active_path
|
||||||
from .discover import discover_incremental
|
from .discover import discover_incremental, extract_handle
|
||||||
from .pipeline import process_video
|
from .pipeline import process_video
|
||||||
from .ratelimit import ThrottleGuard, polite_sleep
|
from .ratelimit import ThrottleGuard, polite_sleep
|
||||||
from .store import Store
|
from .store import Store
|
||||||
@@ -77,13 +77,13 @@ def _run_once(
|
|||||||
max_window=cfg.sync.max_window,
|
max_window=cfg.sync.max_window,
|
||||||
overlap=cfg.sync.overlap,
|
overlap=cfg.sync.overlap,
|
||||||
since=store.latest_upload_date(channel_id) if known else None,
|
since=store.latest_upload_date(channel_id) if known else None,
|
||||||
keep=_keep_ref(cfg),
|
keep=keep_ref(cfg),
|
||||||
)
|
)
|
||||||
channel_name, refs = result.channel_name, result.refs
|
channel_name, refs = result.channel_name, result.refs
|
||||||
target_channel = channel_id or result.channel_id
|
target_channel = channel_id or result.channel_id
|
||||||
if result.full_scan:
|
if result.full_scan:
|
||||||
store.upsert_channel(
|
store.upsert_channel(
|
||||||
target_channel, _extract_handle(cfg.channel_url), channel_name, len(refs), avatar=result.avatar
|
target_channel, extract_handle(cfg.channel_url), channel_name, len(refs), avatar=result.avatar
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
store.update_channel_meta(target_channel, name=channel_name, avatar=result.avatar)
|
store.update_channel_meta(target_channel, name=channel_name, avatar=result.avatar)
|
||||||
@@ -133,23 +133,3 @@ def _run_once(
|
|||||||
_emit(f"watch: throttled, waiting {wait:.1f}s")
|
_emit(f"watch: throttled, waiting {wait:.1f}s")
|
||||||
time.sleep(wait)
|
time.sleep(wait)
|
||||||
polite_sleep(cfg.delay.min_seconds, cfg.delay.max_seconds)
|
polite_sleep(cfg.delay.min_seconds, cfg.delay.max_seconds)
|
||||||
|
|
||||||
|
|
||||||
def _keep_ref(cfg: Config):
|
|
||||||
"""Same shorts/live rules the store was populated under — see jobs._keep_ref."""
|
|
||||||
|
|
||||||
def keep(r) -> bool:
|
|
||||||
url = r.url or ""
|
|
||||||
if not cfg.include_shorts and "/shorts/" in url:
|
|
||||||
return False
|
|
||||||
if not cfg.include_live and url.startswith("https://www.youtube.com/live/"):
|
|
||||||
return False
|
|
||||||
return True
|
|
||||||
|
|
||||||
return keep
|
|
||||||
|
|
||||||
|
|
||||||
def _extract_handle(url: str) -> str:
|
|
||||||
if "@" in url:
|
|
||||||
return "@" + url.split("@", 1)[1].split("/", 1)[0]
|
|
||||||
return ""
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from .chapters import align_chapters, chapters_from_info
|
|||||||
from .config import Config
|
from .config import Config
|
||||||
from .extract import extract_video
|
from .extract import extract_video
|
||||||
from .ratelimit import is_rate_limited
|
from .ratelimit import is_rate_limited
|
||||||
from .render import build_filename_stem, render_markdown
|
from .render import build_filename_stem, render_markdown, safe_dirname
|
||||||
from .store import Store, VideoRow
|
from .store import Store, VideoRow
|
||||||
from ._yt_http import yt_get
|
from ._yt_http import yt_get
|
||||||
|
|
||||||
@@ -44,7 +44,7 @@ def _render_and_retire(
|
|||||||
template=cfg.filename_template,
|
template=cfg.filename_template,
|
||||||
video_id=video_id,
|
video_id=video_id,
|
||||||
)
|
)
|
||||||
out_subdir = Path(cfg.output_dir_resolved) / _safe_dirname(channel_name)
|
out_subdir = Path(cfg.output_dir_resolved) / safe_dirname(channel_name)
|
||||||
md_path = render_markdown(cfg.template_path_resolved, out_subdir, stem, context)
|
md_path = render_markdown(cfg.template_path_resolved, out_subdir, stem, context)
|
||||||
|
|
||||||
out_root = Path(cfg.output_dir_resolved).parent
|
out_root = Path(cfg.output_dir_resolved).parent
|
||||||
@@ -122,13 +122,15 @@ def process_video(
|
|||||||
return "error"
|
return "error"
|
||||||
|
|
||||||
info = data.info
|
info = data.info
|
||||||
store.set_availability(row.video_id, info.get("availability"))
|
|
||||||
|
|
||||||
# Metadata is persisted BEFORE the no-transcript exit. The extraction already
|
# Metadata is persisted BEFORE the no-transcript exit. The extraction already
|
||||||
# cost its requests and the info dict is in hand; discarding it because the
|
# cost its requests and the info dict is in hand; discarding it because the
|
||||||
# separate caption fetch failed means a retry re-spends them for data we
|
# separate caption fetch failed means a retry re-spends them for data we
|
||||||
# already had. Measured after a throttling incident: five rows left with
|
# already had. Measured after a throttling incident: five rows left with
|
||||||
# upload_date, view_count, description and thumbnail all NULL.
|
# upload_date, view_count, description and thumbnail all NULL.
|
||||||
|
# Agrupadas en una transaccion: una conexion/commit en vez de tres.
|
||||||
|
with store.transaction():
|
||||||
|
store.set_availability(row.video_id, info.get("availability"))
|
||||||
_store_metadata(store, row, info)
|
_store_metadata(store, row, info)
|
||||||
|
|
||||||
if not data.segments:
|
if not data.segments:
|
||||||
@@ -148,7 +150,9 @@ def process_video(
|
|||||||
chapters = chapters_from_info(data.info)
|
chapters = chapters_from_info(data.info)
|
||||||
sections = align_chapters(data.segments, chapters)
|
sections = align_chapters(data.segments, chapters)
|
||||||
|
|
||||||
# persist segments + rich metadata to DB (for search, stats, webapp)
|
# persist segments + rich metadata to DB (for search, stats, webapp);
|
||||||
|
# una transaccion: delete+inserts+update atomicos y un solo commit
|
||||||
|
with store.transaction():
|
||||||
store.store_segments(row.video_id, data.segments)
|
store.store_segments(row.video_id, data.segments)
|
||||||
seg_json = json.dumps(
|
seg_json = json.dumps(
|
||||||
[{"start": s.start, "end": s.end, "text": s.text} for s in data.segments],
|
[{"start": s.start, "end": s.end, "text": s.text} for s in data.segments],
|
||||||
@@ -205,11 +209,6 @@ def _normalize_date(d: str | None) -> str:
|
|||||||
return d
|
return d
|
||||||
|
|
||||||
|
|
||||||
def _safe_dirname(name: str) -> str:
|
|
||||||
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
|
|
||||||
return safe.strip().strip(".") or "unknown"
|
|
||||||
|
|
||||||
|
|
||||||
def thumbnail_url_for(video_row) -> str:
|
def thumbnail_url_for(video_row) -> str:
|
||||||
"""Thumbnail URL for a video: stored URL, else the canonical YouTube one derived from its id."""
|
"""Thumbnail URL for a video: stored URL, else the canonical YouTube one derived from its id."""
|
||||||
if video_row and getattr(video_row, "thumbnail", None):
|
if video_row and getattr(video_row, "thumbnail", None):
|
||||||
@@ -259,7 +258,6 @@ def re_render_videos(store: Store, cfg: Config, channel_id: str | None = None) -
|
|||||||
import json
|
import json
|
||||||
from .chapters import align_chapters, Chapter
|
from .chapters import align_chapters, Chapter
|
||||||
from .parse import Segment
|
from .parse import Segment
|
||||||
from .render import build_filename_stem, render_markdown
|
|
||||||
|
|
||||||
videos = [v for v in store.get_all(channel_id) if v.status == "done" and v.segments_json]
|
videos = [v for v in store.get_all(channel_id) if v.status == "done" and v.segments_json]
|
||||||
n = 0
|
n = 0
|
||||||
|
|||||||
@@ -191,6 +191,7 @@ def ydl_throttle_opts(
|
|||||||
*,
|
*,
|
||||||
extractor_retries: int = 3,
|
extractor_retries: int = 3,
|
||||||
socket_timeout: float = 30.0,
|
socket_timeout: float = 30.0,
|
||||||
|
js_runtimes: dict[str, dict] | None = None,
|
||||||
) -> dict[str, object]:
|
) -> dict[str, object]:
|
||||||
"""The politeness half of every `ydl_opts` dict in this project.
|
"""The politeness half of every `ydl_opts` dict in this project.
|
||||||
|
|
||||||
@@ -213,6 +214,7 @@ def ydl_throttle_opts(
|
|||||||
# this only covers transient 5xx and network errors.
|
# this only covers transient 5xx and network errors.
|
||||||
"extractor_retries": int(extractor_retries),
|
"extractor_retries": int(extractor_retries),
|
||||||
"socket_timeout": float(socket_timeout),
|
"socket_timeout": float(socket_timeout),
|
||||||
|
"js_runtimes": js_runtimes if js_runtimes is not None else {"node": {}, "deno": {}, "bun": {}, "quickjs": {}},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -32,7 +32,16 @@ def to_json(value) -> str:
|
|||||||
return json.dumps(value, ensure_ascii=False)
|
return json.dumps(value, ensure_ascii=False)
|
||||||
|
|
||||||
|
|
||||||
|
# One Environment per template directory, cached for the process lifetime:
|
||||||
|
# building it (loader + filters) per rendered note was the dominant cost of
|
||||||
|
# batch renders. Jinja's own per-env template cache keeps the compiled
|
||||||
|
# Template, and its default auto_reload still picks up on-disk edits.
|
||||||
|
_ENV_CACHE: dict[Path, Environment] = {}
|
||||||
|
|
||||||
|
|
||||||
def _make_env(template_dir: Path) -> Environment:
|
def _make_env(template_dir: Path) -> Environment:
|
||||||
|
env = _ENV_CACHE.get(template_dir)
|
||||||
|
if env is None:
|
||||||
env = Environment(
|
env = Environment(
|
||||||
loader=FileSystemLoader(str(template_dir)),
|
loader=FileSystemLoader(str(template_dir)),
|
||||||
autoescape=select_autoescape(disabled_extensions=("j2", "txt")),
|
autoescape=select_autoescape(disabled_extensions=("j2", "txt")),
|
||||||
@@ -42,6 +51,7 @@ def _make_env(template_dir: Path) -> Environment:
|
|||||||
env.filters["format_timestamp"] = format_timestamp
|
env.filters["format_timestamp"] = format_timestamp
|
||||||
env.filters["quote_yaml"] = quote_yaml
|
env.filters["quote_yaml"] = quote_yaml
|
||||||
env.filters["to_json"] = to_json
|
env.filters["to_json"] = to_json
|
||||||
|
_ENV_CACHE[template_dir] = env
|
||||||
return env
|
return env
|
||||||
|
|
||||||
|
|
||||||
@@ -63,6 +73,24 @@ def render_markdown(
|
|||||||
return out_file
|
return out_file
|
||||||
|
|
||||||
|
|
||||||
|
def safe_dirname(name: str | None) -> str:
|
||||||
|
"""Channel name -> markdown subdirectory name, shared by every .md writer.
|
||||||
|
|
||||||
|
Strips the characters Windows forbids in a path segment, then leading and
|
||||||
|
trailing spaces/dots (also illegal there), falling back to "unknown" so the
|
||||||
|
output tree never grows a nameless root. Callers with a bare filename want
|
||||||
|
`safe_filename` instead ("untitled" fallback).
|
||||||
|
"""
|
||||||
|
safe = "".join(c for c in (name or "") if c not in r'\/:*?"<>|')
|
||||||
|
return safe.strip().strip(".") or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def safe_filename(name: str) -> str:
|
||||||
|
"""Free-text (title) -> file-legal stem; "untitled" when nothing survives."""
|
||||||
|
safe = "".join(c for c in name if c not in r'\/:*?"<>|')
|
||||||
|
return safe.strip().strip(".") or "untitled"
|
||||||
|
|
||||||
|
|
||||||
def build_filename_stem(
|
def build_filename_stem(
|
||||||
upload_date: str | None,
|
upload_date: str | None,
|
||||||
title: str,
|
title: str,
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ from pathlib import Path
|
|||||||
from typing import Callable
|
from typing import Callable
|
||||||
|
|
||||||
from .parse import Segment
|
from .parse import Segment
|
||||||
from .store import SearchHit, Store
|
from .store import Store
|
||||||
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
@@ -91,14 +91,6 @@ def _strip_quotes(value: str) -> str:
|
|||||||
return v
|
return v
|
||||||
|
|
||||||
|
|
||||||
def store_segments(store: Store, video_id: str, segments: list[Segment]) -> None:
|
|
||||||
store.store_segments(video_id, segments)
|
|
||||||
|
|
||||||
|
|
||||||
def search(store: Store, query: str, channel_id: str | None = None, limit: int = 50) -> list[SearchHit]:
|
|
||||||
return store.search_segments(query, channel_id=channel_id, limit=limit)
|
|
||||||
|
|
||||||
|
|
||||||
def backfill_from_markdown(
|
def backfill_from_markdown(
|
||||||
store: Store,
|
store: Store,
|
||||||
md_root: Path,
|
md_root: Path,
|
||||||
|
|||||||
+107
-30
@@ -2,6 +2,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import json
|
import json
|
||||||
import sqlite3
|
import sqlite3
|
||||||
|
import threading
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
@@ -393,10 +394,30 @@ class JobRow:
|
|||||||
last_error: str | None
|
last_error: str | None
|
||||||
|
|
||||||
|
|
||||||
|
def order_pending(rows: list[VideoRow], refs: list[VideoRef]) -> list[VideoRow]:
|
||||||
|
"""Newest-window-first ordering for the pending queue.
|
||||||
|
|
||||||
|
`get_pending` is ordered by discovery time, which used to coincide with
|
||||||
|
newest-first because discovery saw the whole channel at once. With windowed
|
||||||
|
sync that no longer holds, so put the videos from this run's window (the
|
||||||
|
`refs` discovery just returned, newest first) at the front and keep the rest
|
||||||
|
of the backlog behind them. Shared by the CLI scrape and the webapp's
|
||||||
|
channel jobs, which must agree on what `--limit` / `limit` mean.
|
||||||
|
"""
|
||||||
|
by_id = {r.video_id: r for r in rows}
|
||||||
|
ordered = [by_id.pop(x.video_id) for x in refs if x.video_id in by_id]
|
||||||
|
ordered.extend(by_id.values())
|
||||||
|
return ordered
|
||||||
|
|
||||||
|
|
||||||
class Store:
|
class Store:
|
||||||
def __init__(self, db_path: str | Path):
|
def __init__(self, db_path: str | Path):
|
||||||
self.db_path = Path(db_path)
|
self.db_path = Path(db_path)
|
||||||
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
# Pila de transacciones ambientales, por hilo: permite agrupar varias
|
||||||
|
# llamadas a Store en una sola conexion/commit sin pasar la conexion
|
||||||
|
# como parametro a cada metodo.
|
||||||
|
self._local = threading.local()
|
||||||
self._init_schema()
|
self._init_schema()
|
||||||
|
|
||||||
def _connect(self) -> sqlite3.Connection:
|
def _connect(self) -> sqlite3.Connection:
|
||||||
@@ -419,13 +440,53 @@ class Store:
|
|||||||
if col not in ch_existing:
|
if col not in ch_existing:
|
||||||
conn.execute(f"ALTER TABLE channels ADD COLUMN {col} {coltype}")
|
conn.execute(f"ALTER TABLE channels ADD COLUMN {col} {coltype}")
|
||||||
conn.executescript(_EXTRA_SCHEMA)
|
conn.executescript(_EXTRA_SCHEMA)
|
||||||
# Unconditional, not "only when the column was just added": a row can
|
# Unconditional in spirit, not in cost: a row can also arrive
|
||||||
# also arrive unranked afterwards, and an unranked row is displayed
|
# unranked afterwards, and an unranked row is displayed in the
|
||||||
# in the wrong place rather than merely in an arbitrary one.
|
# wrong place rather than merely in an arbitrary one. But
|
||||||
|
# _rank_unranked scans the whole videos table, and Store is built
|
||||||
|
# on every webapp import — so gate it behind the same predicate it
|
||||||
|
# matches on, which is a single indexed probe.
|
||||||
|
if conn.execute(
|
||||||
|
"SELECT 1 FROM videos WHERE channel_seq IS NULL LIMIT 1"
|
||||||
|
).fetchone() is not None:
|
||||||
_rank_unranked(conn)
|
_rank_unranked(conn)
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def transaction(self):
|
||||||
|
"""Agrupa varias escrituras de Store en una sola conexion y un commit.
|
||||||
|
|
||||||
|
`process_video` costaba ~6 connect/commit por video (un fsync cada
|
||||||
|
uno en WAL). Dentro de este bloque, toda llamada a Store hecha desde
|
||||||
|
el MISMO hilo se une a la conexion ambiental; un error revierte el
|
||||||
|
grupo entero, que es la atomicidad por video que se quiere de todos
|
||||||
|
modos.
|
||||||
|
"""
|
||||||
|
conn = self._connect()
|
||||||
|
stack: list[sqlite3.Connection] = getattr(self._local, "stack", None) or []
|
||||||
|
self._local.stack = stack
|
||||||
|
stack.append(conn)
|
||||||
|
try:
|
||||||
|
yield conn
|
||||||
|
conn.commit()
|
||||||
|
except BaseException:
|
||||||
|
conn.rollback()
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
stack.remove(conn)
|
||||||
|
conn.close()
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def _cursor(self) -> Iterator[sqlite3.Cursor]:
|
def _cursor(self) -> Iterator[sqlite3.Cursor]:
|
||||||
|
# Dentro de transaction(): misma conexion y sin commit intermedio —
|
||||||
|
# el bloque externo decide cuando el trabajo se vuelve durable.
|
||||||
|
stack: list[sqlite3.Connection] = getattr(self._local, "stack", None) or []
|
||||||
|
if stack:
|
||||||
|
cur = stack[-1].cursor()
|
||||||
|
try:
|
||||||
|
yield cur
|
||||||
|
finally:
|
||||||
|
cur.close()
|
||||||
|
return
|
||||||
conn = self._connect()
|
conn = self._connect()
|
||||||
try:
|
try:
|
||||||
yield conn.cursor()
|
yield conn.cursor()
|
||||||
@@ -576,8 +637,10 @@ class Store:
|
|||||||
for rank, (_, r) in enumerate(ordered):
|
for rank, (_, r) in enumerate(ordered):
|
||||||
seqs[r.video_id] = base + width - rank
|
seqs[r.video_id] = base + width - rank
|
||||||
|
|
||||||
for r in refs:
|
# One executemany instead of a prepared-statement round per ref:
|
||||||
cur.execute(
|
# discovery hands over hundreds of rows at once and the statement
|
||||||
|
# text is identical for all of them.
|
||||||
|
cur.executemany(
|
||||||
"""INSERT INTO videos
|
"""INSERT INTO videos
|
||||||
(video_id, channel_id, title, url, upload_date,
|
(video_id, channel_id, title, url, upload_date,
|
||||||
upload_date_approx, duration, availability,
|
upload_date_approx, duration, availability,
|
||||||
@@ -600,10 +663,14 @@ class Store:
|
|||||||
duration = COALESCE(excluded.duration, videos.duration),
|
duration = COALESCE(excluded.duration, videos.duration),
|
||||||
availability = COALESCE(excluded.availability, videos.availability),
|
availability = COALESCE(excluded.availability, videos.availability),
|
||||||
channel_seq = COALESCE(excluded.channel_seq, videos.channel_seq)""",
|
channel_seq = COALESCE(excluded.channel_seq, videos.channel_seq)""",
|
||||||
|
[
|
||||||
(r.video_id, r.channel_id, r.title, r.url, r.upload_date,
|
(r.video_id, r.channel_id, r.title, r.url, r.upload_date,
|
||||||
int(r.date_approx or 0), r.duration,
|
int(r.date_approx or 0), r.duration, r.availability,
|
||||||
r.availability, seqs.get(r.video_id), now),
|
seqs.get(r.video_id), now)
|
||||||
|
for r in refs
|
||||||
|
],
|
||||||
)
|
)
|
||||||
|
for r in refs:
|
||||||
if r.video_id not in existing:
|
if r.video_id not in existing:
|
||||||
inserted += 1
|
inserted += 1
|
||||||
existing.add(r.video_id)
|
existing.add(r.video_id)
|
||||||
@@ -1038,34 +1105,44 @@ class Store:
|
|||||||
def dashboard(self) -> dict:
|
def dashboard(self) -> dict:
|
||||||
with self._cursor() as cur:
|
with self._cursor() as cur:
|
||||||
channels = [dict(r) for r in cur.execute("SELECT * FROM channels ORDER BY name").fetchall()]
|
channels = [dict(r) for r in cur.execute("SELECT * FROM channels ORDER BY name").fetchall()]
|
||||||
|
# One GROUP BY instead of one aggregate query per channel (N+1).
|
||||||
|
# Channels with zero videos are absent from the grouping and get
|
||||||
|
# the same zeros/NULLs the per-channel query used to return.
|
||||||
|
agg = {
|
||||||
|
r["channel_id"]: r
|
||||||
|
for r in cur.execute(
|
||||||
|
"SELECT channel_id, COUNT(*) AS n, COALESCE(SUM(duration),0) AS dur, "
|
||||||
|
"MIN(upload_date) AS mind, MAX(upload_date) AS maxd, "
|
||||||
|
"COALESCE(SUM(view_count),0) AS views, COALESCE(SUM(like_count),0) AS likes "
|
||||||
|
"FROM videos GROUP BY channel_id"
|
||||||
|
).fetchall()
|
||||||
|
}
|
||||||
for ch in channels:
|
for ch in channels:
|
||||||
cid = ch["channel_id"]
|
row = agg.get(ch["channel_id"])
|
||||||
row = cur.execute(
|
ch["video_count_db"] = row["n"] if row else 0
|
||||||
"SELECT COUNT(*) AS n, COALESCE(SUM(duration),0) AS dur, MIN(upload_date) AS mind, MAX(upload_date) AS maxd, COALESCE(SUM(view_count),0) AS views, COALESCE(SUM(like_count),0) AS likes FROM videos WHERE channel_id = ?",
|
ch["total_duration"] = row["dur"] if row else 0
|
||||||
(cid,),
|
ch["date_min"] = row["mind"] if row else None
|
||||||
).fetchone()
|
ch["date_max"] = row["maxd"] if row else None
|
||||||
ch["video_count_db"] = row["n"]
|
ch["total_views"] = row["views"] if row else 0
|
||||||
ch["total_duration"] = row["dur"]
|
ch["total_likes"] = row["likes"] if row else 0
|
||||||
ch["date_min"] = row["mind"]
|
|
||||||
ch["date_max"] = row["maxd"]
|
|
||||||
ch["total_views"] = row["views"]
|
|
||||||
ch["total_likes"] = row["likes"]
|
|
||||||
status_breakdown = {
|
status_breakdown = {
|
||||||
r["status"]: r["n"]
|
r["status"]: r["n"]
|
||||||
for r in cur.execute("SELECT status, COUNT(*) AS n FROM videos GROUP BY status").fetchall()
|
for r in cur.execute("SELECT status, COUNT(*) AS n FROM videos GROUP BY status").fetchall()
|
||||||
}
|
}
|
||||||
# top tags (tags is JSON array text)
|
# top tags (tags is JSON array text). Counted in SQL via json_each
|
||||||
tag_rows = cur.execute("SELECT tags FROM videos WHERE tags IS NOT NULL AND tags != '[]'").fetchall()
|
# instead of loading every tags string into Python; json_valid
|
||||||
tag_counts: dict[str, int] = {}
|
# skips rows the old json.loads/except path also skipped, and the
|
||||||
for tr in tag_rows:
|
# TRIM/LOWER/empty-filter mirrors the per-tag normalisation.
|
||||||
try:
|
tag_counts = {
|
||||||
for t in json.loads(tr["tags"]):
|
r["tag"]: r["n"]
|
||||||
t = (t or "").strip().lower()
|
for r in cur.execute(
|
||||||
if t:
|
"SELECT LOWER(TRIM(je.value)) AS tag, COUNT(*) AS n "
|
||||||
tag_counts[t] = tag_counts.get(t, 0) + 1
|
"FROM videos v, json_each(v.tags) je "
|
||||||
except (json.JSONDecodeError, TypeError):
|
"WHERE json_valid(v.tags) AND TRIM(je.value) <> '' "
|
||||||
continue
|
"GROUP BY tag ORDER BY n DESC LIMIT 20"
|
||||||
top_tags = sorted(tag_counts.items(), key=lambda x: x[1], reverse=True)[:20]
|
).fetchall()
|
||||||
|
}
|
||||||
|
top_tags = sorted(tag_counts.items(), key=lambda x: x[1], reverse=True)
|
||||||
return {
|
return {
|
||||||
"channels": channels,
|
"channels": channels,
|
||||||
"status_breakdown": status_breakdown,
|
"status_breakdown": status_breakdown,
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
@@ -12,7 +13,9 @@ from .. import analysis as analysis_mod
|
|||||||
from .. import cookies as cookies_mod
|
from .. import cookies as cookies_mod
|
||||||
from .. import export as export_mod
|
from .. import export as export_mod
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
|
from ..discover import extract_handle
|
||||||
from ..ratelimit import polite_sleep
|
from ..ratelimit import polite_sleep
|
||||||
|
from ..render import safe_dirname
|
||||||
from .. import store as store_mod
|
from .. import store as store_mod
|
||||||
from ..store import Store
|
from ..store import Store
|
||||||
|
|
||||||
@@ -21,6 +24,11 @@ from ..store import Store
|
|||||||
#: so this only bounds the eager burst that follows "Add channel".
|
#: so this only bounds the eager burst that follows "Add channel".
|
||||||
THUMBNAIL_AUTO_LIMIT = 60
|
THUMBNAIL_AUTO_LIMIT = 60
|
||||||
|
|
||||||
|
#: Thumbnail fetches go to the i.ytimg.com CDN (not youtube.com), so the
|
||||||
|
#: Pacer does not apply to them; a small pool turns ~60 sequential round
|
||||||
|
#: trips into ~10 batches without hammering the CDN.
|
||||||
|
THUMBNAIL_WORKERS = 6
|
||||||
|
|
||||||
|
|
||||||
def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
||||||
r = APIRouter(prefix="/api")
|
r = APIRouter(prefix="/api")
|
||||||
@@ -50,7 +58,7 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
avatar_cached = False
|
avatar_cached = False
|
||||||
if not avatar:
|
if not avatar:
|
||||||
avatar = deep_channel_avatar(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
|
avatar = deep_channel_avatar(url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
|
||||||
store.upsert_channel(cid, _handle(url), name, len(refs), avatar=avatar)
|
store.upsert_channel(cid, extract_handle(url), name, len(refs), avatar=avatar)
|
||||||
store.upsert_videos(refs)
|
store.upsert_videos(refs)
|
||||||
if avatar:
|
if avatar:
|
||||||
avatars_dir = Path(cfg.output_dir_resolved).parent / "avatars"
|
avatars_dir = Path(cfg.output_dir_resolved).parent / "avatars"
|
||||||
@@ -229,6 +237,31 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
cookies_mod.set_active(store, imported[0])
|
cookies_mod.set_active(store, imported[0])
|
||||||
return {"imported": imported}
|
return {"imported": imported}
|
||||||
|
|
||||||
|
@r.post("/cookies/from-browser")
|
||||||
|
def cookies_from_browser(payload: dict):
|
||||||
|
"""Extract youtube.com cookies from a local browser (default: Brave).
|
||||||
|
|
||||||
|
Requires the browser to be fully closed; Chromium keeps its cookie
|
||||||
|
database locked while running.
|
||||||
|
"""
|
||||||
|
payload = payload or {}
|
||||||
|
browser = (payload.get("browser") or "brave").strip().lower()
|
||||||
|
profile = payload.get("profile") or None
|
||||||
|
try:
|
||||||
|
cid = cookies_mod.import_from_browser(store, browser=browser, profile=profile)
|
||||||
|
except cookies_mod.BrowserCookieLockedError as exc:
|
||||||
|
raise HTTPException(409, str(exc))
|
||||||
|
except Exception as exc:
|
||||||
|
raise HTTPException(400, str(exc))
|
||||||
|
# The user imported it precisely to use it.
|
||||||
|
cookies_mod.set_active(store, cid)
|
||||||
|
row = store.get_cookie(cid)
|
||||||
|
return {
|
||||||
|
"imported": [cid], "active": cid, "browser": browser,
|
||||||
|
"cookie_count": row.cookie_count if row else None,
|
||||||
|
"has_session": row.has_session if row else None,
|
||||||
|
}
|
||||||
|
|
||||||
@r.post("/cookies/{cookie_id}/activate")
|
@r.post("/cookies/{cookie_id}/activate")
|
||||||
def cookies_activate(cookie_id: str):
|
def cookies_activate(cookie_id: str):
|
||||||
cookies_mod.set_active(store, cookie_id)
|
cookies_mod.set_active(store, cookie_id)
|
||||||
@@ -289,8 +322,11 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
"database": _folder_info(Path(cfg.database_path_resolved).parent),
|
"database": _folder_info(Path(cfg.database_path_resolved).parent),
|
||||||
}}
|
}}
|
||||||
|
|
||||||
|
# `def`, not `async def`: same reason as add_channel — the body blocks on
|
||||||
|
# SQLite reads and a subprocess/os.startfile call, which would hold the
|
||||||
|
# event loop if this were async.
|
||||||
@r.post("/folders/open")
|
@r.post("/folders/open")
|
||||||
async def open_folder(payload: dict):
|
def open_folder(payload: dict):
|
||||||
kind = (payload or {}).get("kind")
|
kind = (payload or {}).get("kind")
|
||||||
channel_id = (payload or {}).get("channel_id")
|
channel_id = (payload or {}).get("channel_id")
|
||||||
data_root = Path(cfg.output_dir_resolved).parent
|
data_root = Path(cfg.output_dir_resolved).parent
|
||||||
@@ -299,7 +335,7 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
if channel_id:
|
if channel_id:
|
||||||
ch = store.get_channel(channel_id) or {}
|
ch = store.get_channel(channel_id) or {}
|
||||||
if ch.get("name"):
|
if ch.get("name"):
|
||||||
target = Path(cfg.output_dir_resolved) / _safe_dir(ch["name"])
|
target = Path(cfg.output_dir_resolved) / safe_dirname(ch["name"])
|
||||||
elif kind == "exports":
|
elif kind == "exports":
|
||||||
target = data_root / "exports"
|
target = data_root / "exports"
|
||||||
elif kind == "audio":
|
elif kind == "audio":
|
||||||
@@ -375,6 +411,17 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
raise HTTPException(404, "markdown file missing on disk")
|
raise HTTPException(404, "markdown file missing on disk")
|
||||||
return FileResponse(str(p), filename=p.name, media_type="text/markdown")
|
return FileResponse(str(p), filename=p.name, media_type="text/markdown")
|
||||||
|
|
||||||
|
@r.post("/videos/{video_id}/open-markdown")
|
||||||
|
def open_video_markdown(video_id: str):
|
||||||
|
v = store.get_video(video_id)
|
||||||
|
if not v or not v.markdown_path:
|
||||||
|
raise HTTPException(404, "markdown not generated yet")
|
||||||
|
p = Path(cfg.output_dir_resolved).parent / v.markdown_path
|
||||||
|
if not p.exists():
|
||||||
|
raise HTTPException(404, "markdown file missing on disk")
|
||||||
|
_open_in_os(p)
|
||||||
|
return {"opened": str(p)}
|
||||||
|
|
||||||
@r.api_route("/videos/{video_id}/audio", methods=["GET", "HEAD"])
|
@r.api_route("/videos/{video_id}/audio", methods=["GET", "HEAD"])
|
||||||
def video_audio(video_id: str, request: Request):
|
def video_audio(video_id: str, request: Request):
|
||||||
# audio is stored as data/audio/<video_id>.mp3 (see jobs._run_audio outtmpl)
|
# audio is stored as data/audio/<video_id>.mp3 (see jobs._run_audio outtmpl)
|
||||||
@@ -423,10 +470,13 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
video_ids = video_ids[:limit]
|
video_ids = video_ids[:limit]
|
||||||
out_dir = Path(cfg.output_dir_resolved).parent / "thumbnails"
|
out_dir = Path(cfg.output_dir_resolved).parent / "thumbnails"
|
||||||
out_dir.mkdir(parents=True, exist_ok=True)
|
out_dir.mkdir(parents=True, exist_ok=True)
|
||||||
n = 0
|
# CDN fetches, one per video, each independent: run them on a small
|
||||||
for vid in video_ids:
|
# thread pool instead of sequentially. cache_thumbnail owns its own
|
||||||
if cache_thumbnail(store, vid, out_dir):
|
# SQLite connection and writes one file per id, so this is safe;
|
||||||
n += 1
|
# failures still count as False exactly as before.
|
||||||
|
with ThreadPoolExecutor(max_workers=THUMBNAIL_WORKERS) as pool:
|
||||||
|
results = list(pool.map(lambda vid: cache_thumbnail(store, vid, out_dir), video_ids))
|
||||||
|
n = sum(1 for ok in results if ok)
|
||||||
return {"downloaded": n, "dir": str(out_dir), "skipped": skipped}
|
return {"downloaded": n, "dir": str(out_dir), "skipped": skipped}
|
||||||
|
|
||||||
@r.get("/thumbnails/{video_id}")
|
@r.get("/thumbnails/{video_id}")
|
||||||
@@ -455,8 +505,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
"permanent": counts.get("permanent", 0),
|
"permanent": counts.get("permanent", 0),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# `def`, not `async def`: the UPDATE below blocks on SQLite; run it on the
|
||||||
|
# threadpool like the other blocking handlers.
|
||||||
@r.post("/videos/reset")
|
@r.post("/videos/reset")
|
||||||
async def reset_videos(payload: dict):
|
def reset_videos(payload: dict):
|
||||||
"""Send `error` / `no_subtitles` videos back to pending so they can be retried.
|
"""Send `error` / `no_subtitles` videos back to pending so they can be retried.
|
||||||
|
|
||||||
`no_subtitles` is resettable on purpose: it is recorded whenever the
|
`no_subtitles` is resettable on purpose: it is recorded whenever the
|
||||||
@@ -620,8 +672,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
"top_tags": d["top_tags"], "totals": totals}
|
"top_tags": d["top_tags"], "totals": totals}
|
||||||
|
|
||||||
# -------------------------------------------------- audio (job)
|
# -------------------------------------------------- audio (job)
|
||||||
|
# `def`, not `async def`: no await here, and the SQLite reads below would
|
||||||
|
# block the event loop.
|
||||||
@r.post("/tools/audio")
|
@r.post("/tools/audio")
|
||||||
async def tools_audio(payload: dict):
|
def tools_audio(payload: dict):
|
||||||
video_ids = (payload or {}).get("video_ids") or []
|
video_ids = (payload or {}).get("video_ids") or []
|
||||||
channel_id = (payload or {}).get("channel_id")
|
channel_id = (payload or {}).get("channel_id")
|
||||||
if not video_ids and not channel_id:
|
if not video_ids and not channel_id:
|
||||||
@@ -632,8 +686,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
job_id = jobs.enqueue(channel_id, opts)
|
job_id = jobs.enqueue(channel_id, opts)
|
||||||
return {"job_id": job_id}
|
return {"job_id": job_id}
|
||||||
|
|
||||||
|
# `def`, not `async def`: same as tools_audio — SQLite reads + filesystem
|
||||||
|
# stats only, no await.
|
||||||
@r.post("/tools/video")
|
@r.post("/tools/video")
|
||||||
async def tools_video(payload: dict):
|
def tools_video(payload: dict):
|
||||||
video_ids = (payload or {}).get("video_ids") or []
|
video_ids = (payload or {}).get("video_ids") or []
|
||||||
if len(video_ids) != 1:
|
if len(video_ids) != 1:
|
||||||
raise HTTPException(400, "exactly one video_id required")
|
raise HTTPException(400, "exactly one video_id required")
|
||||||
@@ -652,11 +708,10 @@ def build_router(store: Store, cfg: Config, jobs) -> APIRouter:
|
|||||||
|
|
||||||
|
|
||||||
def _video_dict(v) -> dict:
|
def _video_dict(v) -> dict:
|
||||||
import json as _json
|
|
||||||
tags = []
|
tags = []
|
||||||
if v.tags:
|
if v.tags:
|
||||||
try:
|
try:
|
||||||
tags = _json.loads(v.tags)
|
tags = json.loads(v.tags)
|
||||||
except Exception:
|
except Exception:
|
||||||
tags = []
|
tags = []
|
||||||
return {
|
return {
|
||||||
@@ -710,22 +765,11 @@ def _loads(s):
|
|||||||
return {}
|
return {}
|
||||||
|
|
||||||
|
|
||||||
def _handle(url: str) -> str:
|
|
||||||
if "@" in url:
|
|
||||||
return "@" + url.split("@", 1)[1].split("/", 1)[0]
|
|
||||||
return ""
|
|
||||||
|
|
||||||
|
|
||||||
def _folder_info(path: Path) -> dict:
|
def _folder_info(path: Path) -> dict:
|
||||||
path = Path(path)
|
path = Path(path)
|
||||||
return {"path": str(path.resolve()), "exists": path.exists()}
|
return {"path": str(path.resolve()), "exists": path.exists()}
|
||||||
|
|
||||||
|
|
||||||
def _safe_dir(name: str) -> str:
|
|
||||||
safe = "".join(c for c in (name or "") if c not in r'\/:*?"<>|')
|
|
||||||
return (safe.strip().strip(".") or "unknown")
|
|
||||||
|
|
||||||
|
|
||||||
def _open_in_os(path: Path) -> None:
|
def _open_in_os(path: Path) -> None:
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import threading
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from fastapi import FastAPI
|
from fastapi import FastAPI
|
||||||
@@ -47,18 +48,44 @@ def create_app(cfg: Config | None = None, db_path: str | Path | None = None) ->
|
|||||||
# Reconcile, not just backfill: backfill_from_markdown populates segments
|
# Reconcile, not just backfill: backfill_from_markdown populates segments
|
||||||
# and metadata but never touches `status`/`markdown_path`, so a video whose
|
# and metadata but never touches `status`/`markdown_path`, so a video whose
|
||||||
# .md is already on disk would keep showing a failure after every restart.
|
# .md is already on disk would keep showing a failure after every restart.
|
||||||
|
#
|
||||||
|
# It walks and parses EVERY .md in the tree, which on a large library
|
||||||
|
# blocked uvicorn boot for the whole scan — so it runs on a daemon thread
|
||||||
|
# and the server answers immediately (SQLite/WAL absorbs the concurrent
|
||||||
|
# writes; /api/tools/reconcile still re-runs it on demand). healthz
|
||||||
|
# reports whether the initial pass has finished.
|
||||||
|
reconcile_done = threading.Event()
|
||||||
|
|
||||||
|
def _startup_reconcile() -> None:
|
||||||
try:
|
try:
|
||||||
md_root = Path(cfg.output_dir_resolved)
|
md_root = Path(cfg.output_dir_resolved)
|
||||||
if md_root.exists():
|
if md_root.exists():
|
||||||
reconcile_markdown(store, md_root, log=pkg_log.info)
|
reconcile_markdown(store, md_root, log=pkg_log.info)
|
||||||
except Exception:
|
except Exception:
|
||||||
pkg_log.exception("startup reconcile failed")
|
pkg_log.exception("startup reconcile failed")
|
||||||
|
finally:
|
||||||
|
reconcile_done.set()
|
||||||
|
|
||||||
|
threading.Thread(target=_startup_reconcile, name="startup-reconcile", daemon=True).start()
|
||||||
|
|
||||||
app = FastAPI(title="yt-scraper platform", version="1.0.0")
|
app = FastAPI(title="yt-scraper platform", version="1.0.0")
|
||||||
|
|
||||||
|
# StaticFiles/FileResponse send etag + last-modified but no Cache-Control,
|
||||||
|
# so a browser may heuristically cache an old app.js across updates and
|
||||||
|
# run yesterday's JS against a new backend. "no-cache" forces
|
||||||
|
# revalidation on every load (cheap 304s on a local app) while keeping
|
||||||
|
# the etag benefits.
|
||||||
|
@app.middleware("http")
|
||||||
|
async def _revalidate_shell(request, call_next):
|
||||||
|
response = await call_next(request)
|
||||||
|
path = request.url.path
|
||||||
|
if path == "/" or path.startswith("/static/"):
|
||||||
|
response.headers["Cache-Control"] = "no-cache"
|
||||||
|
return response
|
||||||
|
|
||||||
@app.get("/healthz")
|
@app.get("/healthz")
|
||||||
def healthz():
|
def healthz():
|
||||||
return {"status": "ok"}
|
return {"status": "ok", "reconcile_done": reconcile_done.is_set()}
|
||||||
|
|
||||||
jobs = JobManager(store, cfg)
|
jobs = JobManager(store, cfg)
|
||||||
app.state.jobs = jobs
|
app.state.jobs = jobs
|
||||||
|
|||||||
@@ -9,12 +9,12 @@ from collections import deque
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
from ..config import Config, load_config, parse_languages
|
from ..config import Config, keep_ref, load_config, parse_languages
|
||||||
from ..cookies import resolve_active_path
|
from ..cookies import resolve_active_path
|
||||||
from ..discover import discover_incremental
|
from ..discover import discover_incremental, extract_handle
|
||||||
from ..pipeline import process_video
|
from ..pipeline import process_video
|
||||||
from ..ratelimit import ThrottleGuard, polite_sleep
|
from ..ratelimit import ThrottleGuard, polite_sleep
|
||||||
from ..store import Store, VideoRef
|
from ..store import Store, order_pending
|
||||||
|
|
||||||
|
|
||||||
class JobManager:
|
class JobManager:
|
||||||
@@ -269,6 +269,7 @@ class JobManager:
|
|||||||
"sleep_interval_requests": cfg.yt_dlp.sleep_subrequests,
|
"sleep_interval_requests": cfg.yt_dlp.sleep_subrequests,
|
||||||
"retries": cfg.yt_dlp.retries,
|
"retries": cfg.yt_dlp.retries,
|
||||||
"socket_timeout": 30.0,
|
"socket_timeout": 30.0,
|
||||||
|
"js_runtimes": {"node": {}, "deno": {}, "bun": {}, "quickjs": {}},
|
||||||
}
|
}
|
||||||
if cfg.delay.audio_rate_limit:
|
if cfg.delay.audio_rate_limit:
|
||||||
ydl_opts["ratelimit"] = cfg.delay.audio_rate_limit
|
ydl_opts["ratelimit"] = cfg.delay.audio_rate_limit
|
||||||
@@ -453,7 +454,7 @@ class JobManager:
|
|||||||
max_window=sync.max_window,
|
max_window=sync.max_window,
|
||||||
overlap=sync.overlap,
|
overlap=sync.overlap,
|
||||||
since=self.store.latest_upload_date(channel_id) if incremental else None,
|
since=self.store.latest_upload_date(channel_id) if incremental else None,
|
||||||
keep=_keep_ref(cfg, include_shorts, no_live),
|
keep=keep_ref(cfg, include_shorts, no_live),
|
||||||
)
|
)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
self.store.update_job(job_id, status="error", last_error=str(exc), finished=True)
|
self.store.update_job(job_id, status="error", last_error=str(exc), finished=True)
|
||||||
@@ -478,7 +479,7 @@ class JobManager:
|
|||||||
from ..discover import deep_channel_avatar
|
from ..discover import deep_channel_avatar
|
||||||
avatar = deep_channel_avatar(channel_url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
|
avatar = deep_channel_avatar(channel_url, sleep_subrequests=cfg.yt_dlp.sleep_subrequests)
|
||||||
if result.full_scan:
|
if result.full_scan:
|
||||||
self.store.upsert_channel(_channel_id, _handle(channel_url), channel_name, len(refs), avatar=avatar)
|
self.store.upsert_channel(_channel_id, extract_handle(channel_url), channel_name, len(refs), avatar=avatar)
|
||||||
else:
|
else:
|
||||||
self.store.update_channel_meta(_channel_id, name=channel_name, avatar=avatar)
|
self.store.update_channel_meta(_channel_id, name=channel_name, avatar=avatar)
|
||||||
if avatar:
|
if avatar:
|
||||||
@@ -489,8 +490,8 @@ class JobManager:
|
|||||||
|
|
||||||
# Process the freshly-seen window first, then the older backlog, so a
|
# Process the freshly-seen window first, then the older backlog, so a
|
||||||
# `limit` still means "the newest N" now that discovery stops early.
|
# `limit` still means "the newest N" now that discovery stops early.
|
||||||
pending = _order_pending(self.store.get_pending(_channel_id), refs)
|
pending = order_pending(self.store.get_pending(_channel_id), refs)
|
||||||
pending = [r for r in pending if _keep_ref(cfg, include_shorts, no_live)(r)]
|
pending = [r for r in pending if keep_ref(cfg, include_shorts, no_live)(r)]
|
||||||
if since:
|
if since:
|
||||||
cutoff = since.replace("-", "")
|
cutoff = since.replace("-", "")
|
||||||
pending = [r for r in pending if not r.upload_date or r.upload_date >= cutoff]
|
pending = [r for r in pending if not r.upload_date or r.upload_date >= cutoff]
|
||||||
@@ -646,12 +647,12 @@ class JobManager:
|
|||||||
max_window=sync.max_window,
|
max_window=sync.max_window,
|
||||||
overlap=sync.overlap,
|
overlap=sync.overlap,
|
||||||
since=self.store.latest_upload_date(channel_id) if incremental else None,
|
since=self.store.latest_upload_date(channel_id) if incremental else None,
|
||||||
keep=_keep_ref(self.cfg),
|
keep=keep_ref(self.cfg),
|
||||||
)
|
)
|
||||||
refs = result.refs
|
refs = result.refs
|
||||||
|
|
||||||
if result.full_scan:
|
if result.full_scan:
|
||||||
self.store.upsert_channel(result.channel_id, _handle(channel_url), result.channel_name, len(refs))
|
self.store.upsert_channel(result.channel_id, extract_handle(channel_url), result.channel_name, len(refs))
|
||||||
new_videos = self.store.upsert_videos(refs)
|
new_videos = self.store.upsert_videos(refs)
|
||||||
else:
|
else:
|
||||||
self.store.update_channel_meta(result.channel_id, name=result.channel_name)
|
self.store.update_channel_meta(result.channel_id, name=result.channel_name)
|
||||||
@@ -669,41 +670,6 @@ class JobManager:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def _keep_ref(cfg: Config, include_shorts: bool | None = None, no_live: bool | None = None):
|
|
||||||
"""Predicate matching the shorts/live rules that decide what reaches the DB.
|
|
||||||
|
|
||||||
Incremental discovery needs the same filter its stored ids were created
|
|
||||||
under, otherwise the tail of a window is full of entries that can never be
|
|
||||||
recognised as known and the window keeps widening for nothing.
|
|
||||||
"""
|
|
||||||
shorts = cfg.include_shorts if include_shorts is None else include_shorts
|
|
||||||
skip_live = (not cfg.include_live) if no_live is None else no_live
|
|
||||||
|
|
||||||
def keep(r: Any) -> bool: # VideoRef or VideoRow — both carry `.url`
|
|
||||||
url = r.url or ""
|
|
||||||
if not shorts and "/shorts/" in url:
|
|
||||||
return False
|
|
||||||
if skip_live and url.startswith("https://www.youtube.com/live/"):
|
|
||||||
return False
|
|
||||||
return True
|
|
||||||
|
|
||||||
return keep
|
|
||||||
|
|
||||||
|
|
||||||
def _order_pending(rows: list, refs: list[VideoRef]) -> list:
|
|
||||||
"""Newest-window-first ordering for the pending queue.
|
|
||||||
|
|
||||||
`get_pending` is ordered by discovery time, which used to coincide with
|
|
||||||
newest-first because discovery saw the whole channel at once. With windowed
|
|
||||||
sync that no longer holds, so put the videos from this run's window at the
|
|
||||||
front and keep the rest of the backlog behind them.
|
|
||||||
"""
|
|
||||||
by_id = {r.video_id: r for r in rows}
|
|
||||||
ordered = [by_id.pop(x.video_id) for x in refs if x.video_id in by_id]
|
|
||||||
ordered.extend(by_id.values())
|
|
||||||
return ordered
|
|
||||||
|
|
||||||
|
|
||||||
def _clone_config(cfg: Config) -> Config:
|
def _clone_config(cfg: Config) -> Config:
|
||||||
import copy
|
import copy
|
||||||
return copy.deepcopy(cfg)
|
return copy.deepcopy(cfg)
|
||||||
@@ -721,12 +687,6 @@ def _resolve_channel_url(store: Store, cfg: Config, channel_id: str | None) -> s
|
|||||||
return f"https://www.youtube.com/channel/{channel_id}"
|
return f"https://www.youtube.com/channel/{channel_id}"
|
||||||
|
|
||||||
|
|
||||||
def _handle(url: str) -> str:
|
|
||||||
if "@" in url:
|
|
||||||
return "@" + url.split("@", 1)[1].split("/", 1)[0]
|
|
||||||
return ""
|
|
||||||
|
|
||||||
|
|
||||||
def _safe_video_filename(title: str) -> str:
|
def _safe_video_filename(title: str) -> str:
|
||||||
"""Keep the displayed title while making a valid, bounded Windows name."""
|
"""Keep the displayed title while making a valid, bounded Windows name."""
|
||||||
invalid = set(r'\\/:*?"<>|')
|
invalid = set(r'\\/:*?"<>|')
|
||||||
|
|||||||
@@ -92,19 +92,37 @@
|
|||||||
const size = Number(localStorage.getItem("videos-size"));
|
const size = Number(localStorage.getItem("videos-size"));
|
||||||
if ([10, 25, 50, 100].includes(size)) this.videos.size = size;
|
if ([10, 25, 50, 100].includes(size)) this.videos.size = size;
|
||||||
this.hydrateURL();
|
this.hydrateURL();
|
||||||
|
// Every view transition pushes a real history entry, so the browser's
|
||||||
|
// back/forward buttons have to drive the SPA through popstate.
|
||||||
|
window.addEventListener("popstate", () => this._onPopState());
|
||||||
|
// Global shortcuts ("/" jumps to search) ride the same window wiring.
|
||||||
|
window.addEventListener("keydown", (e) => this._onGlobalKeydown(e));
|
||||||
this.checkHealth();
|
this.checkHealth();
|
||||||
this.loadDashboard();
|
this.loadDashboard();
|
||||||
this.loadChannels();
|
this.loadChannels();
|
||||||
if (this.view === "videos") this.loadVideos(this.videos.page);
|
if (this.view === "videos") this.loadVideos(this.videos.page);
|
||||||
if (this.view === "search" && this.search.q) this.loadSearch();
|
if (this.view === "search" && this.search.q) this.loadSearch();
|
||||||
|
if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id, { push: false });
|
||||||
this.loadRetryable();
|
this.loadRetryable();
|
||||||
this.startLivePolling();
|
this.startLivePolling();
|
||||||
},
|
},
|
||||||
|
|
||||||
hydrateURL() {
|
hydrateURL() {
|
||||||
const p = new URLSearchParams(window.location.search);
|
this._fromParams(new URLSearchParams(window.location.search));
|
||||||
|
},
|
||||||
|
|
||||||
|
// Parse the query string into component state. Shared by the initial
|
||||||
|
// load and by popstate so browser back/forward restore the exact view
|
||||||
|
// (including filters). Resets to defaults first because popstate can go
|
||||||
|
// from a filtered list back to an unfiltered one.
|
||||||
|
_fromParams(p) {
|
||||||
const views = ["dashboard", "channels", "videos", "detail", "search", "analysis", "scrape", "cookies", "tools", "export"];
|
const views = ["dashboard", "channels", "videos", "detail", "search", "analysis", "scrape", "cookies", "tools", "export"];
|
||||||
if (views.includes(p.get("view"))) this.view = p.get("view");
|
const v = p.get("view");
|
||||||
|
this.view = views.includes(v) ? v : "dashboard";
|
||||||
|
if (this.view === "detail" && !p.get("video")) this.view = "videos";
|
||||||
|
this.filters = { channel: "", status: "", from: "", to: "", min_dur: "", q: "", sort: "upload_date" };
|
||||||
|
this.search.q = "";
|
||||||
|
this.search.channel = "";
|
||||||
if (this.view === "videos") {
|
if (this.view === "videos") {
|
||||||
["channel", "status", "from", "to", "min_dur", "q", "sort"].forEach(k => { if (p.has(k)) this.filters[k] = p.get(k); });
|
["channel", "status", "from", "to", "min_dur", "q", "sort"].forEach(k => { if (p.has(k)) this.filters[k] = p.get(k); });
|
||||||
const page = Number(p.get("page"));
|
const page = Number(p.get("page"));
|
||||||
@@ -116,11 +134,11 @@
|
|||||||
this.search.q = p.get("q") || "";
|
this.search.q = p.get("q") || "";
|
||||||
this.search.channel = p.get("channel") || "";
|
this.search.channel = p.get("channel") || "";
|
||||||
}
|
}
|
||||||
|
// A stub is enough for init/popstate to know which video to load.
|
||||||
|
this.detail.video = this.view === "detail" ? { video_id: p.get("video") } : null;
|
||||||
},
|
},
|
||||||
|
|
||||||
syncURL() {
|
_urlParams() {
|
||||||
clearTimeout(this._urlTimer);
|
|
||||||
this._urlTimer = setTimeout(() => {
|
|
||||||
const p = new URLSearchParams({ view: this.view });
|
const p = new URLSearchParams({ view: this.view });
|
||||||
if (this.view === "videos") {
|
if (this.view === "videos") {
|
||||||
Object.entries(this.filters).forEach(([k, v]) => { if (v) p.set(k, v); });
|
Object.entries(this.filters).forEach(([k, v]) => { if (v) p.set(k, v); });
|
||||||
@@ -128,12 +146,45 @@
|
|||||||
} else if (this.view === "search") {
|
} else if (this.view === "search") {
|
||||||
if (this.search.q) p.set("q", this.search.q);
|
if (this.search.q) p.set("q", this.search.q);
|
||||||
if (this.search.channel) p.set("channel", this.search.channel);
|
if (this.search.channel) p.set("channel", this.search.channel);
|
||||||
|
} else if (this.view === "detail" && this.detail.video) {
|
||||||
|
p.set("video", this.detail.video.video_id);
|
||||||
}
|
}
|
||||||
const next = "?" + p.toString();
|
return p;
|
||||||
if (next !== window.location.search) history.replaceState(null, "", next);
|
},
|
||||||
|
|
||||||
|
syncURL() {
|
||||||
|
clearTimeout(this._urlTimer);
|
||||||
|
this._urlTimer = setTimeout(() => {
|
||||||
|
const next = "?" + this._urlParams().toString();
|
||||||
|
if (next !== window.location.search) history.replaceState(history.state, "", next);
|
||||||
}, 120);
|
}, 120);
|
||||||
},
|
},
|
||||||
|
|
||||||
|
// View transitions push a history entry so browser back stays inside
|
||||||
|
// the app; param tweaks (filters, pagination) only replace via syncURL.
|
||||||
|
pushURL(extra) {
|
||||||
|
clearTimeout(this._urlTimer);
|
||||||
|
const p = this._urlParams();
|
||||||
|
if (extra && extra.video) p.set("video", extra.video);
|
||||||
|
const next = "?" + p.toString();
|
||||||
|
if (next === window.location.search) return;
|
||||||
|
history.pushState({ view: p.get("view") }, "", next);
|
||||||
|
this._pushedEntries = (this._pushedEntries || 0) + 1;
|
||||||
|
},
|
||||||
|
|
||||||
|
_onPopState() {
|
||||||
|
if (this._pushedEntries > 0) this._pushedEntries--;
|
||||||
|
this._fromParams(new URLSearchParams(window.location.search));
|
||||||
|
if (this.view === "detail") {
|
||||||
|
const id = (this.detail.video && this.detail.video.video_id) || "";
|
||||||
|
if (id && id !== this._detailLoadedId) this.openVideo(id, { push: false });
|
||||||
|
} else {
|
||||||
|
this._detailLoadedId = null;
|
||||||
|
this._armScrollRestore(this.view);
|
||||||
|
this._loadView(this.view);
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
toggleSidebar() {
|
toggleSidebar() {
|
||||||
this.sidebarOpen = !this.sidebarOpen;
|
this.sidebarOpen = !this.sidebarOpen;
|
||||||
localStorage.setItem("sidebar-open", String(this.sidebarOpen));
|
localStorage.setItem("sidebar-open", String(this.sidebarOpen));
|
||||||
@@ -177,10 +228,18 @@
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
|
||||||
setView(v) {
|
setView(v, opts) {
|
||||||
|
const o = opts || {};
|
||||||
this.view = v;
|
this.view = v;
|
||||||
this.syncURL();
|
if (o.push === false) this.syncURL();
|
||||||
|
else this.pushURL();
|
||||||
if (window.innerWidth < 768) this.closeSidebar();
|
if (window.innerWidth < 768) this.closeSidebar();
|
||||||
|
if (!o.skipLoad) this._loadView(v);
|
||||||
|
},
|
||||||
|
|
||||||
|
// Data loads that accompany entering a view. Extracted from setView so
|
||||||
|
// popstate can run them without pushing a new history entry.
|
||||||
|
_loadView(v) {
|
||||||
// destroy stray canvases when leaving chart-bearing views
|
// destroy stray canvases when leaving chart-bearing views
|
||||||
if (v !== "analysis") this.destroyCharts(["topwords", "timeline"]);
|
if (v !== "analysis") this.destroyCharts(["topwords", "timeline"]);
|
||||||
if (v === "dashboard") this.loadDashboard();
|
if (v === "dashboard") this.loadDashboard();
|
||||||
@@ -194,6 +253,26 @@
|
|||||||
if (v === "tools") { this.loadFolders(); this.loadFormat(); }
|
if (v === "tools") { this.loadFolders(); this.loadFormat(); }
|
||||||
},
|
},
|
||||||
|
|
||||||
|
// One-shot scroll restore, armed ONLY by the back-navigation paths
|
||||||
|
// (_onPopState and goBack's fallback). Ordinary list interactions
|
||||||
|
// (pagination, filter changes, openChannel) never set it, so they keep
|
||||||
|
// landing at the top like a fresh view.
|
||||||
|
// 'videos': the table is fetched asynchronously, and scrolling before
|
||||||
|
// the fetch resolves gets clamped to the top of an empty table — so
|
||||||
|
// the flag is consumed in loadVideos()'s finally, after the rows exist.
|
||||||
|
// 'search': popstate does not reload results (they stay in memory), so
|
||||||
|
// restore right after the view switch; nextTick lets Alpine render the
|
||||||
|
// results and the extra requestAnimationFrame waits for layout, so the
|
||||||
|
// browser does not clamp the scroll to a not-yet-painted page.
|
||||||
|
_armScrollRestore(view) {
|
||||||
|
if (view === "videos") {
|
||||||
|
this._pendingScrollRestore = "videos";
|
||||||
|
} else if (view === "search") {
|
||||||
|
const top = (this._scrollMemory && this._scrollMemory.search) || 0;
|
||||||
|
this.$nextTick(() => requestAnimationFrame(() => window.scrollTo({ top })));
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
// =================================================================
|
// =================================================================
|
||||||
// dashboard
|
// dashboard
|
||||||
// =================================================================
|
// =================================================================
|
||||||
@@ -323,14 +402,46 @@
|
|||||||
const d = await this.api("/api/videos?" + p.toString());
|
const d = await this.api("/api/videos?" + p.toString());
|
||||||
this.videos = Object.assign({}, this.videos, { items: d.items, total: d.total, page: d.page, size: d.size });
|
this.videos = Object.assign({}, this.videos, { items: d.items, total: d.total, page: d.page, size: d.size });
|
||||||
} catch (e) { this.toast("Failed to load videos: " + e.message, "error"); }
|
} catch (e) { this.toast("Failed to load videos: " + e.message, "error"); }
|
||||||
finally { this.loading.videos = false; }
|
finally {
|
||||||
|
this.loading.videos = false;
|
||||||
|
// Back-navigation restore. It must happen here, after the fetch has
|
||||||
|
// resolved — scrolling any earlier is clamped to the top because the
|
||||||
|
// table is still empty while the request is in flight. The nextTick
|
||||||
|
// waits for Alpine to drop the transient "loading…" row (it renders
|
||||||
|
// above the data rows), so the position matches the row the user
|
||||||
|
// actually left. The flag is cleared on EVERY run, not only when it
|
||||||
|
// fired, so a stale one can never affect an unrelated loadVideos
|
||||||
|
// (pagination, filter change).
|
||||||
|
if (this._pendingScrollRestore === "videos") {
|
||||||
|
const top = (this._scrollMemory && this._scrollMemory.videos) || 0;
|
||||||
|
this.$nextTick(() => window.scrollTo({ top }));
|
||||||
|
}
|
||||||
|
this._pendingScrollRestore = null;
|
||||||
|
}
|
||||||
},
|
},
|
||||||
|
|
||||||
async openVideo(id) {
|
async openVideo(id, opts) {
|
||||||
|
const o = opts || {};
|
||||||
|
// Remember where the user came from so the back button can label
|
||||||
|
// itself honestly. Skipped when restoring from history (popstate) or
|
||||||
|
// auto-refreshing after a job, where the origin must not change.
|
||||||
|
if (o.push !== false) {
|
||||||
|
if (this.view !== "detail") {
|
||||||
|
this.detail.fromView = this.view;
|
||||||
|
// Also remember how far down that view the user had scrolled,
|
||||||
|
// so returning from the detail can put them back on the same
|
||||||
|
// row instead of at the top of a reloaded table.
|
||||||
|
if (!this._scrollMemory) this._scrollMemory = {};
|
||||||
|
this._scrollMemory[this.view] = window.scrollY;
|
||||||
|
}
|
||||||
|
this.view = "detail"; // before pushURL: the entry must say where we went
|
||||||
|
this.pushURL({ video: id });
|
||||||
|
}
|
||||||
this.view = "detail";
|
this.view = "detail";
|
||||||
|
this._detailLoadedId = id;
|
||||||
this.loading.detail = true;
|
this.loading.detail = true;
|
||||||
this.loading.transcript = true;
|
this.loading.transcript = true;
|
||||||
this.detail = { video: null, transcript: [], hasAudio: false, audioPlaying: false, activeSeg: -1, videoError: false };
|
this.detail = { video: null, transcript: [], hasAudio: false, audioPlaying: false, activeSeg: -1, videoError: false, fromView: this.detail.fromView };
|
||||||
try {
|
try {
|
||||||
const v = await this.api("/api/videos/" + encodeURIComponent(id));
|
const v = await this.api("/api/videos/" + encodeURIComponent(id));
|
||||||
// chapters may come embedded or be absent
|
// chapters may come embedded or be absent
|
||||||
@@ -347,6 +458,30 @@
|
|||||||
this.checkAudio(id);
|
this.checkAudio(id);
|
||||||
},
|
},
|
||||||
|
|
||||||
|
// The single back affordance for the detail view: prefer real browser
|
||||||
|
// history (restores the list's filters via the URL), and fall back to
|
||||||
|
// the origin view when the detail URL was opened directly (refresh or
|
||||||
|
// shared link) and there is no in-app history to return to.
|
||||||
|
goBack() {
|
||||||
|
if ((this._pushedEntries || 0) > 0) { history.back(); return; }
|
||||||
|
const target = this.detail.fromView && this.detail.fromView !== "detail" ? this.detail.fromView : "videos";
|
||||||
|
// No in-app history to history.back() into — the popstate handler
|
||||||
|
// never fires, so arm the same scroll restore before the fallback
|
||||||
|
// setView() triggers the view's loaders.
|
||||||
|
this._armScrollRestore(target);
|
||||||
|
this.setView(target);
|
||||||
|
},
|
||||||
|
|
||||||
|
get backLabel() {
|
||||||
|
const map = {
|
||||||
|
search: "Back to search results",
|
||||||
|
dashboard: "Back to dashboard",
|
||||||
|
channels: "Back to channels",
|
||||||
|
videos: "Back to videos",
|
||||||
|
};
|
||||||
|
return map[this.detail.fromView] || "Back to videos";
|
||||||
|
},
|
||||||
|
|
||||||
async checkAudio(id) {
|
async checkAudio(id) {
|
||||||
try {
|
try {
|
||||||
const r = await fetch("/api/videos/" + encodeURIComponent(id) + "/audio", { method: "HEAD" });
|
const r = await fetch("/api/videos/" + encodeURIComponent(id) + "/audio", { method: "HEAD" });
|
||||||
@@ -384,7 +519,7 @@
|
|||||||
openVideoPlayer() {
|
openVideoPlayer() {
|
||||||
this.detail.videoError = false;
|
this.detail.videoError = false;
|
||||||
this.$nextTick(() => {
|
this.$nextTick(() => {
|
||||||
const player = this.$refs.videoPlayer;
|
const player = this.ref("videoPlayer");
|
||||||
if (player) { player.load(); player.play().catch(() => {}); }
|
if (player) { player.load(); player.play().catch(() => {}); }
|
||||||
});
|
});
|
||||||
},
|
},
|
||||||
@@ -426,7 +561,7 @@
|
|||||||
},
|
},
|
||||||
|
|
||||||
seekAudio(sec) {
|
seekAudio(sec) {
|
||||||
const a = this.$refs.audioPlayer;
|
const a = this.ref("audioPlayer");
|
||||||
if (!a) return false;
|
if (!a) return false;
|
||||||
try { a.currentTime = sec; a.play().catch(() => {}); } catch (_) {}
|
try { a.currentTime = sec; a.play().catch(() => {}); } catch (_) {}
|
||||||
return true;
|
return true;
|
||||||
@@ -434,7 +569,7 @@
|
|||||||
|
|
||||||
seekTo(sec) {
|
seekTo(sec) {
|
||||||
// seek the local audio player if available; otherwise just scroll the transcript
|
// seek the local audio player if available; otherwise just scroll the transcript
|
||||||
if (this.detail.hasAudio && this.$refs.audioPlayer) {
|
if (this.detail.hasAudio && this.ref("audioPlayer")) {
|
||||||
if (this.seekAudio(sec)) return;
|
if (this.seekAudio(sec)) return;
|
||||||
}
|
}
|
||||||
this.seekTranscript(sec);
|
this.seekTranscript(sec);
|
||||||
@@ -477,6 +612,18 @@
|
|||||||
},
|
},
|
||||||
toggleSelectAll() { if (this.allSelected) this.selectNone(); else this.selectAll(); },
|
toggleSelectAll() { if (this.allSelected) this.selectNone(); else this.selectAll(); },
|
||||||
|
|
||||||
|
// Active-filter chips: removing one must be one click, not a hunt
|
||||||
|
// through the selects — especially the channel filter set by
|
||||||
|
// openChannel(), which used to feel like a trap.
|
||||||
|
clearFilter(key) {
|
||||||
|
this.filters[key] = "";
|
||||||
|
this.loadVideos(1);
|
||||||
|
},
|
||||||
|
clearAllFilters() {
|
||||||
|
this.filters = { channel: "", status: "", from: "", to: "", min_dur: "", q: "", sort: "upload_date" };
|
||||||
|
this.loadVideos(1);
|
||||||
|
},
|
||||||
|
|
||||||
// =================================================================
|
// =================================================================
|
||||||
// downloads (.md / thumbnails / audio / single)
|
// downloads (.md / thumbnails / audio / single)
|
||||||
// =================================================================
|
// =================================================================
|
||||||
@@ -502,6 +649,27 @@
|
|||||||
this.downloadFile("/api/videos/" + encodeURIComponent(videoId) + "/markdown");
|
this.downloadFile("/api/videos/" + encodeURIComponent(videoId) + "/markdown");
|
||||||
},
|
},
|
||||||
|
|
||||||
|
async copyMd(videoId) {
|
||||||
|
try {
|
||||||
|
const res = await fetch("/api/videos/" + encodeURIComponent(videoId) + "/markdown");
|
||||||
|
if (!res.ok) throw new Error("Could not fetch markdown file");
|
||||||
|
const text = await res.text();
|
||||||
|
await navigator.clipboard.writeText(text);
|
||||||
|
this.toast("Contenido .md copiado al portapapeles");
|
||||||
|
} catch (e) {
|
||||||
|
this.toast("Error al copiar .md: " + e.message, "error");
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
|
async openMd(videoId) {
|
||||||
|
try {
|
||||||
|
await this.api("/api/videos/" + encodeURIComponent(videoId) + "/open-markdown", { method: "POST" });
|
||||||
|
this.toast("Abriendo archivo .md...");
|
||||||
|
} catch (e) {
|
||||||
|
this.toast("Error al abrir .md: " + e.message, "error");
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
async processOne(videoId) {
|
async processOne(videoId) {
|
||||||
try {
|
try {
|
||||||
const d = await this.api("/api/scrape/video/" + encodeURIComponent(videoId), { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({}) });
|
const d = await this.api("/api/scrape/video/" + encodeURIComponent(videoId), { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({}) });
|
||||||
@@ -681,9 +849,51 @@
|
|||||||
if (tag === "INPUT" || tag === "TEXTAREA" || (e.target && e.target.isContentEditable)) return;
|
if (tag === "INPUT" || tag === "TEXTAREA" || (e.target && e.target.isContentEditable)) return;
|
||||||
if (this.stats.open) { this.closeStats(); return; }
|
if (this.stats.open) { this.closeStats(); return; }
|
||||||
if (this.clip.open) { this.closeClip(); return; }
|
if (this.clip.open) { this.closeClip(); return; }
|
||||||
|
// Detail is the deepest layer of the app, so Esc backs out of it —
|
||||||
|
// same gesture that closes every modal. Guarded against the Esc that
|
||||||
|
// merely exits fullscreen playback.
|
||||||
|
if (this.view === "detail" && !document.fullscreenElement) { this.goBack(); return; }
|
||||||
if (this.cookies && this.cookies.drag) { this.cookies.drag = false; return; }
|
if (this.cookies && this.cookies.drag) { this.cookies.drag = false; return; }
|
||||||
},
|
},
|
||||||
|
|
||||||
|
// Alpine 3.17 mounts template x-if content under its own scope, so
|
||||||
|
// x-ref elements never reach the root component's $refs — and every
|
||||||
|
// view here lives inside a template x-if. Resolve refs from the DOM
|
||||||
|
// instead; $refs stays as the fast path for anything ever registered.
|
||||||
|
ref(name) {
|
||||||
|
return (this.$refs && this.$refs[name]) || document.querySelector('#root [x-ref="' + name + '"]');
|
||||||
|
},
|
||||||
|
|
||||||
|
// "/" anywhere (outside form fields) jumps to the transcript search
|
||||||
|
// with the query input focused. Guarded like onGlobalEscape so it
|
||||||
|
// never fires while typing, composing or under a modal.
|
||||||
|
_onGlobalKeydown(e) {
|
||||||
|
if (e.key !== "/") return;
|
||||||
|
// IME composition and modifier combos belong to the browser/OS,
|
||||||
|
// not to this shortcut.
|
||||||
|
if (e.isComposing || e.ctrlKey || e.altKey || e.metaKey || e.shiftKey) return;
|
||||||
|
// Let editable fields receive a literal "/" instead of stealing it.
|
||||||
|
const el = document.activeElement;
|
||||||
|
const tag = el && el.tagName;
|
||||||
|
if (tag === "INPUT" || tag === "TEXTAREA" || tag === "SELECT" || (el && el.isContentEditable)) return;
|
||||||
|
// Modals own the keyboard while they are open.
|
||||||
|
if (this.confirmBox.open || this.stats.open || this.clip.open) return;
|
||||||
|
// preventDefault so the "/" never lands in the input we focus next.
|
||||||
|
e.preventDefault();
|
||||||
|
if (this.view !== "search") this.setView("search");
|
||||||
|
this._focusSearchInput(10);
|
||||||
|
},
|
||||||
|
|
||||||
|
// The search view is a template x-if whose content mounts a beat after
|
||||||
|
// the reactive flush, so a single $nextTick can query before the input
|
||||||
|
// exists. Retry across a few animation frames and focus as soon as it
|
||||||
|
// renders (no perceptible delay when it is already there).
|
||||||
|
_focusSearchInput(retries) {
|
||||||
|
const input = this.ref("searchInput");
|
||||||
|
if (input) { input.focus(); return; }
|
||||||
|
if (retries > 0) requestAnimationFrame(() => this._focusSearchInput(retries - 1));
|
||||||
|
},
|
||||||
|
|
||||||
// =================================================================
|
// =================================================================
|
||||||
// channels: pending download
|
// channels: pending download
|
||||||
// =================================================================
|
// =================================================================
|
||||||
@@ -1007,7 +1217,10 @@
|
|||||||
// Scrape tab (where startScrape lives) refreshed nothing at all, and
|
// Scrape tab (where startScrape lives) refreshed nothing at all, and
|
||||||
// the setView cache guard then kept the stale rows on navigation.
|
// the setView cache guard then kept the stale rows on navigation.
|
||||||
this.refreshLiveState({ force: true });
|
this.refreshLiveState({ force: true });
|
||||||
if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id);
|
// Refresh the open detail view after a job, without pushing a new
|
||||||
|
// history entry — Back must still return to where the user came
|
||||||
|
// from, not to a duplicate of the same video.
|
||||||
|
if (this.view === "detail" && this.detail.video) this.openVideo(this.detail.video.video_id, { push: false });
|
||||||
this._armAutoHide();
|
this._armAutoHide();
|
||||||
});
|
});
|
||||||
es.addEventListener("cancelled", () => { this.scrape.log.push("[cancelled]"); this.closeStream(); this.loadJobs(); this._armAutoHide(); });
|
es.addEventListener("cancelled", () => { this.scrape.log.push("[cancelled]"); this.closeStream(); this.loadJobs(); this._armAutoHide(); });
|
||||||
@@ -1132,7 +1345,7 @@
|
|||||||
if (!files.length) { this.toast("Drop .txt cookie files only", "error"); return; }
|
if (!files.length) { this.toast("Drop .txt cookie files only", "error"); return; }
|
||||||
this.uploadCookies(files);
|
this.uploadCookies(files);
|
||||||
// reset the file input so the same file can be picked again
|
// reset the file input so the same file can be picked again
|
||||||
try { if (this.$refs.cookieFile) this.$refs.cookieFile.value = ""; } catch (_) {}
|
try { const input = this.ref("cookieFile"); if (input) input.value = ""; } catch (_) {}
|
||||||
},
|
},
|
||||||
|
|
||||||
async uploadCookies(files) {
|
async uploadCookies(files) {
|
||||||
@@ -1148,6 +1361,24 @@
|
|||||||
} catch (e) { this.toast("Upload failed: " + e.message, "error"); }
|
} catch (e) { this.toast("Upload failed: " + e.message, "error"); }
|
||||||
},
|
},
|
||||||
|
|
||||||
|
async importBrowserCookies() {
|
||||||
|
if (this.loading.cookies) return;
|
||||||
|
this.loading.cookies = true;
|
||||||
|
try {
|
||||||
|
const d = await this.api("/api/cookies/from-browser", {
|
||||||
|
method: "POST",
|
||||||
|
headers: { "Content-Type": "application/json" },
|
||||||
|
body: JSON.stringify({ browser: "brave" }),
|
||||||
|
});
|
||||||
|
this.toast("Imported " + (d.cookie_count || "") + " cookies from Brave and activated them");
|
||||||
|
await this.loadCookies();
|
||||||
|
} catch (e) {
|
||||||
|
this.toast("Brave import failed: " + e.message, "error");
|
||||||
|
} finally {
|
||||||
|
this.loading.cookies = false;
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
async activateCookie(id) {
|
async activateCookie(id) {
|
||||||
try { await this.api("/api/cookies/" + encodeURIComponent(id) + "/activate", { method: "POST" }); this.loadCookies(); }
|
try { await this.api("/api/cookies/" + encodeURIComponent(id) + "/activate", { method: "POST" }); this.loadCookies(); }
|
||||||
catch (e) { this.toast("Activate failed: " + e.message, "error"); }
|
catch (e) { this.toast("Activate failed: " + e.message, "error"); }
|
||||||
@@ -1189,7 +1420,9 @@
|
|||||||
// =================================================================
|
// =================================================================
|
||||||
openChannel(id) {
|
openChannel(id) {
|
||||||
this.filters.channel = id;
|
this.filters.channel = id;
|
||||||
this.setView("videos");
|
// skipLoad: loadVideos(1) below is the authoritative load — setView
|
||||||
|
// would trigger a second one with the stale page number.
|
||||||
|
this.setView("videos", { skipLoad: true });
|
||||||
this.loadVideos(1);
|
this.loadVideos(1);
|
||||||
},
|
},
|
||||||
|
|
||||||
|
|||||||
@@ -86,19 +86,23 @@
|
|||||||
<div class="grid grid-cols-2 md:grid-cols-4 gap-4">
|
<div class="grid grid-cols-2 md:grid-cols-4 gap-4">
|
||||||
<div class="glass p-4">
|
<div class="glass p-4">
|
||||||
<div class="text-xs text-zinc-500 uppercase tracking-wider">Channels</div>
|
<div class="text-xs text-zinc-500 uppercase tracking-wider">Channels</div>
|
||||||
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="dash.channels.length"></div>
|
<template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-12" aria-hidden="true"></div></template>
|
||||||
|
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="dash.channels.length"></div></template>
|
||||||
</div>
|
</div>
|
||||||
<div class="glass p-4">
|
<div class="glass p-4">
|
||||||
<div class="text-xs text-zinc-500 uppercase tracking-wider">Videos</div>
|
<div class="text-xs text-zinc-500 uppercase tracking-wider">Videos</div>
|
||||||
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.video_count||0),0))"></div>
|
<template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-16" aria-hidden="true"></div></template>
|
||||||
|
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.video_count||0),0))"></div></template>
|
||||||
</div>
|
</div>
|
||||||
<div class="glass p-4">
|
<div class="glass p-4">
|
||||||
<div class="text-xs text-zinc-500 uppercase tracking-wider">Total Views</div>
|
<div class="text-xs text-zinc-500 uppercase tracking-wider">Total Views</div>
|
||||||
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.total_views||0),0))"></div>
|
<template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-20" aria-hidden="true"></div></template>
|
||||||
|
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="fmtNum(dash.channels.reduce((s,c)=>s+(c.total_views||0),0))"></div></template>
|
||||||
</div>
|
</div>
|
||||||
<div class="glass p-4">
|
<div class="glass p-4">
|
||||||
<div class="text-xs text-zinc-500 uppercase tracking-wider">Total Duration</div>
|
<div class="text-xs text-zinc-500 uppercase tracking-wider">Total Duration</div>
|
||||||
<div class="mt-1 text-2xl font-bold text-white font-mono" x-text="humanDur(dash.channels.reduce((s,c)=>s+(c.total_duration||0),0))"></div>
|
<template x-if="loading.dashboard"><div class="skeleton mt-1 h-7 w-20" aria-hidden="true"></div></template>
|
||||||
|
<template x-if="!loading.dashboard"><div class="mt-1 text-2xl font-bold text-white font-mono" x-text="humanDur(dash.channels.reduce((s,c)=>s+(c.total_duration||0),0))"></div></template>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
@@ -185,7 +189,31 @@
|
|||||||
<table class="tbl">
|
<table class="tbl">
|
||||||
<thead><tr><th>Name</th><th>Handle</th><th>Videos</th><th>Pending</th><th>Latest video</th><th>Last scraped</th><th class="text-right">Actions</th></tr></thead>
|
<thead><tr><th>Name</th><th>Handle</th><th>Videos</th><th>Pending</th><th>Latest video</th><th>Last scraped</th><th class="text-right">Actions</th></tr></thead>
|
||||||
<tbody>
|
<tbody>
|
||||||
<template x-if="loading.channels"><tr><td colspan="7" class="text-center text-zinc-500 py-8">loading…</td></tr></template>
|
<template x-if="loading.channels">
|
||||||
|
<template x-for="i in 5" :key="i">
|
||||||
|
<tr>
|
||||||
|
<td>
|
||||||
|
<template x-if="i===1"><span class="sr-only" role="status">Loading channels…</span></template>
|
||||||
|
<div class="flex items-center gap-2" aria-hidden="true">
|
||||||
|
<div class="skeleton w-7 h-7 rounded-full"></div>
|
||||||
|
<div class="skeleton h-4 w-28"></div>
|
||||||
|
</div>
|
||||||
|
</td>
|
||||||
|
<td><div class="skeleton h-3.5 w-20" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3.5 w-10" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-5 w-20 rounded-full" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-16" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-16" aria-hidden="true"></div></td>
|
||||||
|
<td>
|
||||||
|
<div class="flex items-center justify-end gap-1" aria-hidden="true">
|
||||||
|
<div class="skeleton h-5 w-14"></div>
|
||||||
|
<div class="skeleton h-5 w-12"></div>
|
||||||
|
<div class="skeleton h-5 w-12"></div>
|
||||||
|
</div>
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
</template>
|
||||||
|
</template>
|
||||||
<template x-if="!loading.channels && channels.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-8">No channels tracked yet.</td></tr></template>
|
<template x-if="!loading.channels && channels.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-8">No channels tracked yet.</td></tr></template>
|
||||||
<template x-for="c in channels.items" :key="c.channel_id">
|
<template x-for="c in channels.items" :key="c.channel_id">
|
||||||
<tr class="row-hover">
|
<tr class="row-hover">
|
||||||
@@ -280,6 +308,25 @@
|
|||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
<!-- active filters: visible at a glance and one click to remove, so
|
||||||
|
a channel filter set from the dashboard never feels like a trap -->
|
||||||
|
<div x-show="filters.channel || filters.status" x-cloak class="flex flex-wrap items-center gap-2">
|
||||||
|
<span class="text-xs text-zinc-500">Showing</span>
|
||||||
|
<template x-if="filters.channel">
|
||||||
|
<button class="filter-chip" @click="clearFilter('channel')" title="Clear this filter and show all channels">
|
||||||
|
<span class="max-w-56 truncate" x-text="chanName(filters.channel)"></span>
|
||||||
|
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" aria-hidden="true"><path stroke-linecap="round" d="M6 6l12 12M18 6L6 18"/></svg>
|
||||||
|
</button>
|
||||||
|
</template>
|
||||||
|
<template x-if="filters.status">
|
||||||
|
<button class="filter-chip" @click="clearFilter('status')" title="Clear this status filter">
|
||||||
|
<span x-text="filters.status === '__blocked__' ? 'locked videos' : filters.status + ' videos'"></span>
|
||||||
|
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" aria-hidden="true"><path stroke-linecap="round" d="M6 6l12 12M18 6L6 18"/></svg>
|
||||||
|
</button>
|
||||||
|
</template>
|
||||||
|
<button class="text-xs text-zinc-400 hover:text-white underline underline-offset-2 transition-colors" @click="clearAllFilters()">Clear all</button>
|
||||||
|
</div>
|
||||||
|
|
||||||
<!-- bulk action bar -->
|
<!-- bulk action bar -->
|
||||||
<template x-if="videos.selected.length > 0">
|
<template x-if="videos.selected.length > 0">
|
||||||
<div class="bulk-bar glass p-3 flex flex-wrap items-center gap-2">
|
<div class="bulk-bar glass p-3 flex flex-wrap items-center gap-2">
|
||||||
@@ -320,8 +367,31 @@
|
|||||||
<th>Duration</th><th>Views</th><th>Status</th><th class="text-right">Actions</th>
|
<th>Duration</th><th>Views</th><th>Status</th><th class="text-right">Actions</th>
|
||||||
</tr></thead>
|
</tr></thead>
|
||||||
<tbody>
|
<tbody>
|
||||||
<template x-if="loading.videos"><tr><td colspan="9" class="text-center text-zinc-500 py-10">loading…</td></tr></template>
|
<!-- skeleton rows mirror the real columns; the sr-only status rides row 1 so each table has exactly one live region -->
|
||||||
<template x-if="!loading.videos && videos.items.length===0"><tr><td colspan="9" class="text-center text-zinc-500 py-10"><div>No videos match these filters.</div><button class="btn btn-ghost mt-3" @click.stop="filters={ channel:'', status:'', from:'', to:'', min_dur:'', q:'', sort:'upload_date' }; loadVideos(1)">Clear filters</button></td></tr></template>
|
<template x-if="loading.videos">
|
||||||
|
<template x-for="i in 8" :key="i">
|
||||||
|
<tr>
|
||||||
|
<td>
|
||||||
|
<template x-if="i===1"><span class="sr-only" role="status">Loading videos…</span></template>
|
||||||
|
<div class="skeleton w-4 h-4 rounded" aria-hidden="true"></div>
|
||||||
|
</td>
|
||||||
|
<td><div class="skeleton w-20 h-12 rounded-lg" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-4 w-full max-w-md" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-24" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-16" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-12" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-14" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-5 w-16 rounded-full" aria-hidden="true"></div></td>
|
||||||
|
<td>
|
||||||
|
<div class="flex items-center justify-end gap-1" aria-hidden="true">
|
||||||
|
<div class="skeleton h-5 w-12"></div>
|
||||||
|
<div class="skeleton h-5 w-10"></div>
|
||||||
|
</div>
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
</template>
|
||||||
|
</template>
|
||||||
|
<template x-if="!loading.videos && videos.items.length===0"><tr><td colspan="9" class="text-center text-zinc-500 py-10"><div>No videos match these filters.</div><button class="btn btn-ghost mt-3" @click.stop="clearAllFilters()">Clear filters</button></td></tr></template>
|
||||||
<template x-for="v in videos.items" :key="v.video_id">
|
<template x-for="v in videos.items" :key="v.video_id">
|
||||||
<tr class="row-hover" :class="isSelected(v.video_id) ? 'row-selected' : ''" @click="openVideo(v.video_id)">
|
<tr class="row-hover" :class="isSelected(v.video_id) ? 'row-selected' : ''" @click="openVideo(v.video_id)">
|
||||||
<td @click.stop><input type="checkbox" class="accent-rose-500" :checked="isSelected(v.video_id)" @change="toggleSelect(v.video_id)" /></td>
|
<td @click.stop><input type="checkbox" class="accent-rose-500" :checked="isSelected(v.video_id)" @change="toggleSelect(v.video_id)" /></td>
|
||||||
@@ -368,7 +438,12 @@
|
|||||||
<!-- ===== Detail ===== -->
|
<!-- ===== Detail ===== -->
|
||||||
<template x-if="view==='detail'">
|
<template x-if="view==='detail'">
|
||||||
<section class="space-y-5" x-transition.opacity>
|
<section class="space-y-5" x-transition.opacity>
|
||||||
<button class="btn btn-ghost" @click="setView('videos')">← Back to videos</button>
|
<div class="flex items-center justify-between gap-3 flex-wrap">
|
||||||
|
<button class="btn btn-ghost" @click="goBack()" title="Go back (or press Esc)">
|
||||||
|
<svg class="w-4 h-4 shrink-0" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" aria-hidden="true"><path stroke-linecap="round" stroke-linejoin="round" d="M15 19l-7-7 7-7"/></svg>
|
||||||
|
<span x-text="backLabel"></span>
|
||||||
|
</button>
|
||||||
|
</div>
|
||||||
<template x-if="loading.detail"><div class="glass p-10 text-center text-zinc-500">loading…</div></template>
|
<template x-if="loading.detail"><div class="glass p-10 text-center text-zinc-500">loading…</div></template>
|
||||||
<template x-if="!loading.detail && detail.video">
|
<template x-if="!loading.detail && detail.video">
|
||||||
<div class="space-y-5">
|
<div class="space-y-5">
|
||||||
@@ -400,6 +475,14 @@
|
|||||||
<div class="flex flex-wrap gap-2 mt-4">
|
<div class="flex flex-wrap gap-2 mt-4">
|
||||||
<button class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="startDiscovery(detail.video.channel_id)" :disabled="jobActive()" title="Find recent videos from this channel without downloading transcripts">Investigate channel</button>
|
<button class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="startDiscovery(detail.video.channel_id)" :disabled="jobActive()" title="Find recent videos from this channel without downloading transcripts">Investigate channel</button>
|
||||||
<button x-show="!isBlocked(detail.video) || detail.video.status==='done'" class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="detail.video.status==='done' ? downloadMd(detail.video.video_id) : processOne(detail.video.video_id)" x-text="detail.video.status==='done' ? 'Download .md' : 'Process to .md'"></button>
|
<button x-show="!isBlocked(detail.video) || detail.video.status==='done'" class="btn accent-grad btn-primary !py-1.5 !px-3 text-xs" @click="detail.video.status==='done' ? downloadMd(detail.video.video_id) : processOne(detail.video.video_id)" x-text="detail.video.status==='done' ? 'Download .md' : 'Process to .md'"></button>
|
||||||
|
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="copyMd(detail.video.video_id)" x-show="detail.video && detail.video.status==='done'" title="Copiar todo el contenido del .md al portapapeles">
|
||||||
|
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><rect x="9" y="9" width="13" height="13" rx="2" ry="2"/><path d="M5 15H4a2 2 0 0 1-2-2V4a2 2 0 0 1 2-2h9a2 2 0 0 1 2 2v1"/></svg>
|
||||||
|
<span>Copiar .md</span>
|
||||||
|
</button>
|
||||||
|
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="openMd(detail.video.video_id)" x-show="detail.video && detail.video.status==='done'" title="Abrir archivo .md en tu aplicación predeterminada (Obsidian, VS Code, etc.)">
|
||||||
|
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"/><polyline points="15 3 21 3 21 9"/><line x1="10" y1="14" x2="21" y2="3"/></svg>
|
||||||
|
<span>Abrir .md</span>
|
||||||
|
</button>
|
||||||
<button x-show="isBlocked(detail.video) && detail.video.status!=='done'" class="btn btn-ghost !py-1.5 !px-3 text-xs opacity-60" @click="processOne(detail.video.video_id)" :title="blockTitle(detail.video) + ' Click anyway if you have since bought the membership and refreshed your cookies.'">Try anyway</button>
|
<button x-show="isBlocked(detail.video) && detail.video.status!=='done'" class="btn btn-ghost !py-1.5 !px-3 text-xs opacity-60" @click="processOne(detail.video.video_id)" :title="blockTitle(detail.video) + ' Click anyway if you have since bought the membership and refreshed your cookies.'">Try anyway</button>
|
||||||
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="downloadAudioOne(detail.video.video_id)" x-show="detail.video.status==='done'" x-text="detail.hasAudio ? 'Re-download audio' : 'Download audio'">
|
<button class="btn btn-ghost !py-1.5 !px-3 text-xs" @click="downloadAudioOne(detail.video.video_id)" x-show="detail.video.status==='done'" x-text="detail.hasAudio ? 'Re-download audio' : 'Download audio'">
|
||||||
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><path stroke-linecap="round" stroke-linejoin="round" d="M9 18V6l10-2v12M9 18a3 3 0 11-6 0 3 3 0 016 0zm10-2a3 3 0 11-6 0 3 3 0 016 0z"/></svg>
|
<svg class="w-3.5 h-3.5" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2"><path stroke-linecap="round" stroke-linejoin="round" d="M9 18V6l10-2v12M9 18a3 3 0 11-6 0 3 3 0 016 0zm10-2a3 3 0 11-6 0 3 3 0 016 0z"/></svg>
|
||||||
@@ -527,7 +610,7 @@
|
|||||||
<section class="space-y-5" x-transition.opacity>
|
<section class="space-y-5" x-transition.opacity>
|
||||||
<header><h1 class="text-2xl font-bold text-white">Search transcripts</h1><p class="text-sm text-zinc-500">Find moments across all videos</p></header>
|
<header><h1 class="text-2xl font-bold text-white">Search transcripts</h1><p class="text-sm text-zinc-500">Find moments across all videos</p></header>
|
||||||
<form class="glass p-4 flex flex-col md:flex-row gap-2" @submit.prevent="loadSearch()">
|
<form class="glass p-4 flex flex-col md:flex-row gap-2" @submit.prevent="loadSearch()">
|
||||||
<input class="field flex-1" type="text" placeholder="search query… (Press Enter)" x-model="search.q" aria-label="Search query" />
|
<input x-ref="searchInput" class="field flex-1" type="text" placeholder="search query… (Press Enter or /)" x-model="search.q" aria-label="Search query" />
|
||||||
<select class="field md:w-56" x-model="search.channel" @change="if(search.ran) loadSearch()" aria-label="Filter by channel"><option value="">All channels</option><template x-for="c in channels.items" :key="c.channel_id"><option :value="c.channel_id" x-text="c.name"></option></template></select>
|
<select class="field md:w-56" x-model="search.channel" @change="if(search.ran) loadSearch()" aria-label="Filter by channel"><option value="">All channels</option><template x-for="c in channels.items" :key="c.channel_id"><option :value="c.channel_id" x-text="c.name"></option></template></select>
|
||||||
<button class="btn accent-grad btn-primary" :disabled="loading.search || !search.q.trim()" @click="loadSearch()">Search</button>
|
<button class="btn accent-grad btn-primary" :disabled="loading.search || !search.q.trim()" @click="loadSearch()">Search</button>
|
||||||
<button class="btn btn-ghost" type="button" @click="resetSearch()" x-show="search.ran || search.q || search.channel" title="Clear search">Clear</button>
|
<button class="btn btn-ghost" type="button" @click="resetSearch()" x-show="search.ran || search.q || search.channel" title="Clear search">Clear</button>
|
||||||
@@ -540,7 +623,7 @@
|
|||||||
</div>
|
</div>
|
||||||
</template>
|
</template>
|
||||||
<template x-if="!loading.search && !search.ran">
|
<template x-if="!loading.search && !search.ran">
|
||||||
<div class="glass p-10 text-center text-zinc-500">Type a query and press <kbd class="kbd">Enter</kbd> to search transcripts across all channels.<br><span class="text-xs text-zinc-500">Tip: try a phrase in its original language for best matches.</span></div>
|
<div class="glass p-10 text-center text-zinc-500">Type a query and press <kbd class="kbd">Enter</kbd> to search transcripts across all channels. Press / anywhere to jump to this search.<br><span class="text-xs text-zinc-500">Tip: try a phrase in its original language for best matches.</span></div>
|
||||||
</template>
|
</template>
|
||||||
<div class="space-y-3">
|
<div class="space-y-3">
|
||||||
<template x-for="r in search.items" :key="r.video_id+r.start_sec">
|
<template x-for="r in search.items" :key="r.video_id+r.start_sec">
|
||||||
@@ -675,7 +758,23 @@
|
|||||||
<table class="tbl">
|
<table class="tbl">
|
||||||
<thead><tr><th>Channel</th><th>Status</th><th>Progress</th><th>Started</th><th>Finished</th><th></th></tr></thead>
|
<thead><tr><th>Channel</th><th>Status</th><th>Progress</th><th>Started</th><th>Finished</th><th></th></tr></thead>
|
||||||
<tbody>
|
<tbody>
|
||||||
<template x-if="loading.jobs"><tr><td colspan="6" class="text-center text-zinc-500 py-6">loading…</td></tr></template>
|
<template x-if="loading.jobs">
|
||||||
|
<template x-for="i in 4" :key="i">
|
||||||
|
<tr>
|
||||||
|
<td>
|
||||||
|
<template x-if="i===1"><span class="sr-only" role="status">Loading jobs…</span></template>
|
||||||
|
<div class="skeleton h-3.5 w-28" aria-hidden="true"></div>
|
||||||
|
</td>
|
||||||
|
<td><div class="skeleton h-5 w-16 rounded-full" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-12" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-24" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-24" aria-hidden="true"></div></td>
|
||||||
|
<td>
|
||||||
|
<div class="flex justify-end" aria-hidden="true"><div class="skeleton h-5 w-12"></div></div>
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
</template>
|
||||||
|
</template>
|
||||||
<template x-if="!loading.jobs && scrape.jobs.length===0"><tr><td colspan="6" class="text-center text-zinc-500 py-6">No jobs yet.</td></tr></template>
|
<template x-if="!loading.jobs && scrape.jobs.length===0"><tr><td colspan="6" class="text-center text-zinc-500 py-6">No jobs yet.</td></tr></template>
|
||||||
<template x-for="j in scrape.jobs" :key="j.id">
|
<template x-for="j in scrape.jobs" :key="j.id">
|
||||||
<tr>
|
<tr>
|
||||||
@@ -698,8 +797,19 @@
|
|||||||
<section class="space-y-5" x-transition.opacity>
|
<section class="space-y-5" x-transition.opacity>
|
||||||
<header><h1 class="text-2xl font-bold text-white">Cookie vault</h1><p class="text-sm text-zinc-500">Upload YouTube cookie files to authenticate</p></header>
|
<header><h1 class="text-2xl font-bold text-white">Cookie vault</h1><p class="text-sm text-zinc-500">Upload YouTube cookie files to authenticate</p></header>
|
||||||
|
|
||||||
|
<div class="glass px-4 py-3 flex items-center justify-between gap-3">
|
||||||
|
<div>
|
||||||
|
<div class="text-sm text-zinc-200 font-medium">Import from Brave</div>
|
||||||
|
<div class="text-xs text-zinc-500">Pulls your logged-in YouTube session straight from the browser's cookie store — no extensions or manual exports. <span class="text-amber-400/90">Close Brave first</span>: it keeps its cookie database locked while running.</div>
|
||||||
|
</div>
|
||||||
|
<button class="btn !py-1.5 !px-3 text-xs shrink-0" :disabled="loading.cookies" @click="importBrowserCookies()">
|
||||||
|
<span x-show="!loading.cookies">Import from Brave</span>
|
||||||
|
<span x-show="loading.cookies">Importing…</span>
|
||||||
|
</button>
|
||||||
|
</div>
|
||||||
|
|
||||||
<div class="dropzone p-8 text-center transition-all" :class="cookies.drag ? 'drag' : ''"
|
<div class="dropzone p-8 text-center transition-all" :class="cookies.drag ? 'drag' : ''"
|
||||||
@click="$refs.cookieFile.click()"
|
@click="ref('cookieFile') && ref('cookieFile').click()"
|
||||||
@dragover.prevent="cookies.drag=true"
|
@dragover.prevent="cookies.drag=true"
|
||||||
@dragleave.prevent="cookies.drag=false"
|
@dragleave.prevent="cookies.drag=false"
|
||||||
@drop.prevent="handleDrop($event)">
|
@drop.prevent="handleDrop($event)">
|
||||||
@@ -718,7 +828,29 @@
|
|||||||
<table class="tbl">
|
<table class="tbl">
|
||||||
<thead><tr><th>Label</th><th>File</th><th>Expires</th><th>Session</th><th>Active</th><th>Test</th><th></th></tr></thead>
|
<thead><tr><th>Label</th><th>File</th><th>Expires</th><th>Session</th><th>Active</th><th>Test</th><th></th></tr></thead>
|
||||||
<tbody>
|
<tbody>
|
||||||
<template x-if="loading.cookies"><tr><td colspan="7" class="text-center text-zinc-500 py-6">loading…</td></tr></template>
|
<template x-if="loading.cookies">
|
||||||
|
<template x-for="i in 4" :key="i">
|
||||||
|
<tr>
|
||||||
|
<td>
|
||||||
|
<template x-if="i===1"><span class="sr-only" role="status">Loading cookies…</span></template>
|
||||||
|
<div class="skeleton h-4 w-24" aria-hidden="true"></div>
|
||||||
|
</td>
|
||||||
|
<td><div class="skeleton h-3 w-40" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-3 w-28" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton h-5 w-16 rounded-full" aria-hidden="true"></div></td>
|
||||||
|
<td><div class="skeleton w-4 h-4 rounded-full" aria-hidden="true"></div></td>
|
||||||
|
<td>
|
||||||
|
<div class="flex items-center gap-2" aria-hidden="true">
|
||||||
|
<div class="skeleton h-5 w-10"></div>
|
||||||
|
<div class="skeleton h-3 w-16"></div>
|
||||||
|
</div>
|
||||||
|
</td>
|
||||||
|
<td>
|
||||||
|
<div class="flex justify-end" aria-hidden="true"><div class="skeleton h-5 w-14"></div></div>
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
</template>
|
||||||
|
</template>
|
||||||
<template x-if="!loading.cookies && cookies.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-6">Vault is empty.</td></tr></template>
|
<template x-if="!loading.cookies && cookies.items.length===0"><tr><td colspan="7" class="text-center text-zinc-500 py-6">Vault is empty.</td></tr></template>
|
||||||
<template x-for="c in cookies.items" :key="c.id">
|
<template x-for="c in cookies.items" :key="c.id">
|
||||||
<tr>
|
<tr>
|
||||||
|
|||||||
@@ -107,6 +107,22 @@ select.field { appearance: none; background-image: linear-gradient(45deg, transp
|
|||||||
border: 1px solid transparent;
|
border: 1px solid transparent;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* ---- active-filter chips (videos) ---- */
|
||||||
|
/* Removable summary of the filters constraining the list: rose tint marks
|
||||||
|
them as the accent-colored "current context", the ✕ is the affordance. */
|
||||||
|
.filter-chip {
|
||||||
|
display: inline-flex; align-items: center; gap: 0.4rem;
|
||||||
|
padding: 0.25rem 0.65rem; border-radius: 999px;
|
||||||
|
background: rgba(244, 63, 94, 0.10);
|
||||||
|
border: 1px solid rgba(244, 63, 94, 0.35);
|
||||||
|
color: #fecdd3; font-size: 0.75rem; font-weight: 600;
|
||||||
|
cursor: pointer; transition: all 0.15s ease;
|
||||||
|
}
|
||||||
|
.filter-chip:hover { background: rgba(244, 63, 94, 0.20); color: #fff; border-color: rgba(244, 63, 94, 0.55); }
|
||||||
|
.filter-chip:active { transform: translateY(1px); }
|
||||||
|
.filter-chip svg { width: 0.75rem; height: 0.75rem; opacity: 0.7; flex-shrink: 0; }
|
||||||
|
.filter-chip:hover svg { opacity: 1; }
|
||||||
|
|
||||||
/* ---- table ---- */
|
/* ---- table ---- */
|
||||||
.tbl { width: 100%; border-collapse: separate; border-spacing: 0; }
|
.tbl { width: 100%; border-collapse: separate; border-spacing: 0; }
|
||||||
.tbl th { text-align: left; font-size: 0.7rem; text-transform: uppercase; letter-spacing: 0.05em; color: #71717a; font-weight: 600; padding: 0.6rem 0.75rem; border-bottom: 1px solid #27272a; }
|
.tbl th { text-align: left; font-size: 0.7rem; text-transform: uppercase; letter-spacing: 0.05em; color: #71717a; font-weight: 600; padding: 0.6rem 0.75rem; border-bottom: 1px solid #27272a; }
|
||||||
@@ -143,6 +159,20 @@ select.field { appearance: none; background-image: linear-gradient(45deg, transp
|
|||||||
.spin { animation: spin 0.9s linear infinite; }
|
.spin { animation: spin 0.9s linear infinite; }
|
||||||
@keyframes spin { to { transform: rotate(360deg); } }
|
@keyframes spin { to { transform: rotate(360deg); } }
|
||||||
|
|
||||||
|
/* ---- skeleton loaders ---- */
|
||||||
|
/* One class only: size/shape is varied per element with Tailwind utilities. */
|
||||||
|
.skeleton {
|
||||||
|
background: linear-gradient(90deg, #18181b 25%, #27272a 50%, #18181b 75%);
|
||||||
|
background-size: 200% 100%;
|
||||||
|
animation: skeleton-shimmer 1.4s ease infinite;
|
||||||
|
border-radius: 0.375rem;
|
||||||
|
}
|
||||||
|
@keyframes skeleton-shimmer {
|
||||||
|
0% { background-position: 200% 0; }
|
||||||
|
100% { background-position: -200% 0; }
|
||||||
|
}
|
||||||
|
@media (prefers-reduced-motion: reduce) { .skeleton { animation: none; } }
|
||||||
|
|
||||||
/* ---- wordcloud ---- */
|
/* ---- wordcloud ---- */
|
||||||
.cloud-word { display: inline-block; margin: 0.35rem 0.4rem; line-height: 1; cursor: default; transition: opacity 0.15s ease; }
|
.cloud-word { display: inline-block; margin: 0.35rem 0.4rem; line-height: 1; cursor: default; transition: opacity 0.15s ease; }
|
||||||
.cloud-word:hover { opacity: 0.75; }
|
.cloud-word:hover { opacity: 0.75; }
|
||||||
|
|||||||
@@ -0,0 +1,152 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from yt_scraper.cookies import (
|
||||||
|
BrowserCookieLockedError,
|
||||||
|
auto_import_dir,
|
||||||
|
delete,
|
||||||
|
import_from_browser,
|
||||||
|
is_expired,
|
||||||
|
parse_netscape,
|
||||||
|
resolve_active_path,
|
||||||
|
)
|
||||||
|
from yt_scraper.store import CookieRow, Store
|
||||||
|
|
||||||
|
|
||||||
|
SAMPLE_NETSCAPE = """# Netscape HTTP Cookie File
|
||||||
|
.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSID\tsample_sid_token
|
||||||
|
.youtube.com\tTRUE\t/\tTRUE\t2000000000\tLOGIN_INFO\tsample_login_info
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_netscape():
|
||||||
|
ok, info = parse_netscape(SAMPLE_NETSCAPE)
|
||||||
|
assert ok is True
|
||||||
|
assert info["count"] == 2
|
||||||
|
assert info["has_session"] is True
|
||||||
|
assert info["expires_at"] is not None
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_netscape_httponly_lines_are_data():
|
||||||
|
# SID/HSID are HttpOnly; Netscape exports prefix those lines and a
|
||||||
|
# comment-skipping parser would silently strip the login out.
|
||||||
|
text = (
|
||||||
|
"# Netscape HTTP Cookie File\n"
|
||||||
|
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSID\ttok\n"
|
||||||
|
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tHSID\ttok\n"
|
||||||
|
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSSID\ttok\n"
|
||||||
|
"# a real comment\n"
|
||||||
|
)
|
||||||
|
ok, info = parse_netscape(text)
|
||||||
|
assert ok is True
|
||||||
|
assert info["count"] == 3
|
||||||
|
assert info["has_session"] is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_partial_session_export_is_not_a_session():
|
||||||
|
# A lone __Secure-3PSID (partial extension export) is anonymous to
|
||||||
|
# YouTube; it must not be reported as a usable session.
|
||||||
|
text = (
|
||||||
|
"# Netscape HTTP Cookie File\n"
|
||||||
|
".youtube.com\tTRUE\t/\tTRUE\t2000000000\t__Secure-3PSID\ttok\n"
|
||||||
|
".youtube.com\tTRUE\t/\tTRUE\t2000000000\t__Secure-3PAPISID\ttok\n"
|
||||||
|
)
|
||||||
|
ok, info = parse_netscape(text)
|
||||||
|
assert ok is True
|
||||||
|
assert info["has_session"] is False
|
||||||
|
|
||||||
|
|
||||||
|
def _browser_cookie(name, value, *, domain=".youtube.com", httponly=False):
|
||||||
|
import http.cookiejar
|
||||||
|
return http.cookiejar.Cookie(
|
||||||
|
version=0, name=name, value=value, port=None, port_specified=False,
|
||||||
|
domain=domain, domain_specified=True, domain_initial_dot=domain.startswith("."),
|
||||||
|
path="/", path_specified=True, secure=True, expires=2000000000,
|
||||||
|
discard=False, comment=None, comment_url=None,
|
||||||
|
rest={"HttpOnly": None} if httponly else {},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_import_from_browser_writes_vault_file(tmp_path: Path, monkeypatch):
|
||||||
|
store = Store(tmp_path / "state.db")
|
||||||
|
cookie_dir = tmp_path / "cookies"
|
||||||
|
|
||||||
|
def fake_extract(browser, profile=None, logger=None, **kw):
|
||||||
|
return iter([
|
||||||
|
_browser_cookie("SID", "sid_tok", httponly=True),
|
||||||
|
_browser_cookie("HSID", "hsid_tok", httponly=True),
|
||||||
|
_browser_cookie("SSID", "ssid_tok", httponly=True),
|
||||||
|
_browser_cookie("__Secure-3PSID", "psid_tok"),
|
||||||
|
_browser_cookie("PREF", "pref_tok", domain=".google.com"), # not youtube -> dropped
|
||||||
|
])
|
||||||
|
|
||||||
|
import yt_dlp.cookies as ydl_cookies
|
||||||
|
monkeypatch.setattr(ydl_cookies, "extract_cookies_from_browser", fake_extract)
|
||||||
|
|
||||||
|
cid = import_from_browser(store, browser="brave", cookie_dir=cookie_dir)
|
||||||
|
row = store.get_cookie(cid)
|
||||||
|
assert row is not None
|
||||||
|
assert row.cookie_count == 4
|
||||||
|
assert row.has_session is True
|
||||||
|
|
||||||
|
# The written file must round-trip as a valid Netscape file whose
|
||||||
|
# HttpOnly session cookies survive the vault's own parser.
|
||||||
|
from yt_scraper.cookies import parse_netscape_file
|
||||||
|
path = cookie_dir / row.filename
|
||||||
|
ok, info = parse_netscape_file(path)
|
||||||
|
assert ok is True
|
||||||
|
assert info["count"] == 4
|
||||||
|
assert info["has_session"] is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_import_from_browser_maps_locked_db(tmp_path: Path, monkeypatch):
|
||||||
|
store = Store(tmp_path / "state.db")
|
||||||
|
|
||||||
|
def fake_extract(browser, profile=None, logger=None, **kw):
|
||||||
|
raise RuntimeError("Could not copy Chrome cookie database. See https://github.com/yt-dlp/yt-dlp/issues/7271 for more info")
|
||||||
|
|
||||||
|
import yt_dlp.cookies as ydl_cookies
|
||||||
|
monkeypatch.setattr(ydl_cookies, "extract_cookies_from_browser", fake_extract)
|
||||||
|
|
||||||
|
with pytest.raises(BrowserCookieLockedError) as exc:
|
||||||
|
import_from_browser(store, browser="brave", cookie_dir=tmp_path / "cookies")
|
||||||
|
assert "close brave" in str(exc.value).lower()
|
||||||
|
|
||||||
|
|
||||||
|
def test_auto_import_and_prune_dead_cookies(tmp_path: Path):
|
||||||
|
db_path = tmp_path / "state.db"
|
||||||
|
store = Store(db_path)
|
||||||
|
cookie_dir = tmp_path / "cookies"
|
||||||
|
cookie_dir.mkdir()
|
||||||
|
|
||||||
|
# Place two cookie files
|
||||||
|
f1 = cookie_dir / "c1.txt"
|
||||||
|
f1.write_text(SAMPLE_NETSCAPE, encoding="utf-8")
|
||||||
|
f2 = cookie_dir / "c2.txt"
|
||||||
|
f2.write_text(SAMPLE_NETSCAPE, encoding="utf-8")
|
||||||
|
|
||||||
|
# auto_import imports both and activates the first
|
||||||
|
n = auto_import_dir(store, dir_path=cookie_dir)
|
||||||
|
assert n == 2
|
||||||
|
assert len(store.list_cookies()) == 2
|
||||||
|
active = store.get_active_cookie()
|
||||||
|
assert active is not None
|
||||||
|
assert active.filename in ("c1.txt", "c2.txt")
|
||||||
|
|
||||||
|
active_path = resolve_active_path(store, cookie_dir=cookie_dir)
|
||||||
|
assert active_path is not None
|
||||||
|
assert Path(active_path).exists()
|
||||||
|
|
||||||
|
# Now simulate user deleting the active cookie file from disk
|
||||||
|
Path(active_path).unlink()
|
||||||
|
|
||||||
|
# resolve_active_path should fall back to the surviving cookie file
|
||||||
|
fallback_path = resolve_active_path(store, cookie_dir=cookie_dir)
|
||||||
|
assert fallback_path is not None
|
||||||
|
assert Path(fallback_path).exists()
|
||||||
|
|
||||||
|
# auto_import_dir should clean up the deleted cookie row from DB
|
||||||
|
auto_import_dir(store, dir_path=cookie_dir)
|
||||||
|
assert len(store.list_cookies()) == 1
|
||||||
@@ -0,0 +1,68 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
from fastapi import FastAPI
|
||||||
|
from fastapi.testclient import TestClient
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from yt_scraper.config import Config
|
||||||
|
from yt_scraper.store import Store, VideoRef
|
||||||
|
from yt_scraper.webapp.api import build_router
|
||||||
|
from yt_scraper.webapp.jobs import JobManager
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def api_client(tmp_path: Path, monkeypatch):
|
||||||
|
db_path = tmp_path / "state.db"
|
||||||
|
md_root = tmp_path / "markdown"
|
||||||
|
md_root.mkdir()
|
||||||
|
|
||||||
|
store = Store(db_path)
|
||||||
|
cfg = Config(database_path=str(db_path), output_dir=str(md_root))
|
||||||
|
jobs = JobManager(store, cfg)
|
||||||
|
router = build_router(store, cfg, jobs)
|
||||||
|
|
||||||
|
app = FastAPI()
|
||||||
|
app.include_router(router)
|
||||||
|
client = TestClient(app)
|
||||||
|
return client, store, cfg, tmp_path
|
||||||
|
|
||||||
|
|
||||||
|
def test_get_and_open_markdown(api_client, monkeypatch):
|
||||||
|
client, store, cfg, tmp_path = api_client
|
||||||
|
store.upsert_channel("UC1", "@test", "Test Channel", 1)
|
||||||
|
store.upsert_videos([
|
||||||
|
VideoRef("vid1", "UC1", "Test Video", "https://www.youtube.com/watch?v=vid1")
|
||||||
|
])
|
||||||
|
|
||||||
|
# 1. Not generated yet -> 404
|
||||||
|
r_get = client.get("/api/videos/vid1/markdown")
|
||||||
|
assert r_get.status_code == 404
|
||||||
|
|
||||||
|
r_open = client.post("/api/videos/vid1/open-markdown")
|
||||||
|
assert r_open.status_code == 404
|
||||||
|
|
||||||
|
# 2. Create markdown file and mark done
|
||||||
|
md_dir = tmp_path / "markdown" / "Test Channel"
|
||||||
|
md_dir.mkdir(parents=True)
|
||||||
|
md_file = md_dir / "2026-08-23_test-video.md"
|
||||||
|
md_content = "# Test Video\n\nContent here"
|
||||||
|
md_file.write_text(md_content, encoding="utf-8", newline="\n")
|
||||||
|
|
||||||
|
rel_path = "markdown/Test Channel/2026-08-23_test-video.md"
|
||||||
|
store.mark_done("vid1", rel_path, "es", "auto", False)
|
||||||
|
|
||||||
|
# 3. GET markdown returns text
|
||||||
|
r_get = client.get("/api/videos/vid1/markdown")
|
||||||
|
assert r_get.status_code == 200
|
||||||
|
assert r_get.text.replace("\r\n", "\n") == md_content
|
||||||
|
|
||||||
|
# 4. POST open-markdown calls _open_in_os
|
||||||
|
opened_paths = []
|
||||||
|
from yt_scraper.webapp import api as api_mod
|
||||||
|
monkeypatch.setattr(api_mod, "_open_in_os", lambda p: opened_paths.append(str(p)))
|
||||||
|
|
||||||
|
r_open = client.post("/api/videos/vid1/open-markdown")
|
||||||
|
assert r_open.status_code == 200
|
||||||
|
assert len(opened_paths) == 1
|
||||||
|
assert str(md_file.resolve()) in [str(Path(p).resolve()) for p in opened_paths]
|
||||||
@@ -136,14 +136,16 @@ def test_cookie_vault(store):
|
|||||||
sample = (
|
sample = (
|
||||||
"# Netscape HTTP Cookie File\n"
|
"# Netscape HTTP Cookie File\n"
|
||||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSID\tabc\n"
|
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSID\tabc\n"
|
||||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSAPISID\tdef\n"
|
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tHSID\tdef\n"
|
||||||
|
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSSID\tghi\n"
|
||||||
|
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSAPISID\tjkl\n"
|
||||||
".youtube.com\tTRUE\t/\tFALSE\t1800287621\tPREF\tf1=2\n"
|
".youtube.com\tTRUE\t/\tFALSE\t1800287621\tPREF\tf1=2\n"
|
||||||
)
|
)
|
||||||
cid = cookies.import_text(store, sample, label="test", cookie_dir=str(Path(store.db_path).parent / "ck"))
|
cid = cookies.import_text(store, sample, label="test", cookie_dir=str(Path(store.db_path).parent / "ck"))
|
||||||
vault = cookies.list_vault(store)
|
vault = cookies.list_vault(store)
|
||||||
assert len(vault) == 1
|
assert len(vault) == 1
|
||||||
assert vault[0].has_session is True
|
assert vault[0].has_session is True
|
||||||
assert vault[0].cookie_count == 3
|
assert vault[0].cookie_count == 5
|
||||||
# set active + resolve
|
# set active + resolve
|
||||||
cookies.set_active(store, cid)
|
cookies.set_active(store, cid)
|
||||||
path = cookies.resolve_active_path(store, cookie_dir=str(Path(store.db_path).parent / "ck"))
|
path = cookies.resolve_active_path(store, cookie_dir=str(Path(store.db_path).parent / "ck"))
|
||||||
|
|||||||
@@ -0,0 +1,88 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import io
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from yt_scraper.extract import extract_via_watch_page
|
||||||
|
|
||||||
|
|
||||||
|
WATCH_PAGE = (
|
||||||
|
"var ytInitialPlayerResponse = "
|
||||||
|
+ json.dumps({
|
||||||
|
"playabilityStatus": {"status": "OK"},
|
||||||
|
"videoDetails": {
|
||||||
|
"title": "Members only test",
|
||||||
|
"author": "Canal",
|
||||||
|
"lengthSeconds": "42",
|
||||||
|
"viewCount": "7",
|
||||||
|
"shortDescription": "desc",
|
||||||
|
"keywords": ["a", "b"],
|
||||||
|
},
|
||||||
|
"microformat": {"playerMicroformatRenderer": {"publishDate": "2026-09-01"}},
|
||||||
|
"captions": {"playerCaptionsTracklistRenderer": {"captionTracks": [
|
||||||
|
{"languageCode": "es", "kind": "asr", "baseUrl": "https://captions.test/es"},
|
||||||
|
]}},
|
||||||
|
})
|
||||||
|
+ ";"
|
||||||
|
)
|
||||||
|
|
||||||
|
JSON3 = json.dumps({
|
||||||
|
"events": [
|
||||||
|
{"tStartMs": 0, "dDurationMs": 1000, "segs": [{"utf8": "hola "}, {"utf8": "mundo"}]},
|
||||||
|
{"tStartMs": 1000, "dDurationMs": 500, "segs": [{"utf8": "segundo"}]},
|
||||||
|
]
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
class _FakeResponse(io.BytesIO):
|
||||||
|
def __enter__(self):
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, *a):
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def test_watch_page_fallback_builds_segments(tmp_path: Path, monkeypatch):
|
||||||
|
from yt_scraper import extract as ex
|
||||||
|
|
||||||
|
def fake_urlopen(cookies_file, url, **kw):
|
||||||
|
if url.startswith("https://www.youtube.com/"):
|
||||||
|
return _FakeResponse(WATCH_PAGE.encode("utf-8"))
|
||||||
|
assert "fmt=json3" in url
|
||||||
|
return _FakeResponse(JSON3.encode("utf-8"))
|
||||||
|
|
||||||
|
monkeypatch.setattr(ex, "_session_urlopen", fake_urlopen)
|
||||||
|
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
|
||||||
|
|
||||||
|
data = extract_via_watch_page(
|
||||||
|
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
|
||||||
|
)
|
||||||
|
|
||||||
|
assert data is not None
|
||||||
|
assert data.subtitle is not None and data.subtitle.lang == "es"
|
||||||
|
# parse_auto_dump merges caption events with no gap between them.
|
||||||
|
assert len(data.segments) == 1
|
||||||
|
assert data.segments[0].text == "hola mundo segundo"
|
||||||
|
assert data.info["title"] == "Members only test"
|
||||||
|
assert data.info["upload_date"] == "20260901"
|
||||||
|
assert data.info["duration"] == 42
|
||||||
|
assert data.skip_reason is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_watch_page_fallback_without_tracks_returns_none(tmp_path: Path, monkeypatch):
|
||||||
|
from yt_scraper import extract as ex
|
||||||
|
|
||||||
|
page = WATCH_PAGE.replace('"captionTracks": [{', '"captionTracks": [{')
|
||||||
|
page = "var ytInitialPlayerResponse = " + json.dumps({
|
||||||
|
"playabilityStatus": {"status": "OK"},
|
||||||
|
"videoDetails": {"title": "x"},
|
||||||
|
}) + ";"
|
||||||
|
|
||||||
|
monkeypatch.setattr(ex, "_session_urlopen",
|
||||||
|
lambda *a, **kw: _FakeResponse(page.encode("utf-8")))
|
||||||
|
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
|
||||||
|
|
||||||
|
assert extract_via_watch_page(
|
||||||
|
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
|
||||||
|
) is None
|
||||||
Reference in New Issue
Block a user