Extraccion autenticada: - import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20 - extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla con sesion logueada (members-only); regex y opener cacheados - js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts - rutas de Brave multiplataforma (Windows/macOS/Linux) Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL, memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics Rendimiento: - entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota) - _rank_unranked con guarda (antes full-scan en cada arranque/import) - upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL) - thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool) - reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md), healthz expone reconcile_done - Store.transaction(): escrituras por video agrupadas (~6 commits -> 3) Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render, order_pending->store, keep_ref->config, seconds_to_ts solo en segments); re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done); fuera wrappers muertos de segments.py
290 lines
10 KiB
Python
290 lines
10 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Callable
|
|
|
|
from .parse import Segment
|
|
from .store import Store
|
|
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
_TS = re.compile(r"^(\d{1,2}):(\d{2})(?::(\d{2}))?$")
|
|
_SEG_LINE = re.compile(r"^\*\*(?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\*\*\s*[·\-]\s*(?P<text>.+?)\s*$")
|
|
_CHAPTER = re.compile(r"^###\s+(?P<title>.+?)\s*\((?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\)\s*$")
|
|
_FRONTMATTER_KEY = re.compile(r"^([A-Za-z0-9_]+)\s*:\s*(.*)$")
|
|
|
|
|
|
def ts_to_seconds(ts: str) -> float:
|
|
m = _TS.match(ts.strip())
|
|
if not m:
|
|
return 0.0
|
|
if m.group(3): # h:m:s
|
|
return int(m.group(1)) * 3600 + int(m.group(2)) * 60 + int(m.group(3))
|
|
return int(m.group(1)) * 60 + int(m.group(2))
|
|
|
|
|
|
def seconds_to_ts(sec: float) -> str:
|
|
total = int(sec)
|
|
h, rem = divmod(total, 3600)
|
|
m, s = divmod(rem, 60)
|
|
return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}"
|
|
|
|
|
|
@dataclass
|
|
class ParsedMarkdown:
|
|
metadata: dict
|
|
segments: list[Segment]
|
|
chapters: list[dict] # {"title": str, "start": float}
|
|
|
|
|
|
def parse_markdown(text: str) -> ParsedMarkdown:
|
|
lines = text.splitlines()
|
|
metadata: dict[str, str] = {}
|
|
segments: list[Segment] = []
|
|
chapters: list[dict] = []
|
|
in_front = False
|
|
front_done = False
|
|
i = 0
|
|
# frontmatter
|
|
if lines and lines[0].strip() == "---":
|
|
i = 1
|
|
in_front = True
|
|
while i < len(lines):
|
|
if lines[i].strip() == "---":
|
|
front_done = True
|
|
i += 1
|
|
break
|
|
m = _FRONTMATTER_KEY.match(lines[i])
|
|
if m:
|
|
metadata[m.group(1).lower()] = _strip_quotes(m.group(2).strip())
|
|
i += 1
|
|
if not front_done:
|
|
i = 0
|
|
# body
|
|
current_start = 0.0
|
|
while i < len(lines):
|
|
line = lines[i]
|
|
s = line.strip()
|
|
cm = _CHAPTER.match(s)
|
|
if cm:
|
|
current_start = ts_to_seconds(cm.group("ts"))
|
|
chapters.append({"title": cm.group("title").strip(), "start": current_start})
|
|
i += 1
|
|
continue
|
|
sm = _SEG_LINE.match(s)
|
|
if sm:
|
|
start = ts_to_seconds(sm.group("ts"))
|
|
segments.append(Segment(start=start, end=start, text=sm.group("text").strip()))
|
|
i += 1
|
|
return ParsedMarkdown(metadata=metadata, segments=segments, chapters=chapters)
|
|
|
|
|
|
def _strip_quotes(value: str) -> str:
|
|
v = value.strip()
|
|
if len(v) >= 2 and v[0] == v[-1] and v[0] in ("'", '"'):
|
|
return v[1:-1]
|
|
return v
|
|
|
|
|
|
def backfill_from_markdown(
|
|
store: Store,
|
|
md_root: Path,
|
|
log: Callable[[str], None] | None = None,
|
|
) -> int:
|
|
"""Parse all done .md files under md_root, populate segments/FTS/metadata. Idempotent."""
|
|
def _log(msg: str) -> None:
|
|
if log:
|
|
log(msg)
|
|
else:
|
|
logmod = logging.getLogger(__name__)
|
|
logmod.info(msg)
|
|
|
|
md_root = Path(md_root)
|
|
if not md_root.exists():
|
|
_log(f"backfill: markdown root not found: {md_root}")
|
|
return 0
|
|
md_files = sorted(md_root.rglob("*.md"))
|
|
_log(f"backfill: scanning {len(md_files)} markdown files")
|
|
n = 0
|
|
for md_path in md_files:
|
|
try:
|
|
text = md_path.read_text(encoding="utf-8")
|
|
except (OSError, UnicodeDecodeError) as exc:
|
|
# UnicodeDecodeError is a ValueError, not an OSError — letting it
|
|
# escape aborted the loop and silently skipped every later file.
|
|
_log(f"backfill: skip unreadable {md_path}: {exc}")
|
|
continue
|
|
parsed = parse_markdown(text)
|
|
video_id = parsed.metadata.get("video_id")
|
|
if not video_id:
|
|
continue
|
|
existing = store.get_video(video_id)
|
|
if not existing:
|
|
_log(f"backfill: video {video_id} not in DB (orphan md), skipping")
|
|
continue
|
|
# always (re)populate upload_date if missing — cheap, idempotent, gated by NULL
|
|
ud = parsed.metadata.get("upload_date")
|
|
if ud:
|
|
compact = ud.replace("-", "")
|
|
if compact.isdigit():
|
|
store.set_upload_date(video_id, compact)
|
|
if store.has_segments(video_id) and existing.segments_json:
|
|
continue # segments + rich metadata already populated
|
|
# metadata from frontmatter
|
|
tags_raw = parsed.metadata.get("tags", "")
|
|
tags: list[str] = []
|
|
if tags_raw:
|
|
try:
|
|
tags = json.loads(tags_raw) if tags_raw.startswith("[") else [t.strip() for t in tags_raw.split(",")]
|
|
except json.JSONDecodeError:
|
|
tags = []
|
|
seg_json = json.dumps(
|
|
[{"start": s.start, "end": s.end, "text": s.text} for s in parsed.segments],
|
|
ensure_ascii=False,
|
|
)
|
|
ch_json = json.dumps(parsed.chapters, ensure_ascii=False)
|
|
store.update_video_metadata(
|
|
video_id,
|
|
view_count=_to_int(parsed.metadata.get("views")),
|
|
like_count=_to_int(parsed.metadata.get("likes")),
|
|
tags=tags or None,
|
|
thumbnail=parsed.metadata.get("thumbnail") or None,
|
|
description=None,
|
|
chapters_json=ch_json,
|
|
segments_json=seg_json,
|
|
)
|
|
store.store_segments(video_id, parsed.segments)
|
|
n += 1
|
|
_log(f"backfill: populated {n} videos")
|
|
return n
|
|
|
|
|
|
def reconcile_markdown(
|
|
store: Store,
|
|
md_root: Path,
|
|
log: Callable[[str], None] | None = None,
|
|
*,
|
|
prune: bool = False,
|
|
) -> dict[str, int]:
|
|
"""Make the DB agree with what is actually on disk.
|
|
|
|
`backfill_from_markdown` fills in segments and metadata but never touches
|
|
`status` or `markdown_path`, so a video whose .md exists can sit at
|
|
`error`/`no_subtitles`/`pending` forever and the UI keeps showing a failure
|
|
for work that is already done. This walks the markdown tree and repairs:
|
|
|
|
- a row with a real .md but a non-done status -> marked done
|
|
|
|
`prune=True` additionally sends `done` rows whose .md has disappeared back
|
|
to pending. That direction is opt-in because it is destructive when aimed
|
|
at the wrong root: pointed at an empty or unrelated markdown tree it would
|
|
demote every finished video in the database. It is also skipped outright
|
|
when the tree contains no .md at all, which is never a real "everything was
|
|
deleted" state — it means the root is wrong.
|
|
|
|
Returns counts so the caller can report what changed. Idempotent.
|
|
"""
|
|
def _log(msg: str) -> None:
|
|
if log:
|
|
log(msg)
|
|
else:
|
|
logging.getLogger(__name__).info(msg)
|
|
|
|
md_root = Path(md_root)
|
|
data_root = md_root.parent
|
|
out = {
|
|
"scanned": 0, "repaired_done": 0, "orphan_md": 0,
|
|
"missing_md": 0, "backfilled": 0, "stale_dupe": 0,
|
|
}
|
|
if not md_root.exists():
|
|
_log(f"reconcile: markdown root not found: {md_root}")
|
|
return out
|
|
|
|
out["backfilled"] = backfill_from_markdown(store, md_root, log=log)
|
|
|
|
seen: dict[str, Path] = {}
|
|
for md_path in sorted(md_root.rglob("*.md")):
|
|
out["scanned"] += 1
|
|
try:
|
|
text = md_path.read_text(encoding="utf-8")
|
|
except (OSError, UnicodeDecodeError) as exc:
|
|
_log(f"reconcile: skip unreadable {md_path}: {exc}")
|
|
continue
|
|
meta = parse_markdown(text).metadata
|
|
video_id = meta.get("video_id")
|
|
if not video_id:
|
|
continue
|
|
row = store.get_video(video_id)
|
|
if not row:
|
|
out["orphan_md"] += 1
|
|
continue
|
|
rel = md_path.relative_to(data_root).as_posix()
|
|
|
|
# A second .md for a video the DB already resolves elsewhere. Older
|
|
# re-renders built the filename from a differently-formatted date, so
|
|
# they wrote a sibling file the DB never learned about; it is dead
|
|
# weight that every later scan has to wade through.
|
|
canonical = (row.markdown_path or "").replace("\\", "/")
|
|
if video_id in seen or (row.status == "done" and canonical and canonical != rel):
|
|
out["stale_dupe"] += 1
|
|
if prune:
|
|
try:
|
|
md_path.unlink()
|
|
_log(f"reconcile: removed stale duplicate {rel}")
|
|
except OSError as exc:
|
|
_log(f"reconcile: could not remove {rel}: {exc}")
|
|
else:
|
|
_log(f"reconcile: stale duplicate (use prune to delete): {rel}")
|
|
continue
|
|
|
|
seen[video_id] = md_path
|
|
if row.status != "done" or not row.markdown_path:
|
|
store.mark_done(
|
|
video_id, rel,
|
|
meta.get("transcript_lang") or row.transcript_lang,
|
|
meta.get("transcript_src") or row.transcript_src,
|
|
bool(meta.get("has_chapters")) or bool(row.has_chapters),
|
|
)
|
|
out["repaired_done"] += 1
|
|
_log(f"reconcile: {video_id} had a .md on disk but status={row.status} -> done")
|
|
|
|
# The other direction is destructive, so it needs both an explicit opt-in
|
|
# and evidence that we are looking at a real markdown tree.
|
|
if prune and out["scanned"]:
|
|
for row in store.get_all():
|
|
if row.status != "done" or row.video_id in seen:
|
|
continue
|
|
path = data_root / row.markdown_path if row.markdown_path else None
|
|
if path is None or not path.exists():
|
|
store.mark_status(row.video_id, "pending", "markdown file missing on disk")
|
|
out["missing_md"] += 1
|
|
_log(f"reconcile: {row.video_id} marked done but .md is gone -> pending")
|
|
elif prune:
|
|
_log("reconcile: markdown tree is empty — refusing to prune (wrong root?)")
|
|
|
|
_log(
|
|
"reconcile: scanned {scanned} .md, repaired {repaired_done}, "
|
|
"re-queued {missing_md}, orphans {orphan_md}, stale duplicates {stale_dupe}".format(**out)
|
|
)
|
|
return out
|
|
|
|
|
|
def _to_int(value: str | None) -> int | None:
|
|
if value is None:
|
|
return None
|
|
value = str(value).strip().strip('"').strip("'")
|
|
if not value:
|
|
return None
|
|
try:
|
|
return int(value)
|
|
except ValueError:
|
|
try:
|
|
return int(float(value))
|
|
except ValueError:
|
|
return None
|