Files
yt-channel-scraper/src/yt_scraper/segments.py
T
urieljareth b3b27ce883 feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo
Extraccion autenticada:
- import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20
- extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla
  con sesion logueada (members-only); regex y opener cacheados
- js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts
- rutas de Brave multiplataforma (Windows/macOS/Linux)

Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL,
memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics

Rendimiento:
- entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota)
- _rank_unranked con guarda (antes full-scan en cada arranque/import)
- upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL)
- thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool)
- reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md),
  healthz expone reconcile_done
- Store.transaction(): escrituras por video agrupadas (~6 commits -> 3)

Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render,
order_pending->store, keep_ref->config, seconds_to_ts solo en segments);
re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done);
fuera wrappers muertos de segments.py
2026-09-10 00:32:19 -06:00

290 lines
10 KiB
Python

from __future__ import annotations
import json
import logging
import re
from dataclasses import dataclass
from pathlib import Path
from typing import Callable
from .parse import Segment
from .store import Store
log = logging.getLogger(__name__)
_TS = re.compile(r"^(\d{1,2}):(\d{2})(?::(\d{2}))?$")
_SEG_LINE = re.compile(r"^\*\*(?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\*\*\s*[·\-]\s*(?P<text>.+?)\s*$")
_CHAPTER = re.compile(r"^###\s+(?P<title>.+?)\s*\((?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\)\s*$")
_FRONTMATTER_KEY = re.compile(r"^([A-Za-z0-9_]+)\s*:\s*(.*)$")
def ts_to_seconds(ts: str) -> float:
m = _TS.match(ts.strip())
if not m:
return 0.0
if m.group(3): # h:m:s
return int(m.group(1)) * 3600 + int(m.group(2)) * 60 + int(m.group(3))
return int(m.group(1)) * 60 + int(m.group(2))
def seconds_to_ts(sec: float) -> str:
total = int(sec)
h, rem = divmod(total, 3600)
m, s = divmod(rem, 60)
return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}"
@dataclass
class ParsedMarkdown:
metadata: dict
segments: list[Segment]
chapters: list[dict] # {"title": str, "start": float}
def parse_markdown(text: str) -> ParsedMarkdown:
lines = text.splitlines()
metadata: dict[str, str] = {}
segments: list[Segment] = []
chapters: list[dict] = []
in_front = False
front_done = False
i = 0
# frontmatter
if lines and lines[0].strip() == "---":
i = 1
in_front = True
while i < len(lines):
if lines[i].strip() == "---":
front_done = True
i += 1
break
m = _FRONTMATTER_KEY.match(lines[i])
if m:
metadata[m.group(1).lower()] = _strip_quotes(m.group(2).strip())
i += 1
if not front_done:
i = 0
# body
current_start = 0.0
while i < len(lines):
line = lines[i]
s = line.strip()
cm = _CHAPTER.match(s)
if cm:
current_start = ts_to_seconds(cm.group("ts"))
chapters.append({"title": cm.group("title").strip(), "start": current_start})
i += 1
continue
sm = _SEG_LINE.match(s)
if sm:
start = ts_to_seconds(sm.group("ts"))
segments.append(Segment(start=start, end=start, text=sm.group("text").strip()))
i += 1
return ParsedMarkdown(metadata=metadata, segments=segments, chapters=chapters)
def _strip_quotes(value: str) -> str:
v = value.strip()
if len(v) >= 2 and v[0] == v[-1] and v[0] in ("'", '"'):
return v[1:-1]
return v
def backfill_from_markdown(
store: Store,
md_root: Path,
log: Callable[[str], None] | None = None,
) -> int:
"""Parse all done .md files under md_root, populate segments/FTS/metadata. Idempotent."""
def _log(msg: str) -> None:
if log:
log(msg)
else:
logmod = logging.getLogger(__name__)
logmod.info(msg)
md_root = Path(md_root)
if not md_root.exists():
_log(f"backfill: markdown root not found: {md_root}")
return 0
md_files = sorted(md_root.rglob("*.md"))
_log(f"backfill: scanning {len(md_files)} markdown files")
n = 0
for md_path in md_files:
try:
text = md_path.read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError) as exc:
# UnicodeDecodeError is a ValueError, not an OSError — letting it
# escape aborted the loop and silently skipped every later file.
_log(f"backfill: skip unreadable {md_path}: {exc}")
continue
parsed = parse_markdown(text)
video_id = parsed.metadata.get("video_id")
if not video_id:
continue
existing = store.get_video(video_id)
if not existing:
_log(f"backfill: video {video_id} not in DB (orphan md), skipping")
continue
# always (re)populate upload_date if missing — cheap, idempotent, gated by NULL
ud = parsed.metadata.get("upload_date")
if ud:
compact = ud.replace("-", "")
if compact.isdigit():
store.set_upload_date(video_id, compact)
if store.has_segments(video_id) and existing.segments_json:
continue # segments + rich metadata already populated
# metadata from frontmatter
tags_raw = parsed.metadata.get("tags", "")
tags: list[str] = []
if tags_raw:
try:
tags = json.loads(tags_raw) if tags_raw.startswith("[") else [t.strip() for t in tags_raw.split(",")]
except json.JSONDecodeError:
tags = []
seg_json = json.dumps(
[{"start": s.start, "end": s.end, "text": s.text} for s in parsed.segments],
ensure_ascii=False,
)
ch_json = json.dumps(parsed.chapters, ensure_ascii=False)
store.update_video_metadata(
video_id,
view_count=_to_int(parsed.metadata.get("views")),
like_count=_to_int(parsed.metadata.get("likes")),
tags=tags or None,
thumbnail=parsed.metadata.get("thumbnail") or None,
description=None,
chapters_json=ch_json,
segments_json=seg_json,
)
store.store_segments(video_id, parsed.segments)
n += 1
_log(f"backfill: populated {n} videos")
return n
def reconcile_markdown(
store: Store,
md_root: Path,
log: Callable[[str], None] | None = None,
*,
prune: bool = False,
) -> dict[str, int]:
"""Make the DB agree with what is actually on disk.
`backfill_from_markdown` fills in segments and metadata but never touches
`status` or `markdown_path`, so a video whose .md exists can sit at
`error`/`no_subtitles`/`pending` forever and the UI keeps showing a failure
for work that is already done. This walks the markdown tree and repairs:
- a row with a real .md but a non-done status -> marked done
`prune=True` additionally sends `done` rows whose .md has disappeared back
to pending. That direction is opt-in because it is destructive when aimed
at the wrong root: pointed at an empty or unrelated markdown tree it would
demote every finished video in the database. It is also skipped outright
when the tree contains no .md at all, which is never a real "everything was
deleted" state — it means the root is wrong.
Returns counts so the caller can report what changed. Idempotent.
"""
def _log(msg: str) -> None:
if log:
log(msg)
else:
logging.getLogger(__name__).info(msg)
md_root = Path(md_root)
data_root = md_root.parent
out = {
"scanned": 0, "repaired_done": 0, "orphan_md": 0,
"missing_md": 0, "backfilled": 0, "stale_dupe": 0,
}
if not md_root.exists():
_log(f"reconcile: markdown root not found: {md_root}")
return out
out["backfilled"] = backfill_from_markdown(store, md_root, log=log)
seen: dict[str, Path] = {}
for md_path in sorted(md_root.rglob("*.md")):
out["scanned"] += 1
try:
text = md_path.read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError) as exc:
_log(f"reconcile: skip unreadable {md_path}: {exc}")
continue
meta = parse_markdown(text).metadata
video_id = meta.get("video_id")
if not video_id:
continue
row = store.get_video(video_id)
if not row:
out["orphan_md"] += 1
continue
rel = md_path.relative_to(data_root).as_posix()
# A second .md for a video the DB already resolves elsewhere. Older
# re-renders built the filename from a differently-formatted date, so
# they wrote a sibling file the DB never learned about; it is dead
# weight that every later scan has to wade through.
canonical = (row.markdown_path or "").replace("\\", "/")
if video_id in seen or (row.status == "done" and canonical and canonical != rel):
out["stale_dupe"] += 1
if prune:
try:
md_path.unlink()
_log(f"reconcile: removed stale duplicate {rel}")
except OSError as exc:
_log(f"reconcile: could not remove {rel}: {exc}")
else:
_log(f"reconcile: stale duplicate (use prune to delete): {rel}")
continue
seen[video_id] = md_path
if row.status != "done" or not row.markdown_path:
store.mark_done(
video_id, rel,
meta.get("transcript_lang") or row.transcript_lang,
meta.get("transcript_src") or row.transcript_src,
bool(meta.get("has_chapters")) or bool(row.has_chapters),
)
out["repaired_done"] += 1
_log(f"reconcile: {video_id} had a .md on disk but status={row.status} -> done")
# The other direction is destructive, so it needs both an explicit opt-in
# and evidence that we are looking at a real markdown tree.
if prune and out["scanned"]:
for row in store.get_all():
if row.status != "done" or row.video_id in seen:
continue
path = data_root / row.markdown_path if row.markdown_path else None
if path is None or not path.exists():
store.mark_status(row.video_id, "pending", "markdown file missing on disk")
out["missing_md"] += 1
_log(f"reconcile: {row.video_id} marked done but .md is gone -> pending")
elif prune:
_log("reconcile: markdown tree is empty — refusing to prune (wrong root?)")
_log(
"reconcile: scanned {scanned} .md, repaired {repaired_done}, "
"re-queued {missing_md}, orphans {orphan_md}, stale duplicates {stale_dupe}".format(**out)
)
return out
def _to_int(value: str | None) -> int | None:
if value is None:
return None
value = str(value).strip().strip('"').strip("'")
if not value:
return None
try:
return int(value)
except ValueError:
try:
return int(float(value))
except ValueError:
return None