from __future__ import annotations import json import logging import re from dataclasses import dataclass from pathlib import Path from typing import Callable from .parse import Segment from .store import Store log = logging.getLogger(__name__) _TS = re.compile(r"^(\d{1,2}):(\d{2})(?::(\d{2}))?$") _SEG_LINE = re.compile(r"^\*\*(?P\d{1,2}:\d{2}(?::\d{2})?)\*\*\s*[·\-]\s*(?P.+?)\s*$") _CHAPTER = re.compile(r"^###\s+(?P.+?)\s*\((?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\)\s*$") _FRONTMATTER_KEY = re.compile(r"^([A-Za-z0-9_]+)\s*:\s*(.*)$") def ts_to_seconds(ts: str) -> float: m = _TS.match(ts.strip()) if not m: return 0.0 if m.group(3): # h:m:s return int(m.group(1)) * 3600 + int(m.group(2)) * 60 + int(m.group(3)) return int(m.group(1)) * 60 + int(m.group(2)) def seconds_to_ts(sec: float) -> str: total = int(sec) h, rem = divmod(total, 3600) m, s = divmod(rem, 60) return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}" @dataclass class ParsedMarkdown: metadata: dict segments: list[Segment] chapters: list[dict] # {"title": str, "start": float} def parse_markdown(text: str) -> ParsedMarkdown: lines = text.splitlines() metadata: dict[str, str] = {} segments: list[Segment] = [] chapters: list[dict] = [] in_front = False front_done = False i = 0 # frontmatter if lines and lines[0].strip() == "---": i = 1 in_front = True while i < len(lines): if lines[i].strip() == "---": front_done = True i += 1 break m = _FRONTMATTER_KEY.match(lines[i]) if m: metadata[m.group(1).lower()] = _strip_quotes(m.group(2).strip()) i += 1 if not front_done: i = 0 # body current_start = 0.0 while i < len(lines): line = lines[i] s = line.strip() cm = _CHAPTER.match(s) if cm: current_start = ts_to_seconds(cm.group("ts")) chapters.append({"title": cm.group("title").strip(), "start": current_start}) i += 1 continue sm = _SEG_LINE.match(s) if sm: start = ts_to_seconds(sm.group("ts")) segments.append(Segment(start=start, end=start, text=sm.group("text").strip())) i += 1 return ParsedMarkdown(metadata=metadata, segments=segments, chapters=chapters) def _strip_quotes(value: str) -> str: v = value.strip() if len(v) >= 2 and v[0] == v[-1] and v[0] in ("'", '"'): return v[1:-1] return v def backfill_from_markdown( store: Store, md_root: Path, log: Callable[[str], None] | None = None, ) -> int: """Parse all done .md files under md_root, populate segments/FTS/metadata. Idempotent.""" def _log(msg: str) -> None: if log: log(msg) else: logmod = logging.getLogger(__name__) logmod.info(msg) md_root = Path(md_root) if not md_root.exists(): _log(f"backfill: markdown root not found: {md_root}") return 0 md_files = sorted(md_root.rglob("*.md")) _log(f"backfill: scanning {len(md_files)} markdown files") n = 0 for md_path in md_files: try: text = md_path.read_text(encoding="utf-8") except (OSError, UnicodeDecodeError) as exc: # UnicodeDecodeError is a ValueError, not an OSError — letting it # escape aborted the loop and silently skipped every later file. _log(f"backfill: skip unreadable {md_path}: {exc}") continue parsed = parse_markdown(text) video_id = parsed.metadata.get("video_id") if not video_id: continue existing = store.get_video(video_id) if not existing: _log(f"backfill: video {video_id} not in DB (orphan md), skipping") continue # always (re)populate upload_date if missing — cheap, idempotent, gated by NULL ud = parsed.metadata.get("upload_date") if ud: compact = ud.replace("-", "") if compact.isdigit(): store.set_upload_date(video_id, compact) if store.has_segments(video_id) and existing.segments_json: continue # segments + rich metadata already populated # metadata from frontmatter tags_raw = parsed.metadata.get("tags", "") tags: list[str] = [] if tags_raw: try: tags = json.loads(tags_raw) if tags_raw.startswith("[") else [t.strip() for t in tags_raw.split(",")] except json.JSONDecodeError: tags = [] seg_json = json.dumps( [{"start": s.start, "end": s.end, "text": s.text} for s in parsed.segments], ensure_ascii=False, ) ch_json = json.dumps(parsed.chapters, ensure_ascii=False) store.update_video_metadata( video_id, view_count=_to_int(parsed.metadata.get("views")), like_count=_to_int(parsed.metadata.get("likes")), tags=tags or None, thumbnail=parsed.metadata.get("thumbnail") or None, description=None, chapters_json=ch_json, segments_json=seg_json, ) store.store_segments(video_id, parsed.segments) n += 1 _log(f"backfill: populated {n} videos") return n def reconcile_markdown( store: Store, md_root: Path, log: Callable[[str], None] | None = None, *, prune: bool = False, ) -> dict[str, int]: """Make the DB agree with what is actually on disk. `backfill_from_markdown` fills in segments and metadata but never touches `status` or `markdown_path`, so a video whose .md exists can sit at `error`/`no_subtitles`/`pending` forever and the UI keeps showing a failure for work that is already done. This walks the markdown tree and repairs: - a row with a real .md but a non-done status -> marked done `prune=True` additionally sends `done` rows whose .md has disappeared back to pending. That direction is opt-in because it is destructive when aimed at the wrong root: pointed at an empty or unrelated markdown tree it would demote every finished video in the database. It is also skipped outright when the tree contains no .md at all, which is never a real "everything was deleted" state — it means the root is wrong. Returns counts so the caller can report what changed. Idempotent. """ def _log(msg: str) -> None: if log: log(msg) else: logging.getLogger(__name__).info(msg) md_root = Path(md_root) data_root = md_root.parent out = { "scanned": 0, "repaired_done": 0, "orphan_md": 0, "missing_md": 0, "backfilled": 0, "stale_dupe": 0, } if not md_root.exists(): _log(f"reconcile: markdown root not found: {md_root}") return out out["backfilled"] = backfill_from_markdown(store, md_root, log=log) seen: dict[str, Path] = {} for md_path in sorted(md_root.rglob("*.md")): out["scanned"] += 1 try: text = md_path.read_text(encoding="utf-8") except (OSError, UnicodeDecodeError) as exc: _log(f"reconcile: skip unreadable {md_path}: {exc}") continue meta = parse_markdown(text).metadata video_id = meta.get("video_id") if not video_id: continue row = store.get_video(video_id) if not row: out["orphan_md"] += 1 continue rel = md_path.relative_to(data_root).as_posix() # A second .md for a video the DB already resolves elsewhere. Older # re-renders built the filename from a differently-formatted date, so # they wrote a sibling file the DB never learned about; it is dead # weight that every later scan has to wade through. canonical = (row.markdown_path or "").replace("\\", "/") if video_id in seen or (row.status == "done" and canonical and canonical != rel): out["stale_dupe"] += 1 if prune: try: md_path.unlink() _log(f"reconcile: removed stale duplicate {rel}") except OSError as exc: _log(f"reconcile: could not remove {rel}: {exc}") else: _log(f"reconcile: stale duplicate (use prune to delete): {rel}") continue seen[video_id] = md_path if row.status != "done" or not row.markdown_path: store.mark_done( video_id, rel, meta.get("transcript_lang") or row.transcript_lang, meta.get("transcript_src") or row.transcript_src, bool(meta.get("has_chapters")) or bool(row.has_chapters), ) out["repaired_done"] += 1 _log(f"reconcile: {video_id} had a .md on disk but status={row.status} -> done") # The other direction is destructive, so it needs both an explicit opt-in # and evidence that we are looking at a real markdown tree. if prune and out["scanned"]: for row in store.get_all(): if row.status != "done" or row.video_id in seen: continue path = data_root / row.markdown_path if row.markdown_path else None if path is None or not path.exists(): store.mark_status(row.video_id, "pending", "markdown file missing on disk") out["missing_md"] += 1 _log(f"reconcile: {row.video_id} marked done but .md is gone -> pending") elif prune: _log("reconcile: markdown tree is empty — refusing to prune (wrong root?)") _log( "reconcile: scanned {scanned} .md, repaired {repaired_done}, " "re-queued {missing_md}, orphans {orphan_md}, stale duplicates {stale_dupe}".format(**out) ) return out def _to_int(value: str | None) -> int | None: if value is None: return None value = str(value).strip().strip('"').strip("'") if not value: return None try: return int(value) except ValueError: try: return int(float(value)) except ValueError: return None