feat: local content-mining platform for YouTube creator scraping
Scraper de canales de YouTube hacia notas Markdown para base de conocimiento (Obsidian-ready), con plataforma web local. Engine + CLI (Workstream A): - Modular pipeline: discover/extract/parse/chapters/render/store + ratelimit - SQLite store con migración idempotente: FTS5 (transcript search), columnas de metadata enriquecida, tablas cookies_meta y scrape_jobs - Módulos: segments, cookies (Netscape vault), export (json/csv/srt/html), analysis (word freq/timeline/wordcloud), monitor (watch loop), pipeline - CLI Click group: search, export, audio, channels, watch, analyze, re-render - Fix del bug de scoping de cookies en cli.py Webapp local (Workstream B): - FastAPI backend: dashboard, channels, videos facetado, transcript, search FTS, analysis, scrape jobs con SSE, cookies drag-and-drop, exports, folders (abrir en OS), tools (re-render, formato) - SPA no-build (Alpine.js + Tailwind + Chart.js por CDN): 9 vistas, tema dark command-center con acento rojo→rosa, cookie vault drag-drop, consola de scrapeo con progreso live vía SSE Launcher + subagentes (Workstream C): - start-server.bat / stop-server.bat con auto port-scan + browser open - .opencode/agent/webapp-builder.md + .opencode/goals/webapp-build.md Tests: 38 pytest verdes. Sin funcionalidad de IA (enfoque data-mining).
This commit is contained in:
@@ -0,0 +1,185 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
from .parse import Segment
|
||||
from .store import SearchHit, Store
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_TS = re.compile(r"^(\d{1,2}):(\d{2})(?::(\d{2}))?$")
|
||||
_SEG_LINE = re.compile(r"^\*\*(?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\*\*\s*[·\-]\s*(?P<text>.+?)\s*$")
|
||||
_CHAPTER = re.compile(r"^###\s+(?P<title>.+?)\s*\((?P<ts>\d{1,2}:\d{2}(?::\d{2})?)\)\s*$")
|
||||
_FRONTMATTER_KEY = re.compile(r"^([A-Za-z0-9_]+)\s*:\s*(.*)$")
|
||||
|
||||
|
||||
def ts_to_seconds(ts: str) -> float:
|
||||
m = _TS.match(ts.strip())
|
||||
if not m:
|
||||
return 0.0
|
||||
if m.group(3): # h:m:s
|
||||
return int(m.group(1)) * 3600 + int(m.group(2)) * 60 + int(m.group(3))
|
||||
return int(m.group(1)) * 60 + int(m.group(2))
|
||||
|
||||
|
||||
def seconds_to_ts(sec: float) -> str:
|
||||
total = int(sec)
|
||||
h, rem = divmod(total, 3600)
|
||||
m, s = divmod(rem, 60)
|
||||
return f"{h:d}:{m:02d}:{s:02d}" if h else f"{m:d}:{s:02d}"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedMarkdown:
|
||||
metadata: dict
|
||||
segments: list[Segment]
|
||||
chapters: list[dict] # {"title": str, "start": float}
|
||||
|
||||
|
||||
def parse_markdown(text: str) -> ParsedMarkdown:
|
||||
lines = text.splitlines()
|
||||
metadata: dict[str, str] = {}
|
||||
segments: list[Segment] = []
|
||||
chapters: list[dict] = []
|
||||
in_front = False
|
||||
front_done = False
|
||||
i = 0
|
||||
# frontmatter
|
||||
if lines and lines[0].strip() == "---":
|
||||
i = 1
|
||||
in_front = True
|
||||
while i < len(lines):
|
||||
if lines[i].strip() == "---":
|
||||
front_done = True
|
||||
i += 1
|
||||
break
|
||||
m = _FRONTMATTER_KEY.match(lines[i])
|
||||
if m:
|
||||
metadata[m.group(1).lower()] = _strip_quotes(m.group(2).strip())
|
||||
i += 1
|
||||
if not front_done:
|
||||
i = 0
|
||||
# body
|
||||
current_start = 0.0
|
||||
while i < len(lines):
|
||||
line = lines[i]
|
||||
s = line.strip()
|
||||
cm = _CHAPTER.match(s)
|
||||
if cm:
|
||||
current_start = ts_to_seconds(cm.group("ts"))
|
||||
chapters.append({"title": cm.group("title").strip(), "start": current_start})
|
||||
i += 1
|
||||
continue
|
||||
sm = _SEG_LINE.match(s)
|
||||
if sm:
|
||||
start = ts_to_seconds(sm.group("ts"))
|
||||
segments.append(Segment(start=start, end=start, text=sm.group("text").strip()))
|
||||
i += 1
|
||||
return ParsedMarkdown(metadata=metadata, segments=segments, chapters=chapters)
|
||||
|
||||
|
||||
def _strip_quotes(value: str) -> str:
|
||||
v = value.strip()
|
||||
if len(v) >= 2 and v[0] == v[-1] and v[0] in ("'", '"'):
|
||||
return v[1:-1]
|
||||
return v
|
||||
|
||||
|
||||
def store_segments(store: Store, video_id: str, segments: list[Segment]) -> None:
|
||||
store.store_segments(video_id, segments)
|
||||
|
||||
|
||||
def search(store: Store, query: str, channel_id: str | None = None, limit: int = 50) -> list[SearchHit]:
|
||||
return store.search_segments(query, channel_id=channel_id, limit=limit)
|
||||
|
||||
|
||||
def backfill_from_markdown(
|
||||
store: Store,
|
||||
md_root: Path,
|
||||
log: Callable[[str], None] | None = None,
|
||||
) -> int:
|
||||
"""Parse all done .md files under md_root, populate segments/FTS/metadata. Idempotent."""
|
||||
def _log(msg: str) -> None:
|
||||
if log:
|
||||
log(msg)
|
||||
else:
|
||||
logmod = logging.getLogger(__name__)
|
||||
logmod.info(msg)
|
||||
|
||||
md_root = Path(md_root)
|
||||
if not md_root.exists():
|
||||
_log(f"backfill: markdown root not found: {md_root}")
|
||||
return 0
|
||||
md_files = sorted(md_root.rglob("*.md"))
|
||||
_log(f"backfill: scanning {len(md_files)} markdown files")
|
||||
n = 0
|
||||
for md_path in md_files:
|
||||
try:
|
||||
text = md_path.read_text(encoding="utf-8")
|
||||
except OSError as exc:
|
||||
_log(f"backfill: skip unreadable {md_path}: {exc}")
|
||||
continue
|
||||
parsed = parse_markdown(text)
|
||||
video_id = parsed.metadata.get("video_id")
|
||||
if not video_id:
|
||||
continue
|
||||
existing = store.get_video(video_id)
|
||||
if not existing:
|
||||
_log(f"backfill: video {video_id} not in DB (orphan md), skipping")
|
||||
continue
|
||||
# always (re)populate upload_date if missing — cheap, idempotent, gated by NULL
|
||||
ud = parsed.metadata.get("upload_date")
|
||||
if ud:
|
||||
compact = ud.replace("-", "")
|
||||
if compact.isdigit():
|
||||
store.set_upload_date(video_id, compact)
|
||||
if store.has_segments(video_id) and existing.segments_json:
|
||||
continue # segments + rich metadata already populated
|
||||
# metadata from frontmatter
|
||||
tags_raw = parsed.metadata.get("tags", "")
|
||||
tags: list[str] = []
|
||||
if tags_raw:
|
||||
try:
|
||||
tags = json.loads(tags_raw) if tags_raw.startswith("[") else [t.strip() for t in tags_raw.split(",")]
|
||||
except json.JSONDecodeError:
|
||||
tags = []
|
||||
seg_json = json.dumps(
|
||||
[{"start": s.start, "end": s.end, "text": s.text} for s in parsed.segments],
|
||||
ensure_ascii=False,
|
||||
)
|
||||
ch_json = json.dumps(parsed.chapters, ensure_ascii=False)
|
||||
store.update_video_metadata(
|
||||
video_id,
|
||||
view_count=_to_int(parsed.metadata.get("views")),
|
||||
like_count=_to_int(parsed.metadata.get("likes")),
|
||||
tags=tags or None,
|
||||
thumbnail=parsed.metadata.get("thumbnail") or None,
|
||||
description=None,
|
||||
chapters_json=ch_json,
|
||||
segments_json=seg_json,
|
||||
)
|
||||
store.store_segments(video_id, parsed.segments)
|
||||
n += 1
|
||||
_log(f"backfill: populated {n} videos")
|
||||
return n
|
||||
|
||||
|
||||
def _to_int(value: str | None) -> int | None:
|
||||
if value is None:
|
||||
return None
|
||||
value = str(value).strip().strip('"').strip("'")
|
||||
if not value:
|
||||
return None
|
||||
try:
|
||||
return int(value)
|
||||
except ValueError:
|
||||
try:
|
||||
return int(float(value))
|
||||
except ValueError:
|
||||
return None
|
||||
Reference in New Issue
Block a user