feat: local content-mining platform for YouTube creator scraping

Scraper de canales de YouTube hacia notas Markdown para base de conocimiento
(Obsidian-ready), con plataforma web local.

Engine + CLI (Workstream A):
- Modular pipeline: discover/extract/parse/chapters/render/store + ratelimit
- SQLite store con migración idempotente: FTS5 (transcript search), columnas
  de metadata enriquecida, tablas cookies_meta y scrape_jobs
- Módulos: segments, cookies (Netscape vault), export (json/csv/srt/html),
  analysis (word freq/timeline/wordcloud), monitor (watch loop), pipeline
- CLI Click group: search, export, audio, channels, watch, analyze, re-render
- Fix del bug de scoping de cookies en cli.py

Webapp local (Workstream B):
- FastAPI backend: dashboard, channels, videos facetado, transcript, search
  FTS, analysis, scrape jobs con SSE, cookies drag-and-drop, exports,
  folders (abrir en OS), tools (re-render, formato)
- SPA no-build (Alpine.js + Tailwind + Chart.js por CDN): 9 vistas, tema
  dark command-center con acento rojo→rosa, cookie vault drag-drop,
  consola de scrapeo con progreso live vía SSE

Launcher + subagentes (Workstream C):
- start-server.bat / stop-server.bat con auto port-scan + browser open
- .opencode/agent/webapp-builder.md + .opencode/goals/webapp-build.md

Tests: 38 pytest verdes. Sin funcionalidad de IA (enfoque data-mining).
This commit is contained in:
urieljareth
2026-07-26 23:19:34 -06:00
commit 621bbc5f5c
45 changed files with 6541 additions and 0 deletions
+163
View File
@@ -0,0 +1,163 @@
from __future__ import annotations
import logging
import re
from collections import Counter
from pathlib import Path
from typing import Iterable
from .store import Store
log = logging.getLogger(__name__)
# Minimal Spanish + English stopword list (no extra deps).
_STOPWORDS = {
# es
"el", "la", "los", "las", "un", "una", "unos", "unas", "de", "del", "al", "a", "y", "o", "u",
"que", "en", "como", "por", "para", "su", "sus", "se", "si", "no", "con", "es", "son", "fue",
"era", "este", "esta", "estos", "estas", "eso", "esa", "esos", "esas", "lo", "le", "les", "te",
"me", "se", "nos", "os", "pero", "mas", "muy", "ya", "cuando", "donde", "quien", "como", "todo",
"todos", "toda", "todas", "nada", "algo", "tambien", "asi", "hay", "habia", "tiene", "tener",
"puede", "pueden", "esto", "esta", "sin", "sobre", "entre", "hasta", "desde", "mi", "tu", "yo",
"el", "ella", "ellos", "ellas", "nosotros", "vosotros", "ustedes", "porque", "pues", "entonces",
# en
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at", "by", "for", "with", "from",
"is", "are", "was", "were", "be", "been", "being", "this", "that", "these", "those", "it", "its",
"as", "if", "not", "no", "so", "do", "does", "did", "have", "has", "had", "i", "you", "he", "she",
"we", "they", "my", "your", "his", "her", "our", "their", "me", "him", "us", "them", "can", "could",
"would", "should", "will", "shall", "may", "might", "just", "like", "what", "which", "who", "when",
"where", "why", "how", "all", "any", "both", "each", "more", "most", "other", "some", "such",
}
_WORD = re.compile(r"[A-Za-zÁÉÍÓÚÜÑáéíóúüñ]{3,}")
def _iter_texts(store: Store, channel_id: str | None) -> Iterable[str]:
sql = "SELECT text FROM transcript_segments"
params: list = []
if channel_id:
sql += " WHERE video_id IN (SELECT video_id FROM videos WHERE channel_id = ?)"
params.append(channel_id)
conn = store._connect()
try:
cur = conn.execute(sql, params)
for row in cur:
yield row["text"]
finally:
conn.close()
def _tokenize(text: str) -> Iterable[str]:
for m in _WORD.finditer(text.lower()):
w = m.group(0)
if w not in _STOPWORDS:
yield w
def word_frequency(store: Store, channel_id: str | None = None, top: int | None = None) -> dict[str, int]:
counter: Counter[str] = Counter()
for text in _iter_texts(store, channel_id):
counter.update(_tokenize(text))
items = counter.most_common(top) if top else counter.most_common()
return dict(items)
def top_words(store: Store, n: int = 50, channel_id: str | None = None) -> list[tuple[str, int]]:
return list(word_frequency(store, channel_id, top=n).items())
def term_timeline(store: Store, term: str, channel_id: str | None = None) -> list[tuple[str, int]]:
"""Return [(YYYY-MM, count)] of months where `term` appears in transcripts."""
term_l = term.lower().strip()
if not term_l:
return []
sql = (
"SELECT substr(v.upload_date,1,6) AS month, COUNT(*) AS n "
"FROM transcript_fts f JOIN videos v ON v.video_id = f.video_id "
"WHERE transcript_fts MATCH ?"
)
params: list = [f'"{term_l}"']
if channel_id:
sql += " AND v.channel_id = ?"
params.append(channel_id)
sql += " AND v.upload_date IS NOT NULL GROUP BY month ORDER BY month"
conn = store._connect()
try:
rows = conn.execute(sql, params).fetchall()
finally:
conn.close()
out: list[tuple[str, int]] = []
for r in rows:
m = r["month"]
if m and len(m) == 6:
out.append((f"{m[:4]}-{m[4:6]}", r["n"]))
return out
def render_wordcloud(freq: dict[str, int], out_path: str | Path) -> Path:
from wordcloud import WordCloud
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
out = Path(out_path)
out.parent.mkdir(parents=True, exist_ok=True)
wc = WordCloud(
width=1600, height=900, background_color="#0a0a0f",
colormap="Reds", max_words=200,
).generate_from_frequencies(freq)
fig, ax = plt.subplots(figsize=(16, 9), dpi=100)
ax.imshow(wc, interpolation="bilinear")
ax.set_axis_off()
fig.tight_layout(pad=0)
fig.savefig(str(out), facecolor="#0a0a0f")
plt.close(fig)
return out
def render_top_words_chart(top: list[tuple[str, int]], out_path: str | Path) -> Path:
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
out = Path(out_path)
out.parent.mkdir(parents=True, exist_ok=True)
items = top[:25][::-1]
labels = [t[0] for t in items]
values = [t[1] for t in items]
fig, ax = plt.subplots(figsize=(10, 8), dpi=100)
ax.barh(labels, values, color="#f43f5e")
ax.set_facecolor("#0a0a0f")
fig.patch.set_facecolor("#0a0a0f")
ax.tick_params(colors="#e5e7eb")
for spine in ax.spines.values():
spine.set_color("#27272a")
ax.set_title("Top terms", color="#e5e7eb")
fig.tight_layout()
fig.savefig(str(out), facecolor="#0a0a0f")
plt.close(fig)
return out
def render_timeline_chart(timeline: list[tuple[str, int]], term: str, out_path: str | Path) -> Path:
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
out = Path(out_path)
out.parent.mkdir(parents=True, exist_ok=True)
labels = [t[0] for t in timeline]
values = [t[1] for t in timeline]
fig, ax = plt.subplots(figsize=(12, 5), dpi=100)
ax.plot(labels, values, marker="o", color="#f43f5e")
ax.set_facecolor("#0a0a0f")
fig.patch.set_facecolor("#0a0a0f")
ax.tick_params(colors="#e5e7eb", axisxlabelrotation=45)
for spine in ax.spines.values():
spine.set_color("#27272a")
ax.set_title(f"Mentions over time: {term}", color="#e5e7eb")
fig.tight_layout()
fig.savefig(str(out), facecolor="#0a0a0f")
plt.close(fig)
return out