feat: local content-mining platform for YouTube creator scraping
Scraper de canales de YouTube hacia notas Markdown para base de conocimiento (Obsidian-ready), con plataforma web local. Engine + CLI (Workstream A): - Modular pipeline: discover/extract/parse/chapters/render/store + ratelimit - SQLite store con migración idempotente: FTS5 (transcript search), columnas de metadata enriquecida, tablas cookies_meta y scrape_jobs - Módulos: segments, cookies (Netscape vault), export (json/csv/srt/html), analysis (word freq/timeline/wordcloud), monitor (watch loop), pipeline - CLI Click group: search, export, audio, channels, watch, analyze, re-render - Fix del bug de scoping de cookies en cli.py Webapp local (Workstream B): - FastAPI backend: dashboard, channels, videos facetado, transcript, search FTS, analysis, scrape jobs con SSE, cookies drag-and-drop, exports, folders (abrir en OS), tools (re-render, formato) - SPA no-build (Alpine.js + Tailwind + Chart.js por CDN): 9 vistas, tema dark command-center con acento rojo→rosa, cookie vault drag-drop, consola de scrapeo con progreso live vía SSE Launcher + subagentes (Workstream C): - start-server.bat / stop-server.bat con auto port-scan + browser open - .opencode/agent/webapp-builder.md + .opencode/goals/webapp-build.md Tests: 38 pytest verdes. Sin funcionalidad de IA (enfoque data-mining).
This commit is contained in:
@@ -0,0 +1,163 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
from .store import Store
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Minimal Spanish + English stopword list (no extra deps).
|
||||
_STOPWORDS = {
|
||||
# es
|
||||
"el", "la", "los", "las", "un", "una", "unos", "unas", "de", "del", "al", "a", "y", "o", "u",
|
||||
"que", "en", "como", "por", "para", "su", "sus", "se", "si", "no", "con", "es", "son", "fue",
|
||||
"era", "este", "esta", "estos", "estas", "eso", "esa", "esos", "esas", "lo", "le", "les", "te",
|
||||
"me", "se", "nos", "os", "pero", "mas", "muy", "ya", "cuando", "donde", "quien", "como", "todo",
|
||||
"todos", "toda", "todas", "nada", "algo", "tambien", "asi", "hay", "habia", "tiene", "tener",
|
||||
"puede", "pueden", "esto", "esta", "sin", "sobre", "entre", "hasta", "desde", "mi", "tu", "yo",
|
||||
"el", "ella", "ellos", "ellas", "nosotros", "vosotros", "ustedes", "porque", "pues", "entonces",
|
||||
# en
|
||||
"the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at", "by", "for", "with", "from",
|
||||
"is", "are", "was", "were", "be", "been", "being", "this", "that", "these", "those", "it", "its",
|
||||
"as", "if", "not", "no", "so", "do", "does", "did", "have", "has", "had", "i", "you", "he", "she",
|
||||
"we", "they", "my", "your", "his", "her", "our", "their", "me", "him", "us", "them", "can", "could",
|
||||
"would", "should", "will", "shall", "may", "might", "just", "like", "what", "which", "who", "when",
|
||||
"where", "why", "how", "all", "any", "both", "each", "more", "most", "other", "some", "such",
|
||||
}
|
||||
|
||||
_WORD = re.compile(r"[A-Za-zÁÉÍÓÚÜÑáéíóúüñ]{3,}")
|
||||
|
||||
|
||||
def _iter_texts(store: Store, channel_id: str | None) -> Iterable[str]:
|
||||
sql = "SELECT text FROM transcript_segments"
|
||||
params: list = []
|
||||
if channel_id:
|
||||
sql += " WHERE video_id IN (SELECT video_id FROM videos WHERE channel_id = ?)"
|
||||
params.append(channel_id)
|
||||
conn = store._connect()
|
||||
try:
|
||||
cur = conn.execute(sql, params)
|
||||
for row in cur:
|
||||
yield row["text"]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def _tokenize(text: str) -> Iterable[str]:
|
||||
for m in _WORD.finditer(text.lower()):
|
||||
w = m.group(0)
|
||||
if w not in _STOPWORDS:
|
||||
yield w
|
||||
|
||||
|
||||
def word_frequency(store: Store, channel_id: str | None = None, top: int | None = None) -> dict[str, int]:
|
||||
counter: Counter[str] = Counter()
|
||||
for text in _iter_texts(store, channel_id):
|
||||
counter.update(_tokenize(text))
|
||||
items = counter.most_common(top) if top else counter.most_common()
|
||||
return dict(items)
|
||||
|
||||
|
||||
def top_words(store: Store, n: int = 50, channel_id: str | None = None) -> list[tuple[str, int]]:
|
||||
return list(word_frequency(store, channel_id, top=n).items())
|
||||
|
||||
|
||||
def term_timeline(store: Store, term: str, channel_id: str | None = None) -> list[tuple[str, int]]:
|
||||
"""Return [(YYYY-MM, count)] of months where `term` appears in transcripts."""
|
||||
term_l = term.lower().strip()
|
||||
if not term_l:
|
||||
return []
|
||||
sql = (
|
||||
"SELECT substr(v.upload_date,1,6) AS month, COUNT(*) AS n "
|
||||
"FROM transcript_fts f JOIN videos v ON v.video_id = f.video_id "
|
||||
"WHERE transcript_fts MATCH ?"
|
||||
)
|
||||
params: list = [f'"{term_l}"']
|
||||
if channel_id:
|
||||
sql += " AND v.channel_id = ?"
|
||||
params.append(channel_id)
|
||||
sql += " AND v.upload_date IS NOT NULL GROUP BY month ORDER BY month"
|
||||
conn = store._connect()
|
||||
try:
|
||||
rows = conn.execute(sql, params).fetchall()
|
||||
finally:
|
||||
conn.close()
|
||||
out: list[tuple[str, int]] = []
|
||||
for r in rows:
|
||||
m = r["month"]
|
||||
if m and len(m) == 6:
|
||||
out.append((f"{m[:4]}-{m[4:6]}", r["n"]))
|
||||
return out
|
||||
|
||||
|
||||
def render_wordcloud(freq: dict[str, int], out_path: str | Path) -> Path:
|
||||
from wordcloud import WordCloud
|
||||
import matplotlib
|
||||
matplotlib.use("Agg")
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
out = Path(out_path)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
wc = WordCloud(
|
||||
width=1600, height=900, background_color="#0a0a0f",
|
||||
colormap="Reds", max_words=200,
|
||||
).generate_from_frequencies(freq)
|
||||
fig, ax = plt.subplots(figsize=(16, 9), dpi=100)
|
||||
ax.imshow(wc, interpolation="bilinear")
|
||||
ax.set_axis_off()
|
||||
fig.tight_layout(pad=0)
|
||||
fig.savefig(str(out), facecolor="#0a0a0f")
|
||||
plt.close(fig)
|
||||
return out
|
||||
|
||||
|
||||
def render_top_words_chart(top: list[tuple[str, int]], out_path: str | Path) -> Path:
|
||||
import matplotlib
|
||||
matplotlib.use("Agg")
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
out = Path(out_path)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
items = top[:25][::-1]
|
||||
labels = [t[0] for t in items]
|
||||
values = [t[1] for t in items]
|
||||
fig, ax = plt.subplots(figsize=(10, 8), dpi=100)
|
||||
ax.barh(labels, values, color="#f43f5e")
|
||||
ax.set_facecolor("#0a0a0f")
|
||||
fig.patch.set_facecolor("#0a0a0f")
|
||||
ax.tick_params(colors="#e5e7eb")
|
||||
for spine in ax.spines.values():
|
||||
spine.set_color("#27272a")
|
||||
ax.set_title("Top terms", color="#e5e7eb")
|
||||
fig.tight_layout()
|
||||
fig.savefig(str(out), facecolor="#0a0a0f")
|
||||
plt.close(fig)
|
||||
return out
|
||||
|
||||
|
||||
def render_timeline_chart(timeline: list[tuple[str, int]], term: str, out_path: str | Path) -> Path:
|
||||
import matplotlib
|
||||
matplotlib.use("Agg")
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
out = Path(out_path)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
labels = [t[0] for t in timeline]
|
||||
values = [t[1] for t in timeline]
|
||||
fig, ax = plt.subplots(figsize=(12, 5), dpi=100)
|
||||
ax.plot(labels, values, marker="o", color="#f43f5e")
|
||||
ax.set_facecolor("#0a0a0f")
|
||||
fig.patch.set_facecolor("#0a0a0f")
|
||||
ax.tick_params(colors="#e5e7eb", axisxlabelrotation=45)
|
||||
for spine in ax.spines.values():
|
||||
spine.set_color("#27272a")
|
||||
ax.set_title(f"Mentions over time: {term}", color="#e5e7eb")
|
||||
fig.tight_layout()
|
||||
fig.savefig(str(out), facecolor="#0a0a0f")
|
||||
plt.close(fig)
|
||||
return out
|
||||
Reference in New Issue
Block a user