from __future__ import annotations import logging import re from collections import Counter from pathlib import Path from typing import Iterable from .store import Store log = logging.getLogger(__name__) # Minimal Spanish + English stopword list (no extra deps). _STOPWORDS = { # es "el", "la", "los", "las", "un", "una", "unos", "unas", "de", "del", "al", "a", "y", "o", "u", "que", "en", "como", "por", "para", "su", "sus", "se", "si", "no", "con", "es", "son", "fue", "era", "este", "esta", "estos", "estas", "eso", "esa", "esos", "esas", "lo", "le", "les", "te", "me", "se", "nos", "os", "pero", "mas", "muy", "ya", "cuando", "donde", "quien", "como", "todo", "todos", "toda", "todas", "nada", "algo", "tambien", "asi", "hay", "habia", "tiene", "tener", "puede", "pueden", "esto", "esta", "sin", "sobre", "entre", "hasta", "desde", "mi", "tu", "yo", "el", "ella", "ellos", "ellas", "nosotros", "vosotros", "ustedes", "porque", "pues", "entonces", # en "the", "a", "an", "and", "or", "but", "of", "to", "in", "on", "at", "by", "for", "with", "from", "is", "are", "was", "were", "be", "been", "being", "this", "that", "these", "those", "it", "its", "as", "if", "not", "no", "so", "do", "does", "did", "have", "has", "had", "i", "you", "he", "she", "we", "they", "my", "your", "his", "her", "our", "their", "me", "him", "us", "them", "can", "could", "would", "should", "will", "shall", "may", "might", "just", "like", "what", "which", "who", "when", "where", "why", "how", "all", "any", "both", "each", "more", "most", "other", "some", "such", } _WORD = re.compile(r"[A-Za-zÁÉÍÓÚÜÑáéíóúüñ]{3,}") def _iter_texts(store: Store, channel_id: str | None) -> Iterable[str]: sql = "SELECT text FROM transcript_segments" params: list = [] if channel_id: sql += " WHERE video_id IN (SELECT video_id FROM videos WHERE channel_id = ?)" params.append(channel_id) conn = store._connect() try: cur = conn.execute(sql, params) for row in cur: yield row["text"] finally: conn.close() def _tokenize(text: str) -> Iterable[str]: for m in _WORD.finditer(text.lower()): w = m.group(0) if w not in _STOPWORDS: yield w def word_frequency(store: Store, channel_id: str | None = None, top: int | None = None) -> dict[str, int]: counter: Counter[str] = Counter() for text in _iter_texts(store, channel_id): counter.update(_tokenize(text)) items = counter.most_common(top) if top else counter.most_common() return dict(items) def top_words(store: Store, n: int = 50, channel_id: str | None = None) -> list[tuple[str, int]]: return list(word_frequency(store, channel_id, top=n).items()) def term_timeline(store: Store, term: str, channel_id: str | None = None) -> list[tuple[str, int]]: """Return [(YYYY-MM, count)] of months where `term` appears in transcripts.""" term_l = term.lower().strip() if not term_l: return [] sql = ( "SELECT substr(v.upload_date,1,6) AS month, COUNT(*) AS n " "FROM transcript_fts f JOIN videos v ON v.video_id = f.video_id " "WHERE transcript_fts MATCH ?" ) params: list = [f'"{term_l}"'] if channel_id: sql += " AND v.channel_id = ?" params.append(channel_id) sql += " AND v.upload_date IS NOT NULL GROUP BY month ORDER BY month" conn = store._connect() try: rows = conn.execute(sql, params).fetchall() finally: conn.close() out: list[tuple[str, int]] = [] for r in rows: m = r["month"] if m and len(m) == 6: out.append((f"{m[:4]}-{m[4:6]}", r["n"])) return out def render_wordcloud(freq: dict[str, int], out_path: str | Path) -> Path: from wordcloud import WordCloud import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt out = Path(out_path) out.parent.mkdir(parents=True, exist_ok=True) wc = WordCloud( width=1600, height=900, background_color="#0a0a0f", colormap="Reds", max_words=200, ).generate_from_frequencies(freq) fig, ax = plt.subplots(figsize=(16, 9), dpi=100) ax.imshow(wc, interpolation="bilinear") ax.set_axis_off() fig.tight_layout(pad=0) fig.savefig(str(out), facecolor="#0a0a0f") plt.close(fig) return out def render_top_words_chart(top: list[tuple[str, int]], out_path: str | Path) -> Path: import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt out = Path(out_path) out.parent.mkdir(parents=True, exist_ok=True) items = top[:25][::-1] labels = [t[0] for t in items] values = [t[1] for t in items] fig, ax = plt.subplots(figsize=(10, 8), dpi=100) ax.barh(labels, values, color="#f43f5e") ax.set_facecolor("#0a0a0f") fig.patch.set_facecolor("#0a0a0f") ax.tick_params(colors="#e5e7eb") for spine in ax.spines.values(): spine.set_color("#27272a") ax.set_title("Top terms", color="#e5e7eb") fig.tight_layout() fig.savefig(str(out), facecolor="#0a0a0f") plt.close(fig) return out def render_timeline_chart(timeline: list[tuple[str, int]], term: str, out_path: str | Path) -> Path: import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt out = Path(out_path) out.parent.mkdir(parents=True, exist_ok=True) labels = [t[0] for t in timeline] values = [t[1] for t in timeline] fig, ax = plt.subplots(figsize=(12, 5), dpi=100) ax.plot(labels, values, marker="o", color="#f43f5e") ax.set_facecolor("#0a0a0f") fig.patch.set_facecolor("#0a0a0f") ax.tick_params(colors="#e5e7eb", axisxlabelrotation=45) for spine in ax.spines.values(): spine.set_color("#27272a") ax.set_title(f"Mentions over time: {term}", color="#e5e7eb") fig.tight_layout() fig.savefig(str(out), facecolor="#0a0a0f") plt.close(fig) return out