wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)

This commit is contained in:
urieljareth
2026-08-22 19:37:23 -06:00
parent 4f5a68b572
commit 8a59b39c98
103 changed files with 70954 additions and 1825 deletions
+290
View File
@@ -0,0 +1,290 @@
from __future__ import annotations
from pathlib import Path
import pytest
import yt_dlp
from yt_scraper.config import Config
from yt_scraper.store import Store, VideoRef
from yt_scraper.webapp.jobs import JobManager
def test_audio_job_downloads_into_audio_directory(tmp_path, monkeypatch):
calls: list[list[str]] = []
class FakeYoutubeDL:
def __init__(self, options):
self.options = options
def __enter__(self):
return self
def __exit__(self, exc_type, exc_value, traceback):
return False
def download(self, urls):
calls.append(urls)
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("abc123", "UC1", "Audio test", "https://www.youtube.com/watch?v=abc123")
])
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_audio("audio-job", {"mode": "audio", "video_ids": ["abc123"]})
assert (tmp_path / "audio").is_dir()
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
assert store.get_job("audio-job").status == "done"
def test_audio_job_with_video_ids_uses_audio_runner(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
calls: list[str] = []
monkeypatch.setattr(manager, "_run_audio", lambda job_id, opts: calls.append("audio"))
monkeypatch.setattr(manager, "_run_batch", lambda job_id, opts: calls.append("batch"))
manager._run_job("audio-job")
assert calls == ["audio"]
def test_video_job_downloads_webm_after_markdown_exists(tmp_path, monkeypatch):
calls: list[list[str]] = []
captured: dict = {}
class FakeYoutubeDL:
def __init__(self, options):
captured.update(options)
def __enter__(self):
return self
def __exit__(self, exc_type, exc_value, traceback):
return False
def download(self, urls):
calls.append(urls)
outtmpl = captured["outtmpl"].replace("%(ext)s", "webm")
output = tmp_path / Path(outtmpl).relative_to(tmp_path)
output.parent.mkdir(parents=True, exist_ok=True)
output.write_bytes(b"webm-test")
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("abc123", "UC1", "A WebM video", "https://www.youtube.com/watch?v=abc123")
])
md = tmp_path / "markdown" / "Alpha" / "a-webm-video.md"
md.parent.mkdir(parents=True)
md.write_text("# A WebM video\n", encoding="utf-8")
store.mark_done("abc123", "markdown/Alpha/a-webm-video.md", "en", "manual", False)
store.create_job("video-job", None, {"mode": "video", "video_ids": ["abc123"]})
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_video("video-job", {"mode": "video", "video_ids": ["abc123"]})
row = store.get_video("abc123")
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
assert captured["merge_output_format"] == "webm"
assert captured["js_runtimes"] == {"node": {}}
assert "protocol^=m3u8_native" in captured["format"]
assert row.video_download_status == "done"
assert row.video_filename == "A WebM video.webm"
assert row.video_path == "videos/abc123/A WebM video.webm"
assert (tmp_path / row.video_path).read_bytes() == b"webm-test"
def test_discovery_job_registers_new_videos_without_processing(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60)
])
store.mark_status("old", "done")
store.create_job("discover-job", "UC1", {"mode": "discover"})
calls: list[str] = []
def fake_discover(url, sleep_subrequests=2.0, limit=None):
calls.append((url, limit))
return ("UC1", "Alpha", None, [
VideoRef("new", "UC1", "New", "https://y/watch?v=new", "20240501", 90),
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60),
])
processed = []
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", lambda *a, **k: processed.append(a))
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("discover-job")
# Windowed, not a full walk: the known "old" video ends the scan in one pass.
assert calls == [("https://www.youtube.com/@alpha/videos", 30)]
assert store.get_video("new").status == "pending"
assert store.get_video("old").status == "done"
assert processed == []
assert store.get_job("discover-job").status == "done"
assert any(
event["event"] == "done" and event["data"]["new_videos"] == 1
for event in manager.events_since("discover-job", 0)
)
def test_all_channel_discovery_continues_after_one_error(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@one", "One", 0)
store.upsert_channel("UC2", "@two", "Two", 0)
store.create_job("discover-all", None, {"mode": "discover"})
def fake_discover(url, sleep_subrequests=2.0, limit=None):
if "@one" in url:
raise RuntimeError("temporary failure")
return ("UC2", "Two", None, [VideoRef("new2", "UC2", "New", "https://y/watch?v=new2")])
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("discover-all")
assert store.get_video("new2").status == "pending"
assert store.get_job("discover-all").status == "done"
done = [e for e in manager.events_since("discover-all", 0) if e["event"] == "done"][-1]
assert done["data"]["errors"] == 1
def _catalog_channel(monkeypatch, catalog, calls):
def fake(url, sleep_subrequests=2.0, limit=None):
calls.append(limit)
return ("UC1", "Alpha", None, catalog[:limit] if limit else list(catalog))
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
def test_discovery_keeps_the_full_video_count_after_a_windowed_pass(tmp_path, monkeypatch):
"""The window is 30 wide but the channel has 200 videos — video_count must
not collapse to the size of what we just looked at."""
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 200)
catalog = [
VideoRef(f"v{i:03d}", "UC1", f"V{i}", f"https://y/watch?v=v{i:03d}", None, 60)
for i in range(200)
]
store.upsert_videos(catalog[2:]) # everything except the 2 newest
store.create_job("disc", "UC1", {"mode": "discover"})
calls: list = []
_catalog_channel(monkeypatch, catalog, calls)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("disc")
assert calls == [30], "must not paginate all 200"
assert store.get_channel("UC1")["video_count"] == 200
done = [e for e in manager.events_since("disc", 0) if e["event"] == "done"][-1]
assert done["data"]["new_videos"] == 2
assert done["data"]["fetched"] == 30
def test_discovery_full_option_walks_the_whole_channel(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 3)
catalog = [
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", None, 60) for i in range(3)
]
store.upsert_videos(catalog)
store.create_job("disc-full", "UC1", {"mode": "discover", "full": True})
calls: list = []
_catalog_channel(monkeypatch, catalog, calls)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("disc-full")
assert calls == [None], "full rescan must not pass a playlistend"
done = [e for e in manager.events_since("disc-full", 0) if e["event"] == "done"][-1]
assert done["data"]["new_videos"] == 0
def test_batch_job_reports_videos_without_transcripts(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("no-captions", "UC1", "No captions", "https://y/watch?v=no-captions")
])
store.create_job("batch-job", None, {"video_ids": ["no-captions"]})
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *args, **kwargs: "no_subtitles")
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_batch("batch-job", {"video_ids": ["no-captions"]})
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
assert done["data"]["processed"] == 0
assert done["data"]["no_subtitles"] == 1
assert store.get_video("no-captions").markdown_path is None
def test_batch_job_downloads_markdown_and_nothing_else(tmp_path, monkeypatch):
"""The .md button downloads notes only.
Thumbnails are already cached by the channel-level paths (add-channel, the
Thumbnails tool) and /api/thumbnails/{id} redirects to the CDN for whatever
is missing, so fetching one per video here spent a request on an image the
UI could already display.
"""
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([VideoRef("vid1", "UC1", "One", "https://y/watch?v=vid1")])
store.create_job("batch-job", None, {"video_ids": ["vid1"]})
thumbs: list[str] = []
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *a, **k: "done")
monkeypatch.setattr(
"yt_scraper.pipeline.cache_thumbnail",
lambda _store, video_id, _dir: bool(thumbs.append(video_id)),
)
monkeypatch.setattr(
"yt_scraper._yt_http.yt_get",
lambda *a, **k: pytest.fail("a .md batch must not fetch images"),
)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_batch("batch-job", {"video_ids": ["vid1"]})
assert thumbs == []
assert not (tmp_path / "thumbnails").exists()
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
assert done["data"]["processed"] == 1