from __future__ import annotations from pathlib import Path import pytest import yt_dlp from yt_scraper.config import Config from yt_scraper.store import Store, VideoRef from yt_scraper.webapp.jobs import JobManager def test_audio_job_downloads_into_audio_directory(tmp_path, monkeypatch): calls: list[list[str]] = [] class FakeYoutubeDL: def __init__(self, options): self.options = options def __enter__(self): return self def __exit__(self, exc_type, exc_value, traceback): return False def download(self, urls): calls.append(urls) monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL) store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 1) store.upsert_videos([ VideoRef("abc123", "UC1", "Audio test", "https://www.youtube.com/watch?v=abc123") ]) store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]}) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_audio("audio-job", {"mode": "audio", "video_ids": ["abc123"]}) assert (tmp_path / "audio").is_dir() assert calls == [["https://www.youtube.com/watch?v=abc123"]] assert store.get_job("audio-job").status == "done" def test_audio_job_with_video_ids_uses_audio_runner(tmp_path, monkeypatch): store = Store(tmp_path / "state.db") store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]}) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) calls: list[str] = [] monkeypatch.setattr(manager, "_run_audio", lambda job_id, opts: calls.append("audio")) monkeypatch.setattr(manager, "_run_batch", lambda job_id, opts: calls.append("batch")) manager._run_job("audio-job") assert calls == ["audio"] def test_video_job_downloads_webm_after_markdown_exists(tmp_path, monkeypatch): calls: list[list[str]] = [] captured: dict = {} class FakeYoutubeDL: def __init__(self, options): captured.update(options) def __enter__(self): return self def __exit__(self, exc_type, exc_value, traceback): return False def download(self, urls): calls.append(urls) outtmpl = captured["outtmpl"].replace("%(ext)s", "webm") output = tmp_path / Path(outtmpl).relative_to(tmp_path) output.parent.mkdir(parents=True, exist_ok=True) output.write_bytes(b"webm-test") monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL) store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 1) store.upsert_videos([ VideoRef("abc123", "UC1", "A WebM video", "https://www.youtube.com/watch?v=abc123") ]) md = tmp_path / "markdown" / "Alpha" / "a-webm-video.md" md.parent.mkdir(parents=True) md.write_text("# A WebM video\n", encoding="utf-8") store.mark_done("abc123", "markdown/Alpha/a-webm-video.md", "en", "manual", False) store.create_job("video-job", None, {"mode": "video", "video_ids": ["abc123"]}) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_video("video-job", {"mode": "video", "video_ids": ["abc123"]}) row = store.get_video("abc123") assert calls == [["https://www.youtube.com/watch?v=abc123"]] assert captured["merge_output_format"] == "webm" assert captured["js_runtimes"] == {"node": {}} assert "protocol^=m3u8_native" in captured["format"] assert row.video_download_status == "done" assert row.video_filename == "A WebM video.webm" assert row.video_path == "videos/abc123/A WebM video.webm" assert (tmp_path / row.video_path).read_bytes() == b"webm-test" def test_discovery_job_registers_new_videos_without_processing(tmp_path, monkeypatch): store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 1) store.upsert_videos([ VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60) ]) store.mark_status("old", "done") store.create_job("discover-job", "UC1", {"mode": "discover"}) calls: list[str] = [] def fake_discover(url, sleep_subrequests=2.0, limit=None): calls.append((url, limit)) return ("UC1", "Alpha", None, [ VideoRef("new", "UC1", "New", "https://y/watch?v=new", "20240501", 90), VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60), ]) processed = [] monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover) monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", lambda *a, **k: processed.append(a)) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_job("discover-job") # Windowed, not a full walk: the known "old" video ends the scan in one pass. assert calls == [("https://www.youtube.com/@alpha/videos", 30)] assert store.get_video("new").status == "pending" assert store.get_video("old").status == "done" assert processed == [] assert store.get_job("discover-job").status == "done" assert any( event["event"] == "done" and event["data"]["new_videos"] == 1 for event in manager.events_since("discover-job", 0) ) def test_all_channel_discovery_continues_after_one_error(tmp_path, monkeypatch): store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@one", "One", 0) store.upsert_channel("UC2", "@two", "Two", 0) store.create_job("discover-all", None, {"mode": "discover"}) def fake_discover(url, sleep_subrequests=2.0, limit=None): if "@one" in url: raise RuntimeError("temporary failure") return ("UC2", "Two", None, [VideoRef("new2", "UC2", "New", "https://y/watch?v=new2")]) monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_job("discover-all") assert store.get_video("new2").status == "pending" assert store.get_job("discover-all").status == "done" done = [e for e in manager.events_since("discover-all", 0) if e["event"] == "done"][-1] assert done["data"]["errors"] == 1 def _catalog_channel(monkeypatch, catalog, calls): def fake(url, sleep_subrequests=2.0, limit=None): calls.append(limit) return ("UC1", "Alpha", None, catalog[:limit] if limit else list(catalog)) monkeypatch.setattr("yt_scraper.discover.discover_channel", fake) def test_discovery_keeps_the_full_video_count_after_a_windowed_pass(tmp_path, monkeypatch): """The window is 30 wide but the channel has 200 videos — video_count must not collapse to the size of what we just looked at.""" store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 200) catalog = [ VideoRef(f"v{i:03d}", "UC1", f"V{i}", f"https://y/watch?v=v{i:03d}", None, 60) for i in range(200) ] store.upsert_videos(catalog[2:]) # everything except the 2 newest store.create_job("disc", "UC1", {"mode": "discover"}) calls: list = [] _catalog_channel(monkeypatch, catalog, calls) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_job("disc") assert calls == [30], "must not paginate all 200" assert store.get_channel("UC1")["video_count"] == 200 done = [e for e in manager.events_since("disc", 0) if e["event"] == "done"][-1] assert done["data"]["new_videos"] == 2 assert done["data"]["fetched"] == 30 def test_discovery_full_option_walks_the_whole_channel(tmp_path, monkeypatch): store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 3) catalog = [ VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", None, 60) for i in range(3) ] store.upsert_videos(catalog) store.create_job("disc-full", "UC1", {"mode": "discover", "full": True}) calls: list = [] _catalog_channel(monkeypatch, catalog, calls) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_job("disc-full") assert calls == [None], "full rescan must not pass a playlistend" done = [e for e in manager.events_since("disc-full", 0) if e["event"] == "done"][-1] assert done["data"]["new_videos"] == 0 def test_batch_job_reports_videos_without_transcripts(tmp_path, monkeypatch): store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 1) store.upsert_videos([ VideoRef("no-captions", "UC1", "No captions", "https://y/watch?v=no-captions") ]) store.create_job("batch-job", None, {"video_ids": ["no-captions"]}) monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *args, **kwargs: "no_subtitles") manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_batch("batch-job", {"video_ids": ["no-captions"]}) done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1] assert done["data"]["processed"] == 0 assert done["data"]["no_subtitles"] == 1 assert store.get_video("no-captions").markdown_path is None def test_batch_job_downloads_markdown_and_nothing_else(tmp_path, monkeypatch): """The .md button downloads notes only. Thumbnails are already cached by the channel-level paths (add-channel, the Thumbnails tool) and /api/thumbnails/{id} redirects to the CDN for whatever is missing, so fetching one per video here spent a request on an image the UI could already display. """ store = Store(tmp_path / "state.db") store.upsert_channel("UC1", "@alpha", "Alpha", 1) store.upsert_videos([VideoRef("vid1", "UC1", "One", "https://y/watch?v=vid1")]) store.create_job("batch-job", None, {"video_ids": ["vid1"]}) thumbs: list[str] = [] monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *a, **k: "done") monkeypatch.setattr( "yt_scraper.pipeline.cache_thumbnail", lambda _store, video_id, _dir: bool(thumbs.append(video_id)), ) monkeypatch.setattr( "yt_scraper._yt_http.yt_get", lambda *a, **k: pytest.fail("a .md batch must not fetch images"), ) manager = JobManager( store, Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")), ) manager._run_batch("batch-job", {"video_ids": ["vid1"]}) assert thumbs == [] assert not (tmp_path / "thumbnails").exists() done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1] assert done["data"]["processed"] == 1