291 lines
11 KiB
Python
291 lines
11 KiB
Python
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
import yt_dlp
|
|
|
|
from yt_scraper.config import Config
|
|
from yt_scraper.store import Store, VideoRef
|
|
from yt_scraper.webapp.jobs import JobManager
|
|
|
|
|
|
def test_audio_job_downloads_into_audio_directory(tmp_path, monkeypatch):
|
|
calls: list[list[str]] = []
|
|
|
|
class FakeYoutubeDL:
|
|
def __init__(self, options):
|
|
self.options = options
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, exc_type, exc_value, traceback):
|
|
return False
|
|
|
|
def download(self, urls):
|
|
calls.append(urls)
|
|
|
|
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
|
|
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
|
store.upsert_videos([
|
|
VideoRef("abc123", "UC1", "Audio test", "https://www.youtube.com/watch?v=abc123")
|
|
])
|
|
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
|
|
manager._run_audio("audio-job", {"mode": "audio", "video_ids": ["abc123"]})
|
|
|
|
assert (tmp_path / "audio").is_dir()
|
|
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
|
|
assert store.get_job("audio-job").status == "done"
|
|
|
|
|
|
def test_audio_job_with_video_ids_uses_audio_runner(tmp_path, monkeypatch):
|
|
store = Store(tmp_path / "state.db")
|
|
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
calls: list[str] = []
|
|
monkeypatch.setattr(manager, "_run_audio", lambda job_id, opts: calls.append("audio"))
|
|
monkeypatch.setattr(manager, "_run_batch", lambda job_id, opts: calls.append("batch"))
|
|
|
|
manager._run_job("audio-job")
|
|
|
|
assert calls == ["audio"]
|
|
|
|
|
|
def test_video_job_downloads_webm_after_markdown_exists(tmp_path, monkeypatch):
|
|
calls: list[list[str]] = []
|
|
captured: dict = {}
|
|
|
|
class FakeYoutubeDL:
|
|
def __init__(self, options):
|
|
captured.update(options)
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, exc_type, exc_value, traceback):
|
|
return False
|
|
|
|
def download(self, urls):
|
|
calls.append(urls)
|
|
outtmpl = captured["outtmpl"].replace("%(ext)s", "webm")
|
|
output = tmp_path / Path(outtmpl).relative_to(tmp_path)
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
output.write_bytes(b"webm-test")
|
|
|
|
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
|
store.upsert_videos([
|
|
VideoRef("abc123", "UC1", "A WebM video", "https://www.youtube.com/watch?v=abc123")
|
|
])
|
|
md = tmp_path / "markdown" / "Alpha" / "a-webm-video.md"
|
|
md.parent.mkdir(parents=True)
|
|
md.write_text("# A WebM video\n", encoding="utf-8")
|
|
store.mark_done("abc123", "markdown/Alpha/a-webm-video.md", "en", "manual", False)
|
|
store.create_job("video-job", None, {"mode": "video", "video_ids": ["abc123"]})
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
|
|
manager._run_video("video-job", {"mode": "video", "video_ids": ["abc123"]})
|
|
|
|
row = store.get_video("abc123")
|
|
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
|
|
assert captured["merge_output_format"] == "webm"
|
|
assert captured["js_runtimes"] == {"node": {}}
|
|
assert "protocol^=m3u8_native" in captured["format"]
|
|
assert row.video_download_status == "done"
|
|
assert row.video_filename == "A WebM video.webm"
|
|
assert row.video_path == "videos/abc123/A WebM video.webm"
|
|
assert (tmp_path / row.video_path).read_bytes() == b"webm-test"
|
|
|
|
|
|
def test_discovery_job_registers_new_videos_without_processing(tmp_path, monkeypatch):
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
|
store.upsert_videos([
|
|
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60)
|
|
])
|
|
store.mark_status("old", "done")
|
|
store.create_job("discover-job", "UC1", {"mode": "discover"})
|
|
calls: list[str] = []
|
|
|
|
def fake_discover(url, sleep_subrequests=2.0, limit=None):
|
|
calls.append((url, limit))
|
|
return ("UC1", "Alpha", None, [
|
|
VideoRef("new", "UC1", "New", "https://y/watch?v=new", "20240501", 90),
|
|
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60),
|
|
])
|
|
|
|
processed = []
|
|
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
|
|
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", lambda *a, **k: processed.append(a))
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
|
|
manager._run_job("discover-job")
|
|
|
|
# Windowed, not a full walk: the known "old" video ends the scan in one pass.
|
|
assert calls == [("https://www.youtube.com/@alpha/videos", 30)]
|
|
assert store.get_video("new").status == "pending"
|
|
assert store.get_video("old").status == "done"
|
|
assert processed == []
|
|
assert store.get_job("discover-job").status == "done"
|
|
assert any(
|
|
event["event"] == "done" and event["data"]["new_videos"] == 1
|
|
for event in manager.events_since("discover-job", 0)
|
|
)
|
|
|
|
|
|
def test_all_channel_discovery_continues_after_one_error(tmp_path, monkeypatch):
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@one", "One", 0)
|
|
store.upsert_channel("UC2", "@two", "Two", 0)
|
|
store.create_job("discover-all", None, {"mode": "discover"})
|
|
|
|
def fake_discover(url, sleep_subrequests=2.0, limit=None):
|
|
if "@one" in url:
|
|
raise RuntimeError("temporary failure")
|
|
return ("UC2", "Two", None, [VideoRef("new2", "UC2", "New", "https://y/watch?v=new2")])
|
|
|
|
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
|
|
manager._run_job("discover-all")
|
|
|
|
assert store.get_video("new2").status == "pending"
|
|
assert store.get_job("discover-all").status == "done"
|
|
done = [e for e in manager.events_since("discover-all", 0) if e["event"] == "done"][-1]
|
|
assert done["data"]["errors"] == 1
|
|
|
|
|
|
def _catalog_channel(monkeypatch, catalog, calls):
|
|
def fake(url, sleep_subrequests=2.0, limit=None):
|
|
calls.append(limit)
|
|
return ("UC1", "Alpha", None, catalog[:limit] if limit else list(catalog))
|
|
|
|
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
|
|
|
|
|
|
def test_discovery_keeps_the_full_video_count_after_a_windowed_pass(tmp_path, monkeypatch):
|
|
"""The window is 30 wide but the channel has 200 videos — video_count must
|
|
not collapse to the size of what we just looked at."""
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 200)
|
|
catalog = [
|
|
VideoRef(f"v{i:03d}", "UC1", f"V{i}", f"https://y/watch?v=v{i:03d}", None, 60)
|
|
for i in range(200)
|
|
]
|
|
store.upsert_videos(catalog[2:]) # everything except the 2 newest
|
|
store.create_job("disc", "UC1", {"mode": "discover"})
|
|
calls: list = []
|
|
_catalog_channel(monkeypatch, catalog, calls)
|
|
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
manager._run_job("disc")
|
|
|
|
assert calls == [30], "must not paginate all 200"
|
|
assert store.get_channel("UC1")["video_count"] == 200
|
|
done = [e for e in manager.events_since("disc", 0) if e["event"] == "done"][-1]
|
|
assert done["data"]["new_videos"] == 2
|
|
assert done["data"]["fetched"] == 30
|
|
|
|
|
|
def test_discovery_full_option_walks_the_whole_channel(tmp_path, monkeypatch):
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 3)
|
|
catalog = [
|
|
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", None, 60) for i in range(3)
|
|
]
|
|
store.upsert_videos(catalog)
|
|
store.create_job("disc-full", "UC1", {"mode": "discover", "full": True})
|
|
calls: list = []
|
|
_catalog_channel(monkeypatch, catalog, calls)
|
|
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
manager._run_job("disc-full")
|
|
|
|
assert calls == [None], "full rescan must not pass a playlistend"
|
|
done = [e for e in manager.events_since("disc-full", 0) if e["event"] == "done"][-1]
|
|
assert done["data"]["new_videos"] == 0
|
|
|
|
|
|
def test_batch_job_reports_videos_without_transcripts(tmp_path, monkeypatch):
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
|
store.upsert_videos([
|
|
VideoRef("no-captions", "UC1", "No captions", "https://y/watch?v=no-captions")
|
|
])
|
|
store.create_job("batch-job", None, {"video_ids": ["no-captions"]})
|
|
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *args, **kwargs: "no_subtitles")
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
|
|
manager._run_batch("batch-job", {"video_ids": ["no-captions"]})
|
|
|
|
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
|
|
assert done["data"]["processed"] == 0
|
|
assert done["data"]["no_subtitles"] == 1
|
|
assert store.get_video("no-captions").markdown_path is None
|
|
|
|
|
|
def test_batch_job_downloads_markdown_and_nothing_else(tmp_path, monkeypatch):
|
|
"""The .md button downloads notes only.
|
|
|
|
Thumbnails are already cached by the channel-level paths (add-channel, the
|
|
Thumbnails tool) and /api/thumbnails/{id} redirects to the CDN for whatever
|
|
is missing, so fetching one per video here spent a request on an image the
|
|
UI could already display.
|
|
"""
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
|
store.upsert_videos([VideoRef("vid1", "UC1", "One", "https://y/watch?v=vid1")])
|
|
store.create_job("batch-job", None, {"video_ids": ["vid1"]})
|
|
|
|
thumbs: list[str] = []
|
|
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *a, **k: "done")
|
|
monkeypatch.setattr(
|
|
"yt_scraper.pipeline.cache_thumbnail",
|
|
lambda _store, video_id, _dir: bool(thumbs.append(video_id)),
|
|
)
|
|
monkeypatch.setattr(
|
|
"yt_scraper._yt_http.yt_get",
|
|
lambda *a, **k: pytest.fail("a .md batch must not fetch images"),
|
|
)
|
|
manager = JobManager(
|
|
store,
|
|
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
|
)
|
|
|
|
manager._run_batch("batch-job", {"video_ids": ["vid1"]})
|
|
|
|
assert thumbs == []
|
|
assert not (tmp_path / "thumbnails").exists()
|
|
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
|
|
assert done["data"]["processed"] == 1
|