wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)
This commit is contained in:
@@ -0,0 +1,290 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import yt_dlp
|
||||
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
from yt_scraper.webapp.jobs import JobManager
|
||||
|
||||
|
||||
def test_audio_job_downloads_into_audio_directory(tmp_path, monkeypatch):
|
||||
calls: list[list[str]] = []
|
||||
|
||||
class FakeYoutubeDL:
|
||||
def __init__(self, options):
|
||||
self.options = options
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False
|
||||
|
||||
def download(self, urls):
|
||||
calls.append(urls)
|
||||
|
||||
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
|
||||
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("abc123", "UC1", "Audio test", "https://www.youtube.com/watch?v=abc123")
|
||||
])
|
||||
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_audio("audio-job", {"mode": "audio", "video_ids": ["abc123"]})
|
||||
|
||||
assert (tmp_path / "audio").is_dir()
|
||||
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
|
||||
assert store.get_job("audio-job").status == "done"
|
||||
|
||||
|
||||
def test_audio_job_with_video_ids_uses_audio_runner(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
calls: list[str] = []
|
||||
monkeypatch.setattr(manager, "_run_audio", lambda job_id, opts: calls.append("audio"))
|
||||
monkeypatch.setattr(manager, "_run_batch", lambda job_id, opts: calls.append("batch"))
|
||||
|
||||
manager._run_job("audio-job")
|
||||
|
||||
assert calls == ["audio"]
|
||||
|
||||
|
||||
def test_video_job_downloads_webm_after_markdown_exists(tmp_path, monkeypatch):
|
||||
calls: list[list[str]] = []
|
||||
captured: dict = {}
|
||||
|
||||
class FakeYoutubeDL:
|
||||
def __init__(self, options):
|
||||
captured.update(options)
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False
|
||||
|
||||
def download(self, urls):
|
||||
calls.append(urls)
|
||||
outtmpl = captured["outtmpl"].replace("%(ext)s", "webm")
|
||||
output = tmp_path / Path(outtmpl).relative_to(tmp_path)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
output.write_bytes(b"webm-test")
|
||||
|
||||
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("abc123", "UC1", "A WebM video", "https://www.youtube.com/watch?v=abc123")
|
||||
])
|
||||
md = tmp_path / "markdown" / "Alpha" / "a-webm-video.md"
|
||||
md.parent.mkdir(parents=True)
|
||||
md.write_text("# A WebM video\n", encoding="utf-8")
|
||||
store.mark_done("abc123", "markdown/Alpha/a-webm-video.md", "en", "manual", False)
|
||||
store.create_job("video-job", None, {"mode": "video", "video_ids": ["abc123"]})
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_video("video-job", {"mode": "video", "video_ids": ["abc123"]})
|
||||
|
||||
row = store.get_video("abc123")
|
||||
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
|
||||
assert captured["merge_output_format"] == "webm"
|
||||
assert captured["js_runtimes"] == {"node": {}}
|
||||
assert "protocol^=m3u8_native" in captured["format"]
|
||||
assert row.video_download_status == "done"
|
||||
assert row.video_filename == "A WebM video.webm"
|
||||
assert row.video_path == "videos/abc123/A WebM video.webm"
|
||||
assert (tmp_path / row.video_path).read_bytes() == b"webm-test"
|
||||
|
||||
|
||||
def test_discovery_job_registers_new_videos_without_processing(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60)
|
||||
])
|
||||
store.mark_status("old", "done")
|
||||
store.create_job("discover-job", "UC1", {"mode": "discover"})
|
||||
calls: list[str] = []
|
||||
|
||||
def fake_discover(url, sleep_subrequests=2.0, limit=None):
|
||||
calls.append((url, limit))
|
||||
return ("UC1", "Alpha", None, [
|
||||
VideoRef("new", "UC1", "New", "https://y/watch?v=new", "20240501", 90),
|
||||
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60),
|
||||
])
|
||||
|
||||
processed = []
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
|
||||
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", lambda *a, **k: processed.append(a))
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_job("discover-job")
|
||||
|
||||
# Windowed, not a full walk: the known "old" video ends the scan in one pass.
|
||||
assert calls == [("https://www.youtube.com/@alpha/videos", 30)]
|
||||
assert store.get_video("new").status == "pending"
|
||||
assert store.get_video("old").status == "done"
|
||||
assert processed == []
|
||||
assert store.get_job("discover-job").status == "done"
|
||||
assert any(
|
||||
event["event"] == "done" and event["data"]["new_videos"] == 1
|
||||
for event in manager.events_since("discover-job", 0)
|
||||
)
|
||||
|
||||
|
||||
def test_all_channel_discovery_continues_after_one_error(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@one", "One", 0)
|
||||
store.upsert_channel("UC2", "@two", "Two", 0)
|
||||
store.create_job("discover-all", None, {"mode": "discover"})
|
||||
|
||||
def fake_discover(url, sleep_subrequests=2.0, limit=None):
|
||||
if "@one" in url:
|
||||
raise RuntimeError("temporary failure")
|
||||
return ("UC2", "Two", None, [VideoRef("new2", "UC2", "New", "https://y/watch?v=new2")])
|
||||
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_job("discover-all")
|
||||
|
||||
assert store.get_video("new2").status == "pending"
|
||||
assert store.get_job("discover-all").status == "done"
|
||||
done = [e for e in manager.events_since("discover-all", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["errors"] == 1
|
||||
|
||||
|
||||
def _catalog_channel(monkeypatch, catalog, calls):
|
||||
def fake(url, sleep_subrequests=2.0, limit=None):
|
||||
calls.append(limit)
|
||||
return ("UC1", "Alpha", None, catalog[:limit] if limit else list(catalog))
|
||||
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
|
||||
|
||||
|
||||
def test_discovery_keeps_the_full_video_count_after_a_windowed_pass(tmp_path, monkeypatch):
|
||||
"""The window is 30 wide but the channel has 200 videos — video_count must
|
||||
not collapse to the size of what we just looked at."""
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 200)
|
||||
catalog = [
|
||||
VideoRef(f"v{i:03d}", "UC1", f"V{i}", f"https://y/watch?v=v{i:03d}", None, 60)
|
||||
for i in range(200)
|
||||
]
|
||||
store.upsert_videos(catalog[2:]) # everything except the 2 newest
|
||||
store.create_job("disc", "UC1", {"mode": "discover"})
|
||||
calls: list = []
|
||||
_catalog_channel(monkeypatch, catalog, calls)
|
||||
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
manager._run_job("disc")
|
||||
|
||||
assert calls == [30], "must not paginate all 200"
|
||||
assert store.get_channel("UC1")["video_count"] == 200
|
||||
done = [e for e in manager.events_since("disc", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["new_videos"] == 2
|
||||
assert done["data"]["fetched"] == 30
|
||||
|
||||
|
||||
def test_discovery_full_option_walks_the_whole_channel(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 3)
|
||||
catalog = [
|
||||
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", None, 60) for i in range(3)
|
||||
]
|
||||
store.upsert_videos(catalog)
|
||||
store.create_job("disc-full", "UC1", {"mode": "discover", "full": True})
|
||||
calls: list = []
|
||||
_catalog_channel(monkeypatch, catalog, calls)
|
||||
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
manager._run_job("disc-full")
|
||||
|
||||
assert calls == [None], "full rescan must not pass a playlistend"
|
||||
done = [e for e in manager.events_since("disc-full", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["new_videos"] == 0
|
||||
|
||||
|
||||
def test_batch_job_reports_videos_without_transcripts(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("no-captions", "UC1", "No captions", "https://y/watch?v=no-captions")
|
||||
])
|
||||
store.create_job("batch-job", None, {"video_ids": ["no-captions"]})
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *args, **kwargs: "no_subtitles")
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_batch("batch-job", {"video_ids": ["no-captions"]})
|
||||
|
||||
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["processed"] == 0
|
||||
assert done["data"]["no_subtitles"] == 1
|
||||
assert store.get_video("no-captions").markdown_path is None
|
||||
|
||||
|
||||
def test_batch_job_downloads_markdown_and_nothing_else(tmp_path, monkeypatch):
|
||||
"""The .md button downloads notes only.
|
||||
|
||||
Thumbnails are already cached by the channel-level paths (add-channel, the
|
||||
Thumbnails tool) and /api/thumbnails/{id} redirects to the CDN for whatever
|
||||
is missing, so fetching one per video here spent a request on an image the
|
||||
UI could already display.
|
||||
"""
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([VideoRef("vid1", "UC1", "One", "https://y/watch?v=vid1")])
|
||||
store.create_job("batch-job", None, {"video_ids": ["vid1"]})
|
||||
|
||||
thumbs: list[str] = []
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *a, **k: "done")
|
||||
monkeypatch.setattr(
|
||||
"yt_scraper.pipeline.cache_thumbnail",
|
||||
lambda _store, video_id, _dir: bool(thumbs.append(video_id)),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"yt_scraper._yt_http.yt_get",
|
||||
lambda *a, **k: pytest.fail("a .md batch must not fetch images"),
|
||||
)
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_batch("batch-job", {"video_ids": ["vid1"]})
|
||||
|
||||
assert thumbs == []
|
||||
assert not (tmp_path / "thumbnails").exists()
|
||||
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["processed"] == 1
|
||||
Reference in New Issue
Block a user