"""Windowed channel sync: only fetch what is newer than what we already have. The whole point is request economy against YouTube, so these tests assert on the `limit` values handed to yt-dlp (which become `playlistend`, i.e. how many continuation pages get requested), not just on the refs that come back. """ from __future__ import annotations import pytest from yt_scraper.discover import discover_incremental from yt_scraper.store import VideoRef def _ref(vid: str, upload_date: str | None = None, url: str | None = None) -> VideoRef: return VideoRef( video_id=vid, channel_id="UC1", title=vid, url=url or f"https://www.youtube.com/watch?v={vid}", upload_date=upload_date, ) def _fake_channel(monkeypatch, catalog: list[VideoRef], calls: list | None = None): """Patch discover_channel with a channel whose /videos tab is `catalog` (newest first) and that honours `limit` the way playlistend does.""" def fake(url, sleep_subrequests=2.0, limit=None): if calls is not None: calls.append(limit) return ("UC1", "Alpha", "http://avatar", catalog[:limit] if limit else list(catalog)) monkeypatch.setattr("yt_scraper.discover.discover_channel", fake) def test_stops_at_first_run_of_known_videos(monkeypatch): catalog = [_ref(f"v{i:03d}") for i in range(500)] known = {r.video_id for r in catalog[5:]} # everything except the 5 newest calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3) assert calls == [30], "one pass only — must not paginate the whole channel" assert result.fetched == 30 assert [r.video_id for r in result.new_refs] == [f"v{i:03d}" for i in range(5)] assert result.caught_up is True assert result.full_scan is False def test_widens_window_when_the_whole_window_is_new(monkeypatch): catalog = [_ref(f"v{i:03d}") for i in range(500)] known = {r.video_id for r in catalog[100:]} # 100 new uploads since last sync calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3) # 30 -> 60 -> 120: doubles only as far as needed, never the full 500. assert calls == [30, 60, 120] assert result.new_count == 100 assert result.caught_up is True def test_gives_up_at_max_window_and_says_so(monkeypatch): catalog = [_ref(f"v{i:03d}") for i in range(500)] calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental( "https://y/@alpha/videos", {"not-in-this-channel"}, window=10, max_window=40, overlap=3 ) assert calls == [10, 20, 40] assert result.caught_up is False, "caller must be able to tell the scan was truncated" assert result.fetched == 40 def test_no_local_history_walks_the_whole_channel(monkeypatch): catalog = [_ref(f"v{i:03d}") for i in range(120)] calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental("https://y/@alpha/videos", set(), window=30) assert calls == [None], "a first sync has no boundary to stop at" assert result.full_scan is True assert result.new_count == 120 def test_short_channel_is_exhausted_in_one_pass(monkeypatch): catalog = [_ref("a"), _ref("b")] calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental("https://y/@alpha/videos", {"b"}, window=30) assert calls == [30] assert result.exhausted is True assert result.caught_up is True assert [r.video_id for r in result.new_refs] == ["a"] def test_keep_filter_applies_before_the_overlap_check(monkeypatch): """Shorts the store never recorded must not keep the window widening.""" catalog = [ _ref("s1", url="https://www.youtube.com/shorts/s1"), _ref("s2", url="https://www.youtube.com/shorts/s2"), _ref("s3", url="https://www.youtube.com/shorts/s3"), _ref("known1"), _ref("known2"), _ref("known3"), ] + [_ref(f"v{i}") for i in range(50)] calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental( "https://y/@alpha/videos", {"known1", "known2", "known3"}, window=6, overlap=3, keep=lambda r: "/shorts/" not in (r.url or ""), ) assert calls == [6], "the three known long-form videos end the scan" assert result.new_refs == [] def test_upload_date_cutoff_stops_the_scan_when_dates_are_available(monkeypatch): catalog = [ _ref("n1", "20260701"), _ref("n2", "20260630"), _ref("o1", "20250101"), _ref("o2", "20241231"), ] + [_ref(f"old{i}", "20200101") for i in range(50)] calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental( "https://y/@alpha/videos", {"unrelated"}, window=4, overlap=2, since="20260601" ) assert calls == [4], "entries older than the watermark end the scan" assert result.caught_up is True @pytest.mark.parametrize("window,overlap", [(0, 0), (-5, -1)]) def test_degenerate_settings_are_clamped(monkeypatch, window, overlap): catalog = [_ref("a"), _ref("b")] calls: list = [] _fake_channel(monkeypatch, catalog, calls) result = discover_incremental("https://y/@alpha/videos", {"b"}, window=window, overlap=overlap) assert calls and calls[0] >= 1 assert result.fetched >= 1