158 lines
5.4 KiB
Python
158 lines
5.4 KiB
Python
"""Windowed channel sync: only fetch what is newer than what we already have.
|
|
|
|
The whole point is request economy against YouTube, so these tests assert on the
|
|
`limit` values handed to yt-dlp (which become `playlistend`, i.e. how many
|
|
continuation pages get requested), not just on the refs that come back.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from yt_scraper.discover import discover_incremental
|
|
from yt_scraper.store import VideoRef
|
|
|
|
|
|
def _ref(vid: str, upload_date: str | None = None, url: str | None = None) -> VideoRef:
|
|
return VideoRef(
|
|
video_id=vid,
|
|
channel_id="UC1",
|
|
title=vid,
|
|
url=url or f"https://www.youtube.com/watch?v={vid}",
|
|
upload_date=upload_date,
|
|
)
|
|
|
|
|
|
def _fake_channel(monkeypatch, catalog: list[VideoRef], calls: list | None = None):
|
|
"""Patch discover_channel with a channel whose /videos tab is `catalog`
|
|
(newest first) and that honours `limit` the way playlistend does."""
|
|
|
|
def fake(url, sleep_subrequests=2.0, limit=None):
|
|
if calls is not None:
|
|
calls.append(limit)
|
|
return ("UC1", "Alpha", "http://avatar", catalog[:limit] if limit else list(catalog))
|
|
|
|
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
|
|
|
|
|
|
def test_stops_at_first_run_of_known_videos(monkeypatch):
|
|
catalog = [_ref(f"v{i:03d}") for i in range(500)]
|
|
known = {r.video_id for r in catalog[5:]} # everything except the 5 newest
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
|
|
|
|
assert calls == [30], "one pass only — must not paginate the whole channel"
|
|
assert result.fetched == 30
|
|
assert [r.video_id for r in result.new_refs] == [f"v{i:03d}" for i in range(5)]
|
|
assert result.caught_up is True
|
|
assert result.full_scan is False
|
|
|
|
|
|
def test_widens_window_when_the_whole_window_is_new(monkeypatch):
|
|
catalog = [_ref(f"v{i:03d}") for i in range(500)]
|
|
known = {r.video_id for r in catalog[100:]} # 100 new uploads since last sync
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
|
|
|
|
# 30 -> 60 -> 120: doubles only as far as needed, never the full 500.
|
|
assert calls == [30, 60, 120]
|
|
assert result.new_count == 100
|
|
assert result.caught_up is True
|
|
|
|
|
|
def test_gives_up_at_max_window_and_says_so(monkeypatch):
|
|
catalog = [_ref(f"v{i:03d}") for i in range(500)]
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental(
|
|
"https://y/@alpha/videos", {"not-in-this-channel"}, window=10, max_window=40, overlap=3
|
|
)
|
|
|
|
assert calls == [10, 20, 40]
|
|
assert result.caught_up is False, "caller must be able to tell the scan was truncated"
|
|
assert result.fetched == 40
|
|
|
|
|
|
def test_no_local_history_walks_the_whole_channel(monkeypatch):
|
|
catalog = [_ref(f"v{i:03d}") for i in range(120)]
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental("https://y/@alpha/videos", set(), window=30)
|
|
|
|
assert calls == [None], "a first sync has no boundary to stop at"
|
|
assert result.full_scan is True
|
|
assert result.new_count == 120
|
|
|
|
|
|
def test_short_channel_is_exhausted_in_one_pass(monkeypatch):
|
|
catalog = [_ref("a"), _ref("b")]
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=30)
|
|
|
|
assert calls == [30]
|
|
assert result.exhausted is True
|
|
assert result.caught_up is True
|
|
assert [r.video_id for r in result.new_refs] == ["a"]
|
|
|
|
|
|
def test_keep_filter_applies_before_the_overlap_check(monkeypatch):
|
|
"""Shorts the store never recorded must not keep the window widening."""
|
|
catalog = [
|
|
_ref("s1", url="https://www.youtube.com/shorts/s1"),
|
|
_ref("s2", url="https://www.youtube.com/shorts/s2"),
|
|
_ref("s3", url="https://www.youtube.com/shorts/s3"),
|
|
_ref("known1"),
|
|
_ref("known2"),
|
|
_ref("known3"),
|
|
] + [_ref(f"v{i}") for i in range(50)]
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental(
|
|
"https://y/@alpha/videos",
|
|
{"known1", "known2", "known3"},
|
|
window=6,
|
|
overlap=3,
|
|
keep=lambda r: "/shorts/" not in (r.url or ""),
|
|
)
|
|
|
|
assert calls == [6], "the three known long-form videos end the scan"
|
|
assert result.new_refs == []
|
|
|
|
|
|
def test_upload_date_cutoff_stops_the_scan_when_dates_are_available(monkeypatch):
|
|
catalog = [
|
|
_ref("n1", "20260701"),
|
|
_ref("n2", "20260630"),
|
|
_ref("o1", "20250101"),
|
|
_ref("o2", "20241231"),
|
|
] + [_ref(f"old{i}", "20200101") for i in range(50)]
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental(
|
|
"https://y/@alpha/videos", {"unrelated"}, window=4, overlap=2, since="20260601"
|
|
)
|
|
|
|
assert calls == [4], "entries older than the watermark end the scan"
|
|
assert result.caught_up is True
|
|
|
|
|
|
@pytest.mark.parametrize("window,overlap", [(0, 0), (-5, -1)])
|
|
def test_degenerate_settings_are_clamped(monkeypatch, window, overlap):
|
|
catalog = [_ref("a"), _ref("b")]
|
|
calls: list = []
|
|
_fake_channel(monkeypatch, catalog, calls)
|
|
|
|
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=window, overlap=overlap)
|
|
|
|
assert calls and calls[0] >= 1
|
|
assert result.fetched >= 1
|