wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)

This commit is contained in:
urieljareth
2026-08-22 19:37:23 -06:00
parent 4f5a68b572
commit 8a59b39c98
103 changed files with 70954 additions and 1825 deletions
+157
View File
@@ -0,0 +1,157 @@
"""Windowed channel sync: only fetch what is newer than what we already have.
The whole point is request economy against YouTube, so these tests assert on the
`limit` values handed to yt-dlp (which become `playlistend`, i.e. how many
continuation pages get requested), not just on the refs that come back.
"""
from __future__ import annotations
import pytest
from yt_scraper.discover import discover_incremental
from yt_scraper.store import VideoRef
def _ref(vid: str, upload_date: str | None = None, url: str | None = None) -> VideoRef:
return VideoRef(
video_id=vid,
channel_id="UC1",
title=vid,
url=url or f"https://www.youtube.com/watch?v={vid}",
upload_date=upload_date,
)
def _fake_channel(monkeypatch, catalog: list[VideoRef], calls: list | None = None):
"""Patch discover_channel with a channel whose /videos tab is `catalog`
(newest first) and that honours `limit` the way playlistend does."""
def fake(url, sleep_subrequests=2.0, limit=None):
if calls is not None:
calls.append(limit)
return ("UC1", "Alpha", "http://avatar", catalog[:limit] if limit else list(catalog))
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
def test_stops_at_first_run_of_known_videos(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(500)]
known = {r.video_id for r in catalog[5:]} # everything except the 5 newest
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
assert calls == [30], "one pass only — must not paginate the whole channel"
assert result.fetched == 30
assert [r.video_id for r in result.new_refs] == [f"v{i:03d}" for i in range(5)]
assert result.caught_up is True
assert result.full_scan is False
def test_widens_window_when_the_whole_window_is_new(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(500)]
known = {r.video_id for r in catalog[100:]} # 100 new uploads since last sync
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
# 30 -> 60 -> 120: doubles only as far as needed, never the full 500.
assert calls == [30, 60, 120]
assert result.new_count == 100
assert result.caught_up is True
def test_gives_up_at_max_window_and_says_so(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(500)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental(
"https://y/@alpha/videos", {"not-in-this-channel"}, window=10, max_window=40, overlap=3
)
assert calls == [10, 20, 40]
assert result.caught_up is False, "caller must be able to tell the scan was truncated"
assert result.fetched == 40
def test_no_local_history_walks_the_whole_channel(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(120)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", set(), window=30)
assert calls == [None], "a first sync has no boundary to stop at"
assert result.full_scan is True
assert result.new_count == 120
def test_short_channel_is_exhausted_in_one_pass(monkeypatch):
catalog = [_ref("a"), _ref("b")]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=30)
assert calls == [30]
assert result.exhausted is True
assert result.caught_up is True
assert [r.video_id for r in result.new_refs] == ["a"]
def test_keep_filter_applies_before_the_overlap_check(monkeypatch):
"""Shorts the store never recorded must not keep the window widening."""
catalog = [
_ref("s1", url="https://www.youtube.com/shorts/s1"),
_ref("s2", url="https://www.youtube.com/shorts/s2"),
_ref("s3", url="https://www.youtube.com/shorts/s3"),
_ref("known1"),
_ref("known2"),
_ref("known3"),
] + [_ref(f"v{i}") for i in range(50)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental(
"https://y/@alpha/videos",
{"known1", "known2", "known3"},
window=6,
overlap=3,
keep=lambda r: "/shorts/" not in (r.url or ""),
)
assert calls == [6], "the three known long-form videos end the scan"
assert result.new_refs == []
def test_upload_date_cutoff_stops_the_scan_when_dates_are_available(monkeypatch):
catalog = [
_ref("n1", "20260701"),
_ref("n2", "20260630"),
_ref("o1", "20250101"),
_ref("o2", "20241231"),
] + [_ref(f"old{i}", "20200101") for i in range(50)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental(
"https://y/@alpha/videos", {"unrelated"}, window=4, overlap=2, since="20260601"
)
assert calls == [4], "entries older than the watermark end the scan"
assert result.caught_up is True
@pytest.mark.parametrize("window,overlap", [(0, 0), (-5, -1)])
def test_degenerate_settings_are_clamped(monkeypatch, window, overlap):
catalog = [_ref("a"), _ref("b")]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=window, overlap=overlap)
assert calls and calls[0] >= 1
assert result.fetched >= 1