81 lines
2.9 KiB
Python
81 lines
2.9 KiB
Python
"""Store-side support for incremental sync: the known-id boundary and the
|
|
watermark that survives a re-discovery."""
|
|
from __future__ import annotations
|
|
|
|
from yt_scraper.store import Store, VideoRef
|
|
|
|
|
|
def _store(tmp_path) -> Store:
|
|
store = Store(tmp_path / "state.db")
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
|
return store
|
|
|
|
|
|
def test_known_video_ids_is_scoped_to_the_channel(tmp_path):
|
|
store = _store(tmp_path)
|
|
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
|
store.upsert_videos([
|
|
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
|
|
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
|
|
VideoRef("c", "UC2", "C", "https://y/watch?v=c"),
|
|
])
|
|
|
|
assert store.known_video_ids("UC1") == {"a", "b"}
|
|
assert store.known_video_ids("UC2") == {"c"}
|
|
assert store.known_video_ids("nope") == set()
|
|
|
|
|
|
def test_rediscovery_does_not_wipe_a_known_upload_date(tmp_path):
|
|
"""Flat discovery reports upload_date=None; without COALESCE a routine sync
|
|
would erase the dates learned during extraction — and with them the very
|
|
watermark this feature is built on."""
|
|
store = _store(tmp_path)
|
|
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260715", 120)])
|
|
|
|
store.upsert_videos([VideoRef("a", "UC1", "A (renamed)", "https://y/watch?v=a", None, None)])
|
|
|
|
row = store.get_video("a")
|
|
assert row.upload_date == "20260715"
|
|
assert row.duration == 120
|
|
assert row.title == "A (renamed)", "titles should still refresh"
|
|
|
|
|
|
def test_latest_upload_date_ignores_undated_rows(tmp_path):
|
|
store = _store(tmp_path)
|
|
store.upsert_videos([
|
|
VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260101"),
|
|
VideoRef("b", "UC1", "B", "https://y/watch?v=b", None),
|
|
VideoRef("c", "UC1", "C", "https://y/watch?v=c", "20260720"),
|
|
])
|
|
|
|
assert store.latest_upload_date("UC1") == "20260720"
|
|
|
|
|
|
def test_mark_channel_synced_recounts_instead_of_trusting_the_window(tmp_path):
|
|
"""An incremental pass only sees the newest slice, so video_count must come
|
|
from the DB — otherwise an 848-video channel shrinks to the window size."""
|
|
store = _store(tmp_path)
|
|
store.upsert_videos([
|
|
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", "20260101")
|
|
for i in range(40)
|
|
])
|
|
|
|
marks = store.mark_channel_synced("UC1")
|
|
|
|
assert marks["video_count"] == 40
|
|
channel = store.get_channel("UC1")
|
|
assert channel["video_count"] == 40
|
|
assert channel["last_video_date"] == "20260101"
|
|
assert channel["last_synced_at"]
|
|
|
|
|
|
def test_mark_channel_synced_keeps_the_last_date_when_nothing_is_dated(tmp_path):
|
|
store = _store(tmp_path)
|
|
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260101")])
|
|
store.mark_channel_synced("UC1")
|
|
|
|
store.upsert_videos([VideoRef("b", "UC1", "B", "https://y/watch?v=b", None)])
|
|
store.mark_channel_synced("UC1")
|
|
|
|
assert store.get_channel("UC1")["last_video_date"] == "20260101"
|