"""Chronological ordering of the videos list. The list has to read like a YouTube channel page — newest upload first — and that has to hold for videos nothing has scraped yet. Those have no `upload_date` at all: yt-dlp's flat listing does not report one for YouTube entries, so the date only arrives with the extraction that produces the .md. `channel_seq` (the video's slot in the reverse-chronological /videos tab) is what keeps them in place until then. """ from __future__ import annotations import sqlite3 import pytest from yt_scraper.store import NO_DATE_SENTINEL, SCHEMA, SORT_DATE_SQL, Store, VideoRef @pytest.fixture() def store(tmp_path): s = Store(tmp_path / "state.db") s.upsert_channel("UC1", "@alpha", "Alpha", 0) return s def _ref(video_id: str, upload_date: str | None = None, channel_id: str = "UC1") -> VideoRef: return VideoRef( video_id, channel_id, video_id.upper(), f"https://y/watch?v={video_id}", upload_date, ) def _ids(rows) -> list[str]: return [r.video_id for r in rows] # A channel as discovery hands it over: /videos-tab order, newest first, with # two videos nobody has extracted yet ("a" and "c") interleaved among two that # have been ("b" and "d"). MIXED = [_ref("a"), _ref("b", "20240301"), _ref("c"), _ref("d", "20240101")] def test_default_order_is_newest_upload_first(store): store.upsert_videos([_ref("mid", "20240201"), _ref("new", "20240301"), _ref("old", "20240101")]) rows, total = store.query_videos() assert total == 3 assert _ids(rows) == ["new", "mid", "old"] def test_videos_without_a_markdown_keep_their_slot_in_the_channel(store): """The regression this exists for: the fallback used to be `discovered_at`, which dates every un-scraped video "today" and floats the whole backlog above videos that really were uploaded this week.""" store.upsert_videos(MIXED) rows, _ = store.query_videos() assert _ids(rows) == ["a", "b", "c", "d"] def test_an_undated_video_reports_the_date_it_was_sorted_by(store): store.upsert_videos(MIXED) by_id = {r.video_id: r for r in store.query_videos()[0]} assert by_id["b"].sort_date == "20240301" # "c" has no date of its own, so it borrows the nearest NEWER dated video's. assert by_id["c"].upload_date is None assert by_id["c"].sort_date == "20240301" # Nothing dated sits above "a", so the most it can claim is the newest date # known in its channel. Rank puts it above that video; inventing a later # date would let an un-scraped channel outrank every scraped one. assert by_id["a"].sort_date == "20240301" def test_a_channel_with_nothing_scraped_does_not_outrank_channels_that_are(store): """With no dated video anywhere in the channel there is zero evidence about when anything was published. Measured on the real library, one such channel (212 videos, none extracted) took over the whole first page.""" store.upsert_channel("UC2", "@beta", "Beta", 0) store.upsert_videos([_ref("known", "20240101", channel_id="UC2")]) store.upsert_videos([_ref("u1"), _ref("u2")]) # UC1: nothing dated at all rows, _ = store.query_videos() assert _ids(rows) == ["known", "u1", "u2"] # ...but the blind channel is still internally in listing order. assert {r.sort_date for r in rows if r.channel_id == "UC1"} == {NO_DATE_SENTINEL} def test_a_new_upload_ranks_above_the_backlog_it_did_not_refetch(store): """An incremental sync only fetches the newest window. Everything outside it is older by construction, so the window is ranked above the stored rows instead of restarting from zero and colliding with them.""" store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")]) # first full pass store.upsert_videos([_ref("n2"), _ref("n1"), _ref("v3"), _ref("v2")]) # later window rows, _ = store.query_videos() assert _ids(rows) == ["n2", "n1", "v3", "v2", "v1"] def test_repeated_syncs_do_not_reshuffle_a_channel(store): store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")]) first = _ids(store.query_videos()[0]) for _ in range(3): store.upsert_videos([_ref("v3"), _ref("v2")]) # same window, nothing new assert _ids(store.query_videos()[0]) == first @pytest.mark.parametrize("sort", ["view_count", "like_count", "duration", "", "nonsense"]) def test_sort_keys_the_ui_sends_do_not_degrade_the_chronology(store, sort): """The web UI sends `view_count`/`like_count`/`duration`; the store only knew `views_desc`/`duration_desc`. Every miss fell through to a raw `upload_date DESC` that ignored the inference and dumped undated videos at the bottom, whatever the channel order said.""" store.upsert_videos(MIXED) rows, _ = store.query_videos(sort=sort) assert _ids(rows) == ["a", "b", "c", "d"] def test_sorting_by_views_ranks_counted_videos_and_keeps_the_rest_chronological(store): store.upsert_videos(MIXED) store.update_video_metadata("d", view_count=500) store.update_video_metadata("b", view_count=10) rows, _ = store.query_videos(sort="view_count") assert _ids(rows) == ["d", "b", "a", "c"] def test_oldest_first_reverses_the_default(store): store.upsert_videos(MIXED) rows, _ = store.query_videos(sort="oldest") assert _ids(rows) == ["d", "c", "b", "a"] def test_oldest_first_keeps_unknown_dates_at_the_end_not_the_start(store): """"We do not know when this is from" is not "the beginning of time". NO_DATE_SENTINEL sorts below every real date, which is what keeps those rows off the front page under the default sort — and is exactly why they came FIRST under ascending. Measured: "oldest" opened with the 212 videos of a channel nothing had extracted, ahead of a genuine 2017 upload. """ store.upsert_channel("UC2", "@beta", "Beta", 0) store.upsert_videos([_ref("ancient", "20170414", channel_id="UC2")]) store.upsert_videos([_ref("unknown1"), _ref("unknown2")]) # UC1: no dates at all rows, _ = store.query_videos(sort="oldest") assert _ids(rows)[0] == "ancient" assert set(_ids(rows)[1:]) == {"unknown1", "unknown2"} def test_a_tie_does_not_rank_channels_by_how_big_their_catalogue_is(store): """`channel_seq` counts up to a channel's video count, so comparing it ACROSS channels ranks by catalogue size. With 92% of adjacent pairs tying on sort_date, that decided most of the list: an entire channel preceded another purely because 575 > 476. Ties group by channel instead, so no channel's internal order is ever interleaved away.""" store.upsert_channel("UC2", "@beta", "Beta", 0) # Same date everywhere: the tiebreak decides the whole ordering. store.upsert_videos([_ref(f"big{i}", "20240101") for i in range(5)]) store.upsert_videos([_ref(f"small{i}", "20240101", channel_id="UC2") for i in range(2)]) ids = _ids(store.query_videos()[0]) big = [i for i, v in enumerate(ids) if v.startswith("big")] small = [i for i, v in enumerate(ids) if v.startswith("small")] assert max(big) < min(small) or max(small) < min(big), f"channels interleaved: {ids}" # and each channel is still in its own listing order assert [v for v in ids if v.startswith("big")] == ["big0", "big1", "big2", "big3", "big4"] assert [v for v in ids if v.startswith("small")] == ["small0", "small1"] def test_a_row_that_arrives_without_a_rank_is_ranked_on_the_next_open(tmp_path): """A server still running the pre-column code writes rows with a NULL rank. Left NULL they do not just lose their place — SORT_DATE_SQL makes them inherit the OLDEST date in the channel, so a video discovery has only just found is shown last. Measured on the live database: two brand-new uploads displayed with sort_date 20180720, second to last of 513.""" db = tmp_path / "state.db" store = Store(db) store.upsert_channel("UC1", "@alpha", "Alpha", 0) store.upsert_videos([_ref("older", "20180720"), _ref("oldest", "20180101")]) with sqlite3.connect(db) as conn: # what the old code path produced conn.execute( "INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)" " VALUES ('brand_new', 'UC1', 'Brand new', 'https://y/watch?v=brand_new'," " 'pending', '2026-08-08T13:25:32+00:00')" ) reopened = Store(db) # migration runs here rows, _ = reopened.query_videos() assert _ids(rows) == ["brand_new", "older", "oldest"] assert rows[0].channel_seq is not None assert rows[0].sort_date == "20180720" # newest known in the channel, not the oldest def test_paging_neither_repeats_nor_loses_rows(store): store.upsert_videos([_ref(f"v{i}") for i in range(7)]) full = _ids(store.query_videos(size=50)[0]) paged = [vid for page in (1, 2, 3) for vid in _ids(store.query_videos(page=page, size=3)[0])] assert full == ["v0", "v1", "v2", "v3", "v4", "v5", "v6"] assert paged == full def test_filters_do_not_disturb_the_order(store): store.upsert_videos(MIXED) store.mark_status("a", "no_subtitles", "no captions") rows, total = store.query_videos(channel_id="UC1", status="pending") assert total == 3 assert _ids(rows) == ["b", "c", "d"] def test_a_real_date_outranks_a_stale_listing_position(store): """Why the date is the primary key and the rank only the tiebreak. A rank that predates the column was reconstructed offline, not observed from YouTube, so it can be wrong. Sorting by date first means the dates we did pay an extraction to learn CORRECT that reconstruction instead of being overridden by it. Measured against the live listing on Código Espinoza: ordering by rank alone was strictly worse than date-then-rank. """ # The rank claims "old" is the newer of the two; its upload_date says otherwise. store.upsert_videos([_ref("old", "20240101"), _ref("new", "20260801")]) assert _ids(store.query_videos()[0]) == ["new", "old"] def test_a_discovery_pass_repairs_a_rank_the_seed_got_wrong(store): """Two videos published the same day: the date cannot separate them, so the listing rank decides — and a wrong rank shows a wrong order. One ordinary sync re-observes the window from YouTube and fixes it. Measured: Código Espinoza went from 9 inverted pairs to 0 after a single 30-entry pass.""" store.upsert_videos([_ref("a", "20260801"), _ref("b", "20260801")]) assert _ids(store.query_videos()[0]) == ["a", "b"] store.upsert_videos([_ref("b", "20260801"), _ref("a", "20260801")]) # real order assert _ids(store.query_videos()[0]) == ["b", "a"] def test_the_ordering_lookups_stay_on_their_partial_indexes(store): """Both subqueries run once per undated row, so the plan is the difference between a usable page and an unusable one: measured on the real library, 576 ms per page without these indexes and 22 ms with them. SQLite drops a partial index the moment the query's WHERE stops implying the index's, and says nothing about it.""" store.upsert_videos(MIXED) sql = ( f"SELECT videos.*, {SORT_DATE_SQL} AS sort_date FROM videos " "ORDER BY sort_date DESC, videos.channel_seq DESC, videos.video_id DESC LIMIT 25" ) with store._cursor() as cur: plan = [str(row["detail"]) for row in cur.execute("EXPLAIN QUERY PLAN " + sql)] assert any("COVERING INDEX idx_videos_dated_seq" in line for line in plan), plan assert any("COVERING INDEX idx_videos_dated" in line and "idx_videos_dated_seq" not in line for line in plan), plan def _legacy_db(path): """A database written before `channel_seq` existed.""" conn = sqlite3.connect(path) conn.executescript(SCHEMA) conn.execute("INSERT INTO channels (channel_id, name) VALUES ('UC1', 'Alpha')") # One discovery batch shares a timestamp and is inserted newest-first... for vid in ("b1", "b2", "b3"): conn.execute( "INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)" " VALUES (?, 'UC1', ?, ?, 'pending', '2026-01-01T00:00:00+00:00')", (vid, vid, f"https://y/watch?v={vid}"), ) # ...and a later sync can only add videos newer than every one of them. conn.execute( "INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)" " VALUES ('later', 'UC1', 'later', 'https://y/watch?v=later', 'pending'," " '2026-02-01T00:00:00+00:00')" ) conn.commit() conn.close() def test_migration_ranks_a_database_that_predates_the_column(tmp_path): db = tmp_path / "legacy.db" _legacy_db(db) rows, _ = Store(db).query_videos() assert _ids(rows) == ["later", "b1", "b2", "b3"] assert all(r.channel_seq is not None for r in rows) def test_migration_seeds_once_and_does_not_reshuffle_on_reopen(tmp_path): db = tmp_path / "legacy.db" _legacy_db(db) first = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]} again = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]} assert again == first