320 lines
13 KiB
Python
320 lines
13 KiB
Python
"""Chronological ordering of the videos list.
|
|
|
|
The list has to read like a YouTube channel page — newest upload first — and
|
|
that has to hold for videos nothing has scraped yet. Those have no
|
|
`upload_date` at all: yt-dlp's flat listing does not report one for YouTube
|
|
entries, so the date only arrives with the extraction that produces the .md.
|
|
`channel_seq` (the video's slot in the reverse-chronological /videos tab) is
|
|
what keeps them in place until then.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import sqlite3
|
|
|
|
import pytest
|
|
|
|
from yt_scraper.store import NO_DATE_SENTINEL, SCHEMA, SORT_DATE_SQL, Store, VideoRef
|
|
|
|
|
|
@pytest.fixture()
|
|
def store(tmp_path):
|
|
s = Store(tmp_path / "state.db")
|
|
s.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
|
return s
|
|
|
|
|
|
def _ref(video_id: str, upload_date: str | None = None, channel_id: str = "UC1") -> VideoRef:
|
|
return VideoRef(
|
|
video_id, channel_id, video_id.upper(),
|
|
f"https://y/watch?v={video_id}", upload_date,
|
|
)
|
|
|
|
|
|
def _ids(rows) -> list[str]:
|
|
return [r.video_id for r in rows]
|
|
|
|
|
|
# A channel as discovery hands it over: /videos-tab order, newest first, with
|
|
# two videos nobody has extracted yet ("a" and "c") interleaved among two that
|
|
# have been ("b" and "d").
|
|
MIXED = [_ref("a"), _ref("b", "20240301"), _ref("c"), _ref("d", "20240101")]
|
|
|
|
|
|
def test_default_order_is_newest_upload_first(store):
|
|
store.upsert_videos([_ref("mid", "20240201"), _ref("new", "20240301"), _ref("old", "20240101")])
|
|
|
|
rows, total = store.query_videos()
|
|
|
|
assert total == 3
|
|
assert _ids(rows) == ["new", "mid", "old"]
|
|
|
|
|
|
def test_videos_without_a_markdown_keep_their_slot_in_the_channel(store):
|
|
"""The regression this exists for: the fallback used to be `discovered_at`,
|
|
which dates every un-scraped video "today" and floats the whole backlog
|
|
above videos that really were uploaded this week."""
|
|
store.upsert_videos(MIXED)
|
|
|
|
rows, _ = store.query_videos()
|
|
|
|
assert _ids(rows) == ["a", "b", "c", "d"]
|
|
|
|
|
|
def test_an_undated_video_reports_the_date_it_was_sorted_by(store):
|
|
store.upsert_videos(MIXED)
|
|
|
|
by_id = {r.video_id: r for r in store.query_videos()[0]}
|
|
|
|
assert by_id["b"].sort_date == "20240301"
|
|
# "c" has no date of its own, so it borrows the nearest NEWER dated video's.
|
|
assert by_id["c"].upload_date is None
|
|
assert by_id["c"].sort_date == "20240301"
|
|
# Nothing dated sits above "a", so the most it can claim is the newest date
|
|
# known in its channel. Rank puts it above that video; inventing a later
|
|
# date would let an un-scraped channel outrank every scraped one.
|
|
assert by_id["a"].sort_date == "20240301"
|
|
|
|
|
|
def test_a_channel_with_nothing_scraped_does_not_outrank_channels_that_are(store):
|
|
"""With no dated video anywhere in the channel there is zero evidence about
|
|
when anything was published. Measured on the real library, one such channel
|
|
(212 videos, none extracted) took over the whole first page."""
|
|
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
|
store.upsert_videos([_ref("known", "20240101", channel_id="UC2")])
|
|
store.upsert_videos([_ref("u1"), _ref("u2")]) # UC1: nothing dated at all
|
|
|
|
rows, _ = store.query_videos()
|
|
|
|
assert _ids(rows) == ["known", "u1", "u2"]
|
|
# ...but the blind channel is still internally in listing order.
|
|
assert {r.sort_date for r in rows if r.channel_id == "UC1"} == {NO_DATE_SENTINEL}
|
|
|
|
|
|
def test_a_new_upload_ranks_above_the_backlog_it_did_not_refetch(store):
|
|
"""An incremental sync only fetches the newest window. Everything outside it
|
|
is older by construction, so the window is ranked above the stored rows
|
|
instead of restarting from zero and colliding with them."""
|
|
store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")]) # first full pass
|
|
store.upsert_videos([_ref("n2"), _ref("n1"), _ref("v3"), _ref("v2")]) # later window
|
|
|
|
rows, _ = store.query_videos()
|
|
|
|
assert _ids(rows) == ["n2", "n1", "v3", "v2", "v1"]
|
|
|
|
|
|
def test_repeated_syncs_do_not_reshuffle_a_channel(store):
|
|
store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")])
|
|
first = _ids(store.query_videos()[0])
|
|
|
|
for _ in range(3):
|
|
store.upsert_videos([_ref("v3"), _ref("v2")]) # same window, nothing new
|
|
|
|
assert _ids(store.query_videos()[0]) == first
|
|
|
|
|
|
@pytest.mark.parametrize("sort", ["view_count", "like_count", "duration", "", "nonsense"])
|
|
def test_sort_keys_the_ui_sends_do_not_degrade_the_chronology(store, sort):
|
|
"""The web UI sends `view_count`/`like_count`/`duration`; the store only
|
|
knew `views_desc`/`duration_desc`. Every miss fell through to a raw
|
|
`upload_date DESC` that ignored the inference and dumped undated videos at
|
|
the bottom, whatever the channel order said."""
|
|
store.upsert_videos(MIXED)
|
|
|
|
rows, _ = store.query_videos(sort=sort)
|
|
|
|
assert _ids(rows) == ["a", "b", "c", "d"]
|
|
|
|
|
|
def test_sorting_by_views_ranks_counted_videos_and_keeps_the_rest_chronological(store):
|
|
store.upsert_videos(MIXED)
|
|
store.update_video_metadata("d", view_count=500)
|
|
store.update_video_metadata("b", view_count=10)
|
|
|
|
rows, _ = store.query_videos(sort="view_count")
|
|
|
|
assert _ids(rows) == ["d", "b", "a", "c"]
|
|
|
|
|
|
def test_oldest_first_reverses_the_default(store):
|
|
store.upsert_videos(MIXED)
|
|
|
|
rows, _ = store.query_videos(sort="oldest")
|
|
|
|
assert _ids(rows) == ["d", "c", "b", "a"]
|
|
|
|
|
|
def test_oldest_first_keeps_unknown_dates_at_the_end_not_the_start(store):
|
|
""""We do not know when this is from" is not "the beginning of time".
|
|
|
|
NO_DATE_SENTINEL sorts below every real date, which is what keeps those rows
|
|
off the front page under the default sort — and is exactly why they came
|
|
FIRST under ascending. Measured: "oldest" opened with the 212 videos of a
|
|
channel nothing had extracted, ahead of a genuine 2017 upload.
|
|
"""
|
|
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
|
store.upsert_videos([_ref("ancient", "20170414", channel_id="UC2")])
|
|
store.upsert_videos([_ref("unknown1"), _ref("unknown2")]) # UC1: no dates at all
|
|
|
|
rows, _ = store.query_videos(sort="oldest")
|
|
|
|
assert _ids(rows)[0] == "ancient"
|
|
assert set(_ids(rows)[1:]) == {"unknown1", "unknown2"}
|
|
|
|
|
|
def test_a_tie_does_not_rank_channels_by_how_big_their_catalogue_is(store):
|
|
"""`channel_seq` counts up to a channel's video count, so comparing it
|
|
ACROSS channels ranks by catalogue size. With 92% of adjacent pairs tying on
|
|
sort_date, that decided most of the list: an entire channel preceded another
|
|
purely because 575 > 476. Ties group by channel instead, so no channel's
|
|
internal order is ever interleaved away."""
|
|
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
|
# Same date everywhere: the tiebreak decides the whole ordering.
|
|
store.upsert_videos([_ref(f"big{i}", "20240101") for i in range(5)])
|
|
store.upsert_videos([_ref(f"small{i}", "20240101", channel_id="UC2") for i in range(2)])
|
|
|
|
ids = _ids(store.query_videos()[0])
|
|
|
|
big = [i for i, v in enumerate(ids) if v.startswith("big")]
|
|
small = [i for i, v in enumerate(ids) if v.startswith("small")]
|
|
assert max(big) < min(small) or max(small) < min(big), f"channels interleaved: {ids}"
|
|
# and each channel is still in its own listing order
|
|
assert [v for v in ids if v.startswith("big")] == ["big0", "big1", "big2", "big3", "big4"]
|
|
assert [v for v in ids if v.startswith("small")] == ["small0", "small1"]
|
|
|
|
|
|
def test_a_row_that_arrives_without_a_rank_is_ranked_on_the_next_open(tmp_path):
|
|
"""A server still running the pre-column code writes rows with a NULL rank.
|
|
Left NULL they do not just lose their place — SORT_DATE_SQL makes them
|
|
inherit the OLDEST date in the channel, so a video discovery has only just
|
|
found is shown last. Measured on the live database: two brand-new uploads
|
|
displayed with sort_date 20180720, second to last of 513."""
|
|
db = tmp_path / "state.db"
|
|
store = Store(db)
|
|
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
|
store.upsert_videos([_ref("older", "20180720"), _ref("oldest", "20180101")])
|
|
|
|
with sqlite3.connect(db) as conn: # what the old code path produced
|
|
conn.execute(
|
|
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
|
|
" VALUES ('brand_new', 'UC1', 'Brand new', 'https://y/watch?v=brand_new',"
|
|
" 'pending', '2026-08-08T13:25:32+00:00')"
|
|
)
|
|
|
|
reopened = Store(db) # migration runs here
|
|
|
|
rows, _ = reopened.query_videos()
|
|
assert _ids(rows) == ["brand_new", "older", "oldest"]
|
|
assert rows[0].channel_seq is not None
|
|
assert rows[0].sort_date == "20180720" # newest known in the channel, not the oldest
|
|
|
|
|
|
def test_paging_neither_repeats_nor_loses_rows(store):
|
|
store.upsert_videos([_ref(f"v{i}") for i in range(7)])
|
|
|
|
full = _ids(store.query_videos(size=50)[0])
|
|
paged = [vid for page in (1, 2, 3) for vid in _ids(store.query_videos(page=page, size=3)[0])]
|
|
|
|
assert full == ["v0", "v1", "v2", "v3", "v4", "v5", "v6"]
|
|
assert paged == full
|
|
|
|
|
|
def test_filters_do_not_disturb_the_order(store):
|
|
store.upsert_videos(MIXED)
|
|
store.mark_status("a", "no_subtitles", "no captions")
|
|
|
|
rows, total = store.query_videos(channel_id="UC1", status="pending")
|
|
|
|
assert total == 3
|
|
assert _ids(rows) == ["b", "c", "d"]
|
|
|
|
|
|
def test_a_real_date_outranks_a_stale_listing_position(store):
|
|
"""Why the date is the primary key and the rank only the tiebreak.
|
|
|
|
A rank that predates the column was reconstructed offline, not observed from
|
|
YouTube, so it can be wrong. Sorting by date first means the dates we did
|
|
pay an extraction to learn CORRECT that reconstruction instead of being
|
|
overridden by it. Measured against the live listing on Código Espinoza:
|
|
ordering by rank alone was strictly worse than date-then-rank.
|
|
"""
|
|
# The rank claims "old" is the newer of the two; its upload_date says otherwise.
|
|
store.upsert_videos([_ref("old", "20240101"), _ref("new", "20260801")])
|
|
|
|
assert _ids(store.query_videos()[0]) == ["new", "old"]
|
|
|
|
|
|
def test_a_discovery_pass_repairs_a_rank_the_seed_got_wrong(store):
|
|
"""Two videos published the same day: the date cannot separate them, so the
|
|
listing rank decides — and a wrong rank shows a wrong order. One ordinary
|
|
sync re-observes the window from YouTube and fixes it. Measured: Código
|
|
Espinoza went from 9 inverted pairs to 0 after a single 30-entry pass."""
|
|
store.upsert_videos([_ref("a", "20260801"), _ref("b", "20260801")])
|
|
assert _ids(store.query_videos()[0]) == ["a", "b"]
|
|
|
|
store.upsert_videos([_ref("b", "20260801"), _ref("a", "20260801")]) # real order
|
|
|
|
assert _ids(store.query_videos()[0]) == ["b", "a"]
|
|
|
|
|
|
def test_the_ordering_lookups_stay_on_their_partial_indexes(store):
|
|
"""Both subqueries run once per undated row, so the plan is the difference
|
|
between a usable page and an unusable one: measured on the real library,
|
|
576 ms per page without these indexes and 22 ms with them. SQLite drops a
|
|
partial index the moment the query's WHERE stops implying the index's, and
|
|
says nothing about it."""
|
|
store.upsert_videos(MIXED)
|
|
sql = (
|
|
f"SELECT videos.*, {SORT_DATE_SQL} AS sort_date FROM videos "
|
|
"ORDER BY sort_date DESC, videos.channel_seq DESC, videos.video_id DESC LIMIT 25"
|
|
)
|
|
|
|
with store._cursor() as cur:
|
|
plan = [str(row["detail"]) for row in cur.execute("EXPLAIN QUERY PLAN " + sql)]
|
|
|
|
assert any("COVERING INDEX idx_videos_dated_seq" in line for line in plan), plan
|
|
assert any("COVERING INDEX idx_videos_dated" in line
|
|
and "idx_videos_dated_seq" not in line for line in plan), plan
|
|
|
|
|
|
def _legacy_db(path):
|
|
"""A database written before `channel_seq` existed."""
|
|
conn = sqlite3.connect(path)
|
|
conn.executescript(SCHEMA)
|
|
conn.execute("INSERT INTO channels (channel_id, name) VALUES ('UC1', 'Alpha')")
|
|
# One discovery batch shares a timestamp and is inserted newest-first...
|
|
for vid in ("b1", "b2", "b3"):
|
|
conn.execute(
|
|
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
|
|
" VALUES (?, 'UC1', ?, ?, 'pending', '2026-01-01T00:00:00+00:00')",
|
|
(vid, vid, f"https://y/watch?v={vid}"),
|
|
)
|
|
# ...and a later sync can only add videos newer than every one of them.
|
|
conn.execute(
|
|
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
|
|
" VALUES ('later', 'UC1', 'later', 'https://y/watch?v=later', 'pending',"
|
|
" '2026-02-01T00:00:00+00:00')"
|
|
)
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
|
|
def test_migration_ranks_a_database_that_predates_the_column(tmp_path):
|
|
db = tmp_path / "legacy.db"
|
|
_legacy_db(db)
|
|
|
|
rows, _ = Store(db).query_videos()
|
|
|
|
assert _ids(rows) == ["later", "b1", "b2", "b3"]
|
|
assert all(r.channel_seq is not None for r in rows)
|
|
|
|
|
|
def test_migration_seeds_once_and_does_not_reshuffle_on_reopen(tmp_path):
|
|
db = tmp_path / "legacy.db"
|
|
_legacy_db(db)
|
|
first = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]}
|
|
|
|
again = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]}
|
|
|
|
assert again == first
|