wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)
This commit is contained in:
@@ -0,0 +1,143 @@
|
||||
"""Members-only / gated videos: identified from discovery, labelled, and kept
|
||||
out of bulk work without becoming unreachable.
|
||||
|
||||
yt-dlp's flat listing reports availability="subscriber_only" for members-only
|
||||
videos, so they are knowable before an extraction attempt is ever spent on them.
|
||||
Measured on a real channel: 120 flat entries in one request, exactly 4 flagged,
|
||||
matching exactly the 4 the DB had learned about the expensive way.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.store import BLOCKING_AVAILABILITY, Store, VideoRef
|
||||
|
||||
|
||||
def _store(tmp_path) -> Store:
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
||||
return store
|
||||
|
||||
|
||||
def _ref(vid: str, availability: str | None = None) -> VideoRef:
|
||||
return VideoRef(vid, "UC1", vid, f"https://www.youtube.com/watch?v={vid}", availability=availability)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- classification
|
||||
|
||||
def test_availability_from_discovery_is_stored_and_classified(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("gated", "subscriber_only"), _ref("open", "public")])
|
||||
|
||||
assert store.get_video("gated").availability == "subscriber_only"
|
||||
assert store.get_video("gated").block_reason == "members_only"
|
||||
assert store.get_video("open").block_reason is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("availability,expected", [
|
||||
("subscriber_only", "members_only"),
|
||||
("premium_only", "premium_only"),
|
||||
("private", "private"),
|
||||
("needs_auth", "needs_auth"),
|
||||
])
|
||||
def test_every_blocking_availability_maps_to_a_reason(tmp_path, availability, expected):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("v", availability)])
|
||||
assert store.get_video("v").block_reason == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("availability", ["public", "unlisted", None])
|
||||
def test_fetchable_availability_is_never_blocked(tmp_path, availability):
|
||||
"""`unlisted` downloads perfectly well — mislabelling it would hide videos
|
||||
the user can actually have."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("v", availability)])
|
||||
assert store.get_video("v").block_reason is None
|
||||
assert "unlisted" not in BLOCKING_AVAILABILITY
|
||||
|
||||
|
||||
def test_block_reason_falls_back_to_the_recorded_error(tmp_path):
|
||||
"""Rows burned into `error` before availability was captured must still be
|
||||
identifiable without re-fetching them."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("old")])
|
||||
store.mark_error("old", "ERROR: [youtube] x: Join this channel to get access to members-only content")
|
||||
|
||||
assert store.get_video("old").availability is None
|
||||
assert store.get_video("old").block_reason == "members_only"
|
||||
|
||||
|
||||
def test_rate_limit_error_is_not_a_block_reason(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("t")])
|
||||
store.mark_error("t", "ERROR: Video unavailable. The current session has been rate-limited by YouTube")
|
||||
|
||||
assert store.get_video("t").block_reason is None
|
||||
|
||||
|
||||
def test_rediscovery_does_not_wipe_a_known_availability(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("v", "subscriber_only")])
|
||||
store.upsert_videos([_ref("v", None)]) # a later flat pass omitted the field
|
||||
|
||||
assert store.get_video("v").availability == "subscriber_only"
|
||||
|
||||
|
||||
def test_set_availability_records_what_extraction_learned(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("v")])
|
||||
store.set_availability("v", "subscriber_only")
|
||||
assert store.get_video("v").block_reason == "members_only"
|
||||
|
||||
store.set_availability("v", None) # must not clear it
|
||||
assert store.get_video("v").availability == "subscriber_only"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- queue behaviour
|
||||
|
||||
def test_blocked_videos_are_kept_out_of_the_pending_queue(tmp_path):
|
||||
"""Bulk runs must not spend requests on videos that cannot be fetched."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("gated", "subscriber_only"), _ref("ok"), _ref("unlisted", "unlisted")])
|
||||
|
||||
ids = {v.video_id for v in store.get_pending("UC1")}
|
||||
assert ids == {"ok", "unlisted"}
|
||||
|
||||
all_ids = {v.video_id for v in store.get_pending("UC1", include_blocked=True)}
|
||||
assert all_ids == {"ok", "unlisted", "gated"}
|
||||
|
||||
|
||||
def test_a_blocked_video_is_still_reachable_by_id(tmp_path):
|
||||
"""Buying the membership must not leave the video permanently stranded."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("gated", "subscriber_only")])
|
||||
|
||||
assert store.get_video("gated") is not None, "explicit per-video processing still works"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- filtering
|
||||
|
||||
def test_blocked_filter_finds_them_across_statuses(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
_ref("gated", "subscriber_only"),
|
||||
_ref("legacy"),
|
||||
_ref("fine"),
|
||||
])
|
||||
store.mark_error("legacy", "ERROR: Join this channel to get access to members-only content")
|
||||
store.mark_status("fine", "no_subtitles")
|
||||
|
||||
rows, total = store.query_videos(blocked=True)
|
||||
assert {r.video_id for r in rows} == {"gated", "legacy"}
|
||||
assert total == 2
|
||||
|
||||
rows, total = store.query_videos(blocked=False)
|
||||
assert {r.video_id for r in rows} == {"fine"}
|
||||
|
||||
|
||||
def test_blocked_filter_absent_means_everything(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([_ref("gated", "subscriber_only"), _ref("fine")])
|
||||
|
||||
_rows, total = store.query_videos()
|
||||
assert total == 2
|
||||
@@ -0,0 +1,75 @@
|
||||
"""Tests for the YAML config loader.
|
||||
|
||||
Focuses on the ``languages`` and ``prefer_manual`` fields, including the
|
||||
legacy/compat behaviour. Network-free, DB-free.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import textwrap
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.config import load_config, parse_languages
|
||||
|
||||
|
||||
def _write_config(tmp_path, body: str) -> str:
|
||||
p = tmp_path / "config.yaml"
|
||||
p.write_text(textwrap.dedent(body), encoding="utf-8")
|
||||
return str(p)
|
||||
|
||||
|
||||
def test_load_legacy_languages_list(tmp_path) -> None:
|
||||
p = _write_config(tmp_path, """
|
||||
channel_url: "https://example.com/@x/videos"
|
||||
languages: ["es", "en"]
|
||||
prefer_manual: true
|
||||
""")
|
||||
cfg = load_config(p)
|
||||
assert cfg.languages == {"es": "manual", "en": "manual"}
|
||||
|
||||
|
||||
def test_load_legacy_languages_list_with_prefer_manual_false(tmp_path) -> None:
|
||||
p = _write_config(tmp_path, """
|
||||
languages: ["es", "en"]
|
||||
prefer_manual: false
|
||||
""")
|
||||
cfg = load_config(p)
|
||||
assert cfg.languages == {"es": "auto", "en": "auto"}
|
||||
|
||||
|
||||
def test_load_new_dict_languages(tmp_path) -> None:
|
||||
p = _write_config(tmp_path, """
|
||||
languages:
|
||||
en: manual
|
||||
es: auto
|
||||
pt: any
|
||||
prefer_manual: false
|
||||
""")
|
||||
cfg = load_config(p)
|
||||
assert cfg.languages == {"en": "manual", "es": "auto", "pt": "any"}
|
||||
# prefer_manual remains accessible for fallback on `"any"` entries
|
||||
assert cfg.prefer_manual is False
|
||||
|
||||
|
||||
def test_load_unknown_mode_normalises_to_any(tmp_path) -> None:
|
||||
p = _write_config(tmp_path, """
|
||||
languages:
|
||||
en: garbage
|
||||
""")
|
||||
cfg = load_config(p)
|
||||
assert cfg.languages == {"en": "any"}
|
||||
|
||||
|
||||
def test_load_missing_languages_defaults_to_any_not_manual_only(tmp_path) -> None:
|
||||
"""The default must fall back to auto captions.
|
||||
|
||||
A manual-only default silently produces zero transcripts on the many
|
||||
channels that publish only auto-generated captions, and records them as
|
||||
`no_subtitles` — a terminal status that hides a purely configural failure.
|
||||
"""
|
||||
p = _write_config(tmp_path, """
|
||||
channel_url: "https://example.com/@x/videos"
|
||||
""")
|
||||
cfg = load_config(p)
|
||||
assert "es" in cfg.languages and "en" in cfg.languages
|
||||
assert cfg.languages["es"] == "any"
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Windowed channel sync: only fetch what is newer than what we already have.
|
||||
|
||||
The whole point is request economy against YouTube, so these tests assert on the
|
||||
`limit` values handed to yt-dlp (which become `playlistend`, i.e. how many
|
||||
continuation pages get requested), not just on the refs that come back.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.discover import discover_incremental
|
||||
from yt_scraper.store import VideoRef
|
||||
|
||||
|
||||
def _ref(vid: str, upload_date: str | None = None, url: str | None = None) -> VideoRef:
|
||||
return VideoRef(
|
||||
video_id=vid,
|
||||
channel_id="UC1",
|
||||
title=vid,
|
||||
url=url or f"https://www.youtube.com/watch?v={vid}",
|
||||
upload_date=upload_date,
|
||||
)
|
||||
|
||||
|
||||
def _fake_channel(monkeypatch, catalog: list[VideoRef], calls: list | None = None):
|
||||
"""Patch discover_channel with a channel whose /videos tab is `catalog`
|
||||
(newest first) and that honours `limit` the way playlistend does."""
|
||||
|
||||
def fake(url, sleep_subrequests=2.0, limit=None):
|
||||
if calls is not None:
|
||||
calls.append(limit)
|
||||
return ("UC1", "Alpha", "http://avatar", catalog[:limit] if limit else list(catalog))
|
||||
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
|
||||
|
||||
|
||||
def test_stops_at_first_run_of_known_videos(monkeypatch):
|
||||
catalog = [_ref(f"v{i:03d}") for i in range(500)]
|
||||
known = {r.video_id for r in catalog[5:]} # everything except the 5 newest
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
|
||||
|
||||
assert calls == [30], "one pass only — must not paginate the whole channel"
|
||||
assert result.fetched == 30
|
||||
assert [r.video_id for r in result.new_refs] == [f"v{i:03d}" for i in range(5)]
|
||||
assert result.caught_up is True
|
||||
assert result.full_scan is False
|
||||
|
||||
|
||||
def test_widens_window_when_the_whole_window_is_new(monkeypatch):
|
||||
catalog = [_ref(f"v{i:03d}") for i in range(500)]
|
||||
known = {r.video_id for r in catalog[100:]} # 100 new uploads since last sync
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
|
||||
|
||||
# 30 -> 60 -> 120: doubles only as far as needed, never the full 500.
|
||||
assert calls == [30, 60, 120]
|
||||
assert result.new_count == 100
|
||||
assert result.caught_up is True
|
||||
|
||||
|
||||
def test_gives_up_at_max_window_and_says_so(monkeypatch):
|
||||
catalog = [_ref(f"v{i:03d}") for i in range(500)]
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental(
|
||||
"https://y/@alpha/videos", {"not-in-this-channel"}, window=10, max_window=40, overlap=3
|
||||
)
|
||||
|
||||
assert calls == [10, 20, 40]
|
||||
assert result.caught_up is False, "caller must be able to tell the scan was truncated"
|
||||
assert result.fetched == 40
|
||||
|
||||
|
||||
def test_no_local_history_walks_the_whole_channel(monkeypatch):
|
||||
catalog = [_ref(f"v{i:03d}") for i in range(120)]
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental("https://y/@alpha/videos", set(), window=30)
|
||||
|
||||
assert calls == [None], "a first sync has no boundary to stop at"
|
||||
assert result.full_scan is True
|
||||
assert result.new_count == 120
|
||||
|
||||
|
||||
def test_short_channel_is_exhausted_in_one_pass(monkeypatch):
|
||||
catalog = [_ref("a"), _ref("b")]
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=30)
|
||||
|
||||
assert calls == [30]
|
||||
assert result.exhausted is True
|
||||
assert result.caught_up is True
|
||||
assert [r.video_id for r in result.new_refs] == ["a"]
|
||||
|
||||
|
||||
def test_keep_filter_applies_before_the_overlap_check(monkeypatch):
|
||||
"""Shorts the store never recorded must not keep the window widening."""
|
||||
catalog = [
|
||||
_ref("s1", url="https://www.youtube.com/shorts/s1"),
|
||||
_ref("s2", url="https://www.youtube.com/shorts/s2"),
|
||||
_ref("s3", url="https://www.youtube.com/shorts/s3"),
|
||||
_ref("known1"),
|
||||
_ref("known2"),
|
||||
_ref("known3"),
|
||||
] + [_ref(f"v{i}") for i in range(50)]
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental(
|
||||
"https://y/@alpha/videos",
|
||||
{"known1", "known2", "known3"},
|
||||
window=6,
|
||||
overlap=3,
|
||||
keep=lambda r: "/shorts/" not in (r.url or ""),
|
||||
)
|
||||
|
||||
assert calls == [6], "the three known long-form videos end the scan"
|
||||
assert result.new_refs == []
|
||||
|
||||
|
||||
def test_upload_date_cutoff_stops_the_scan_when_dates_are_available(monkeypatch):
|
||||
catalog = [
|
||||
_ref("n1", "20260701"),
|
||||
_ref("n2", "20260630"),
|
||||
_ref("o1", "20250101"),
|
||||
_ref("o2", "20241231"),
|
||||
] + [_ref(f"old{i}", "20200101") for i in range(50)]
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental(
|
||||
"https://y/@alpha/videos", {"unrelated"}, window=4, overlap=2, since="20260601"
|
||||
)
|
||||
|
||||
assert calls == [4], "entries older than the watermark end the scan"
|
||||
assert result.caught_up is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("window,overlap", [(0, 0), (-5, -1)])
|
||||
def test_degenerate_settings_are_clamped(monkeypatch, window, overlap):
|
||||
catalog = [_ref("a"), _ref("b")]
|
||||
calls: list = []
|
||||
_fake_channel(monkeypatch, catalog, calls)
|
||||
|
||||
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=window, overlap=overlap)
|
||||
|
||||
assert calls and calls[0] >= 1
|
||||
assert result.fetched >= 1
|
||||
@@ -0,0 +1,154 @@
|
||||
"""Tests for per-language subtitle selection and dict-typed ``Config.languages``.
|
||||
|
||||
These tests are pure: no network, no yt-dlp, no DB. They cover the policy
|
||||
logic that decides which subtitle track to pick for a given
|
||||
``info`` dict produced by yt-dlp, plus the legacy/back-compat shims that
|
||||
let older YAML configs still load.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.config import Config, parse_languages
|
||||
from yt_scraper.extract import pick_subtitle
|
||||
|
||||
|
||||
# ---------- fixtures ----------
|
||||
|
||||
def _track(ext: str = "json3", url: str = "u") -> dict:
|
||||
return {"ext": ext, "url": url}
|
||||
|
||||
|
||||
def _info(manual: dict | None = None, auto: dict | None = None) -> dict:
|
||||
return {"subtitles": manual or {}, "automatic_captions": auto or {}}
|
||||
|
||||
|
||||
# ---------- pick_subtitle: mixed per-language mode ----------
|
||||
|
||||
def test_mixed_manual_and_auto_per_language() -> None:
|
||||
"""`es: auto` only takes auto; iterating `es` first means it wins.
|
||||
|
||||
With ``{"es": "auto", "en": "manual"}`` the iteration matches `es`
|
||||
against the auto dict and picks the auto track; for `en` only the
|
||||
manual dict is consulted and the manual track is returned. The two
|
||||
policies do NOT cross-pollinate between languages.
|
||||
"""
|
||||
info = _info(
|
||||
manual={"en": [_track(url="man-en")]},
|
||||
auto={"en": [_track(url="auto-en")], "es": [_track(url="auto-es")]},
|
||||
)
|
||||
pick = pick_subtitle(info, {"es": "auto", "en": "manual"}, prefer_manual=True)
|
||||
assert pick is not None
|
||||
assert pick.lang == "es"
|
||||
assert pick.source == "auto"
|
||||
|
||||
|
||||
def test_manual_only_skips_auto_even_when_present() -> None:
|
||||
info = _info(manual={}, auto={"es": [_track(url="auto")]})
|
||||
pick = pick_subtitle(info, {"es": "manual"}, prefer_manual=True)
|
||||
assert pick is None
|
||||
|
||||
|
||||
def test_auto_only_skips_manual_even_when_present() -> None:
|
||||
info = _info(manual={"es": [_track(url="man")]}, auto={})
|
||||
pick = pick_subtitle(info, {"es": "auto"}, prefer_manual=True)
|
||||
assert pick is None
|
||||
|
||||
|
||||
def test_any_defer_to_prefer_manual_default_true() -> None:
|
||||
"""`any` honours the legacy prefer_manual=True default."""
|
||||
info = _info(
|
||||
manual={"es": [_track(url="man")]},
|
||||
auto={"es": [_track(url="auto")]},
|
||||
)
|
||||
pick = pick_subtitle(info, {"es": "any"}, prefer_manual=True)
|
||||
assert pick is not None
|
||||
assert pick.source == "manual"
|
||||
|
||||
|
||||
def test_any_defer_to_prefer_manual_default_false() -> None:
|
||||
info = _info(
|
||||
manual={"es": [_track(url="man")]},
|
||||
auto={"es": [_track(url="auto")]},
|
||||
)
|
||||
pick = pick_subtitle(info, {"es": "any"}, prefer_manual=False)
|
||||
assert pick is not None
|
||||
assert pick.source == "auto"
|
||||
|
||||
|
||||
def test_picks_best_format_within_track() -> None:
|
||||
info = _info(manual={"en": [_track(ext="ttml", url="t"), _track(ext="json3", url="j")]})
|
||||
pick = pick_subtitle(info, {"en": "manual"}, prefer_manual=True)
|
||||
assert pick.url == "j"
|
||||
|
||||
|
||||
# ---------- language normalisation ----------
|
||||
|
||||
def test_language_base_match() -> None:
|
||||
"""`es-419` preference matches the `es` caption track."""
|
||||
info = _info(manual={"es": [_track(url="u")]})
|
||||
pick = pick_subtitle(info, {"es-419": "manual"}, prefer_manual=True)
|
||||
assert pick is not None
|
||||
assert pick.lang == "es"
|
||||
|
||||
|
||||
# ---------- legacy list input ----------
|
||||
|
||||
def test_legacy_list_input_uses_prefer_manual() -> None:
|
||||
info = _info(
|
||||
manual={"es": [_track(url="m")]},
|
||||
auto={"es": [_track(url="a")]},
|
||||
)
|
||||
pick = pick_subtitle(info, ["es"], prefer_manual=True)
|
||||
assert pick.source == "manual"
|
||||
pick = pick_subtitle(info, ["es"], prefer_manual=False)
|
||||
assert pick.source == "auto"
|
||||
|
||||
|
||||
def test_unknown_mode_falls_back_to_any() -> None:
|
||||
info = _info(manual={"es": [_track()]}, auto={"es": [_track()]})
|
||||
pick = pick_subtitle(info, {"es": "garbage"}, prefer_manual=True)
|
||||
assert pick is not None
|
||||
assert pick.source == "manual"
|
||||
|
||||
|
||||
# ---------- parse_languages config helper ----------
|
||||
|
||||
def test_parse_languages_dict_pass_through() -> None:
|
||||
assert parse_languages({"en": "manual", "es": "auto"}, True) == {
|
||||
"en": "manual", "es": "auto",
|
||||
}
|
||||
|
||||
|
||||
def test_parse_languages_dict_unknown_mode_to_any() -> None:
|
||||
assert parse_languages({"en": "garbage"}, True) == {"en": "any"}
|
||||
|
||||
|
||||
def test_parse_languages_legacy_list_to_manual_by_default() -> None:
|
||||
assert parse_languages(["es", "en"], prefer_manual=True) == {
|
||||
"es": "manual", "en": "manual",
|
||||
}
|
||||
|
||||
|
||||
def test_parse_languages_legacy_list_to_auto_when_prefer_manual_false() -> None:
|
||||
assert parse_languages(["es", "en"], prefer_manual=False) == {
|
||||
"es": "auto", "en": "auto",
|
||||
}
|
||||
|
||||
|
||||
def test_parse_languages_none_returns_empty() -> None:
|
||||
assert parse_languages(None, True) == {}
|
||||
|
||||
|
||||
# ---------- Config dataclass default shape ----------
|
||||
|
||||
def test_config_languages_defaults_to_dict() -> None:
|
||||
cfg = Config()
|
||||
assert isinstance(cfg.languages, dict)
|
||||
# "any", not "manual": the default must not exclude auto-generated captions.
|
||||
assert cfg.languages == {"es": "any", "en": "any"}
|
||||
|
||||
|
||||
def test_config_prefer_manual_defaults_true() -> None:
|
||||
cfg = Config()
|
||||
assert cfg.prefer_manual is True
|
||||
@@ -0,0 +1,118 @@
|
||||
"""One video must own exactly one .md, whatever happens to its title.
|
||||
|
||||
The filename is derived from the title, and titles are not stable: YouTube
|
||||
serves them localised, so the same video came back as "La controversia de
|
||||
Claude Fable 5" on one pass and "The Claude Fable controversy 5" on the next.
|
||||
Creators also simply rename videos.
|
||||
|
||||
`re_render_videos` already deleted the superseded file; `process_video` did not,
|
||||
so a re-scrape after a title change left the old file orphaned on disk. The DB
|
||||
repointed, the stale file stayed, and every later scan had to wade through it —
|
||||
the same shape as the incident that left 94 files for 61 rows.
|
||||
|
||||
The identity that matters is the video id, which never changes. These tests pin
|
||||
that: the row's `markdown_path` is authoritative, and anything it used to point
|
||||
at gets cleaned up.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.extract import SubtitlePick, VideoData
|
||||
from yt_scraper.parse import Segment
|
||||
from yt_scraper.pipeline import process_video
|
||||
from yt_scraper.render import build_filename_stem
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def env(tmp_path):
|
||||
cfg = Config(
|
||||
database_path=str(tmp_path / "state.db"),
|
||||
output_dir=str(tmp_path / "markdown"),
|
||||
template_path="templates/video.md.j2",
|
||||
)
|
||||
store = Store(cfg.database_path_resolved)
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([VideoRef("vid123", "UC1", "t", "https://y/watch?v=vid123", "20260101", 60)])
|
||||
return cfg, store, tmp_path
|
||||
|
||||
|
||||
def _data(title: str) -> VideoData:
|
||||
return VideoData(
|
||||
info={"id": "vid123", "title": title, "upload_date": "20260101", "channel": "Alpha"},
|
||||
segments=[Segment(start=0.0, end=2.0, text="hello")],
|
||||
subtitle=SubtitlePick(url="u", ext="json3", lang="en-orig", source="auto"),
|
||||
has_chapters=False,
|
||||
)
|
||||
|
||||
|
||||
def _md_files(root: Path) -> list[str]:
|
||||
return sorted(p.name for p in root.rglob("*.md"))
|
||||
|
||||
|
||||
def test_a_retitled_video_does_not_leave_a_second_file(env, monkeypatch):
|
||||
"""The production case: the same video, title localised differently."""
|
||||
cfg, store, tmp_path = env
|
||||
md_root = Path(cfg.output_dir_resolved)
|
||||
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
|
||||
lambda *a, **k: _data("La controversia de Claude Fable 5"))
|
||||
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
|
||||
first = _md_files(md_root)
|
||||
assert len(first) == 1
|
||||
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
|
||||
lambda *a, **k: _data("The Claude Fable controversy 5"))
|
||||
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
|
||||
|
||||
after = _md_files(md_root)
|
||||
assert len(after) == 1, f"one video, {len(after)} files on disk: {after}"
|
||||
# And the DB points at the one that exists.
|
||||
row = store.get_video("vid123")
|
||||
assert (Path(cfg.output_dir_resolved).parent / row.markdown_path).exists()
|
||||
assert Path(row.markdown_path).name == after[0]
|
||||
|
||||
|
||||
def test_rescraping_an_unchanged_video_is_idempotent(env, monkeypatch):
|
||||
cfg, store, tmp_path = env
|
||||
md_root = Path(cfg.output_dir_resolved)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("Same Title"))
|
||||
for _ in range(3):
|
||||
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
|
||||
assert len(_md_files(md_root)) == 1
|
||||
|
||||
|
||||
def test_changing_the_filename_template_relocates_rather_than_duplicates(env, monkeypatch):
|
||||
"""Opting into ids in the filename must not strand the old files."""
|
||||
cfg, store, tmp_path = env
|
||||
md_root = Path(cfg.output_dir_resolved)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("A Title"))
|
||||
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
|
||||
|
||||
cfg.filename_template = "{upload_date}_{slug}_{video_id}"
|
||||
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
|
||||
|
||||
files = _md_files(md_root)
|
||||
assert len(files) == 1, f"template change duplicated the file: {files}"
|
||||
assert "vid123" in files[0]
|
||||
|
||||
|
||||
# ------------------------------------------------------------ template vars
|
||||
|
||||
|
||||
def test_video_id_is_available_to_the_filename_template():
|
||||
"""Lets an operator make the file self-identifying without the DB."""
|
||||
stem = build_filename_stem(
|
||||
"20260101", "Some Title", template="{upload_date}_{slug}_{video_id}", video_id="abc123XYZ_-"
|
||||
)
|
||||
assert stem == "20260101_some-title_abc123XYZ_-"
|
||||
|
||||
|
||||
def test_default_template_is_unchanged():
|
||||
"""Existing libraries keep their filenames; adding the variable is opt-in."""
|
||||
assert build_filename_stem("20260101", "Some Title", video_id="abc123") == "20260101_some-title"
|
||||
@@ -0,0 +1,246 @@
|
||||
"""Unit tests for the politeness primitives.
|
||||
|
||||
Nothing here sleeps for real: `Pacer` takes an injectable clock and sleeper, and
|
||||
`backoff_delay` returns the delay instead of consuming it. A test suite that
|
||||
actually waited would be the first thing anyone deleted.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.ratelimit import (
|
||||
Pacer,
|
||||
ThrottleGuard,
|
||||
backoff_delay,
|
||||
is_quota_exhausted,
|
||||
is_rate_limited,
|
||||
ydl_throttle_opts,
|
||||
)
|
||||
|
||||
# Verbatim from the production database, where 343 of 350 `error` rows carried
|
||||
# one of these. If the detector stops matching them the circuit breaker becomes
|
||||
# decorative, so they are pinned here rather than paraphrased.
|
||||
REAL_THROTTLE_MESSAGES = [
|
||||
"ERROR: [youtube] abcdefghijk: Video unavailable. This content isn't available, "
|
||||
"try again later. The current session has been rate-limited by YouTube for up to an hour.",
|
||||
"ERROR: [youtube] abcdefghijk: This content isn't available, try again later. "
|
||||
"The current session has been rate-limited by YouTube for up to an hour. It is recommended t",
|
||||
"HTTPError: 429 Client Error: Too Many Requests for url: https://www.youtube.com/api/timedtext",
|
||||
"ERROR: [youtube] xyz: Sign in to confirm you're not a bot",
|
||||
"HTTP Error 429: Too Many Requests",
|
||||
]
|
||||
|
||||
# Equally verbatim: these are permanent, must NOT trip the breaker, and must
|
||||
# stay distinguishable from throttling.
|
||||
REAL_PERMANENT_MESSAGES = [
|
||||
"ERROR: [youtube] abcdefghijk: Join this channel to get access to members-only "
|
||||
"content like this video, and other exclusive perks.",
|
||||
"ERROR: [youtube] abcdefghijk: Private video. Sign in if you've been granted access to this video",
|
||||
"ERROR: [youtube] abcdefghijk: This video has been removed by the uploader",
|
||||
"no caption tracks published for this video",
|
||||
"subtitle downloaded but parsed empty (lang=es, format=json3)",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("msg", REAL_THROTTLE_MESSAGES)
|
||||
def test_detects_real_throttle_messages(msg):
|
||||
assert is_rate_limited(msg) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("msg", REAL_PERMANENT_MESSAGES)
|
||||
def test_ignores_permanent_failures(msg):
|
||||
assert is_rate_limited(msg) is False
|
||||
|
||||
|
||||
def test_quota_is_not_treated_as_plain_throttling():
|
||||
"""Google documents quota exhaustion as daily; backing off cannot fix it."""
|
||||
assert is_quota_exhausted("403 quotaExceeded") is True
|
||||
assert is_quota_exhausted("dailyLimitExceeded") is True
|
||||
assert is_quota_exhausted("rate-limited by YouTube") is False
|
||||
|
||||
|
||||
def test_rate_limited_accepts_exception_objects():
|
||||
assert is_rate_limited(RuntimeError("HTTP Error 429: Too Many Requests")) is True
|
||||
|
||||
|
||||
# Both spellings occur, and they are different strings. yt-dlp raises
|
||||
# "HTTP Error 429: ..."; `requests` raises "429 Client Error: ... for url: ...",
|
||||
# which the project wraps as "HTTPError: 429 ...". Detection used to rely on the
|
||||
# prose for the second form, so a 429 with no reason phrase — routine over
|
||||
# HTTP/2 — went unnoticed and the breaker never counted it.
|
||||
@pytest.mark.parametrize(
|
||||
"msg",
|
||||
[
|
||||
"HTTP Error 429: Too Many Requests",
|
||||
"HTTPError: 429 Client Error: Too Many Requests for url: https://youtube.com/api/timedtext",
|
||||
"HTTP Error 429: HTTPError: 429 Client Error: for url: https://youtube.com/api/timedtext",
|
||||
"429 Client Error: for url: https://www.youtube.com/api/timedtext?v=x",
|
||||
"HTTP Error 408: Request Timeout",
|
||||
"HTTPError: 408 Client Error: Request Timeout for url: https://youtube.com/",
|
||||
],
|
||||
)
|
||||
def test_detects_the_status_code_without_relying_on_the_reason_phrase(msg):
|
||||
assert is_rate_limited(msg) is True, f"undetected throttle: {msg}"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"msg",
|
||||
[
|
||||
"HTTP Error 404: Not Found",
|
||||
"HTTPError: 403 Client Error: Forbidden for url: https://youtube.com/",
|
||||
"HTTP Error 500: Internal Server Error",
|
||||
"no caption tracks published for this video",
|
||||
# A bare number must not be read as a status code.
|
||||
"video 429 seconds long with 408 segments",
|
||||
],
|
||||
)
|
||||
def test_does_not_treat_other_statuses_as_throttling(msg):
|
||||
assert is_rate_limited(msg) is False, f"false positive: {msg}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- backoff
|
||||
|
||||
|
||||
def test_backoff_grows_and_is_capped():
|
||||
delays = [backoff_delay(n, base=2.0, cap=60.0) for n in range(8)]
|
||||
# Jitter is < 1s so successive doublings still order strictly until the cap.
|
||||
assert delays[0] < delays[1] < delays[2] < delays[3]
|
||||
assert all(d <= 60.0 for d in delays)
|
||||
assert delays[-1] == 60.0
|
||||
|
||||
|
||||
def test_backoff_jitters():
|
||||
"""Same attempt must not produce the same delay twice, or concurrent
|
||||
clients would re-synchronise into waves — the reason Google mandates it."""
|
||||
seen = {backoff_delay(2, base=2.0, cap=60.0) for _ in range(30)}
|
||||
assert len(seen) > 1
|
||||
|
||||
|
||||
def test_backoff_survives_a_runaway_counter():
|
||||
assert backoff_delay(10_000, base=2.0, cap=60.0) == 60.0
|
||||
assert backoff_delay(-5, base=2.0, cap=60.0) <= 3.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- pacer
|
||||
|
||||
|
||||
class FakeClock:
|
||||
def __init__(self):
|
||||
self.now = 1000.0
|
||||
self.slept: list[float] = []
|
||||
|
||||
def time(self) -> float:
|
||||
return self.now
|
||||
|
||||
def sleep(self, seconds: float) -> None:
|
||||
self.slept.append(seconds)
|
||||
self.now += seconds
|
||||
|
||||
|
||||
def test_pacer_spaces_calls():
|
||||
clock = FakeClock()
|
||||
pacer = Pacer(2.0, clock=clock.time, sleeper=clock.sleep)
|
||||
assert pacer.wait() == 0.0 # first call is free
|
||||
assert pacer.wait() == pytest.approx(2.0)
|
||||
assert pacer.wait() == pytest.approx(2.0)
|
||||
assert clock.slept == [2.0, 2.0]
|
||||
|
||||
|
||||
def test_pacer_does_not_charge_for_time_already_spent():
|
||||
"""A caller slower than the interval should never wait on top of its own work."""
|
||||
clock = FakeClock()
|
||||
pacer = Pacer(2.0, clock=clock.time, sleeper=clock.sleep)
|
||||
pacer.wait()
|
||||
clock.now += 10.0 # the request itself took 10s
|
||||
assert pacer.wait() == 0.0
|
||||
assert clock.slept == []
|
||||
|
||||
|
||||
def test_pacer_disabled_by_default_interval():
|
||||
clock = FakeClock()
|
||||
pacer = Pacer(0.0, clock=clock.time, sleeper=clock.sleep)
|
||||
assert [pacer.wait() for _ in range(5)] == [0.0] * 5
|
||||
assert clock.slept == []
|
||||
|
||||
|
||||
def test_pacer_charges_for_multi_request_callers():
|
||||
"""One extract_info is two HTTP requests; billing it as one halves the budget."""
|
||||
clock = FakeClock()
|
||||
pacer = Pacer(2.0, clock=clock.time, sleeper=clock.sleep)
|
||||
pacer.wait(cost=2) # first call still free...
|
||||
assert pacer.wait() == pytest.approx(4.0) # ...but it reserved two slots
|
||||
|
||||
|
||||
def test_penalise_pushes_the_next_slot_out():
|
||||
clock = FakeClock()
|
||||
pacer = Pacer(1.0, clock=clock.time, sleeper=clock.sleep)
|
||||
pacer.wait()
|
||||
pacer.penalise(30.0)
|
||||
assert pacer.wait() == pytest.approx(30.0)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- guard
|
||||
|
||||
|
||||
def _guard() -> ThrottleGuard:
|
||||
return ThrottleGuard(threshold=3, base=0.01, cap=0.05)
|
||||
|
||||
|
||||
def test_guard_trips_after_consecutive_throttling():
|
||||
g = _guard()
|
||||
assert g.note_failure(REAL_THROTTLE_MESSAGES[0]) > 0
|
||||
assert not g.tripped
|
||||
g.note_failure(REAL_THROTTLE_MESSAGES[0])
|
||||
assert not g.tripped
|
||||
g.note_failure(REAL_THROTTLE_MESSAGES[0])
|
||||
assert g.tripped
|
||||
assert "consecutive" in (g.tripped_reason or "")
|
||||
|
||||
|
||||
def test_success_resets_the_streak():
|
||||
"""Isolated throttled videos between successes are noise, not a banned session."""
|
||||
g = _guard()
|
||||
for _ in range(10):
|
||||
g.note_failure(REAL_THROTTLE_MESSAGES[0])
|
||||
g.note_success()
|
||||
assert not g.tripped
|
||||
assert g.throttled_total == 10
|
||||
|
||||
|
||||
def test_permanent_failures_never_trip_the_breaker():
|
||||
"""A channel with a few members-only videos must not look like a ban."""
|
||||
g = _guard()
|
||||
for msg in REAL_PERMANENT_MESSAGES * 5:
|
||||
g.note_failure(msg)
|
||||
assert not g.tripped
|
||||
assert g.throttled_total == 0
|
||||
|
||||
|
||||
def test_mixed_failures_do_not_accumulate_into_a_trip():
|
||||
g = _guard()
|
||||
g.note_failure(REAL_THROTTLE_MESSAGES[0])
|
||||
g.note_failure("Private video")
|
||||
g.note_failure(REAL_THROTTLE_MESSAGES[0])
|
||||
g.note_failure("no caption tracks published for this video")
|
||||
g.note_failure(REAL_THROTTLE_MESSAGES[0])
|
||||
assert not g.tripped
|
||||
|
||||
|
||||
def test_quota_trips_immediately_without_backoff():
|
||||
g = _guard()
|
||||
assert g.note_failure("403 quotaExceeded") == 0.0
|
||||
assert g.tripped
|
||||
assert "quota" in (g.tripped_reason or "").lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- ydl opts
|
||||
|
||||
|
||||
def test_throttle_opts_use_names_yt_dlp_actually_reads():
|
||||
opts = ydl_throttle_opts(2.5, extractor_retries=4, socket_timeout=15.0)
|
||||
assert opts["sleep_interval_requests"] == 2.5
|
||||
assert opts["extractor_retries"] == 4
|
||||
assert opts["socket_timeout"] == 15.0
|
||||
# The bug this whole module exists to prevent.
|
||||
assert "sleep_subrequests" not in opts
|
||||
@@ -0,0 +1,356 @@
|
||||
"""Recovery paths: retryable statuses, recorded skip reasons, disk<->DB reconcile.
|
||||
|
||||
The motivating incident: 511 videos were stored as `no_subtitles` because the
|
||||
language policy was manual-only while the channel publishes only auto-generated
|
||||
captions. `reset_errors` could not reach them, and nothing recorded why they
|
||||
were skipped, so the failure was both invisible and irreversible.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from yt_scraper.extract import describe_missing_subtitle
|
||||
from yt_scraper.segments import reconcile_markdown
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
|
||||
|
||||
def _store(tmp_path) -> Store:
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
||||
return store
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- reset
|
||||
|
||||
def test_reset_reaches_no_subtitles_not_just_error(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
|
||||
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
|
||||
VideoRef("c", "UC1", "C", "https://y/watch?v=c"),
|
||||
])
|
||||
store.mark_status("a", "no_subtitles", "policy rejected auto captions")
|
||||
store.mark_error("b", "rate limited")
|
||||
store.mark_done("c", "markdown/c.md", "es", "auto", False)
|
||||
|
||||
n = store.reset_videos("UC1", ("error", "no_subtitles"))
|
||||
|
||||
assert n == 2
|
||||
assert store.get_video("a").status == "pending"
|
||||
assert store.get_video("b").status == "pending"
|
||||
assert store.get_video("c").status == "done", "finished work must not be re-queued"
|
||||
|
||||
|
||||
def test_reset_never_touches_done_even_if_asked(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("c", "UC1", "C", "https://y/watch?v=c")])
|
||||
store.mark_done("c", "markdown/c.md", "es", "auto", False)
|
||||
|
||||
assert store.reset_videos("UC1", ("done",)) == 0
|
||||
assert store.get_video("c").status == "done"
|
||||
|
||||
|
||||
def test_reset_errors_still_only_resets_errors(tmp_path):
|
||||
"""The narrower legacy helper keeps its old meaning."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
|
||||
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
|
||||
])
|
||||
store.mark_status("a", "no_subtitles")
|
||||
store.mark_error("b", "boom")
|
||||
|
||||
assert store.reset_errors("UC1") == 1
|
||||
assert store.get_video("a").status == "no_subtitles"
|
||||
assert store.get_video("b").status == "pending"
|
||||
|
||||
|
||||
def test_permanent_failures_are_excluded_from_retries(tmp_path):
|
||||
"""Members-only videos cannot be fixed by retrying; re-running them only
|
||||
spends requests the recoverable videos need."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef(v, "UC1", v, f"https://y/watch?v={v}") for v in ("members", "throttled", "priv")
|
||||
])
|
||||
store.mark_error("members", "ERROR: [youtube] x: Join this channel to get access to members-only content")
|
||||
store.mark_error("throttled", "ERROR: Video unavailable. The current session has been rate-limited by YouTube")
|
||||
store.mark_error("priv", "ERROR: Private video. Sign in if you've been granted access")
|
||||
|
||||
counts = store.retryable_counts("UC1")
|
||||
assert counts["error"] == 1, "only the throttled one is worth retrying"
|
||||
assert counts["permanent"] == 2
|
||||
|
||||
assert store.reset_videos("UC1", ("error",)) == 1
|
||||
assert store.get_video("throttled").status == "pending"
|
||||
assert store.get_video("members").status == "error"
|
||||
assert store.get_video("priv").status == "error"
|
||||
|
||||
|
||||
def test_permanent_failures_can_be_reset_when_explicitly_asked(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("members", "UC1", "M", "https://y/watch?v=members")])
|
||||
store.mark_error("members", "ERROR: members-only content")
|
||||
|
||||
assert store.reset_videos("UC1", ("error",)) == 0
|
||||
assert store.reset_videos("UC1", ("error",), include_permanent=True) == 1
|
||||
assert store.get_video("members").status == "pending"
|
||||
|
||||
|
||||
def test_rate_limit_wording_is_never_treated_as_permanent(tmp_path):
|
||||
"""The throttling message is the one that must stay retryable."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("t", "UC1", "T", "https://y/watch?v=t")])
|
||||
store.mark_error(
|
||||
"t",
|
||||
"ERROR: [youtube] t: Video unavailable. This content isn't available, try again later. "
|
||||
"The current session has been rate-limited by YouTube for up to an hour.",
|
||||
)
|
||||
|
||||
assert store.retryable_counts("UC1")["permanent"] == 0
|
||||
assert store.reset_videos("UC1", ("error",)) == 1
|
||||
|
||||
|
||||
def test_retryable_counts_reports_both_statuses(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef(v, "UC1", v, f"https://y/watch?v={v}") for v in ("a", "b", "c")
|
||||
])
|
||||
store.mark_status("a", "no_subtitles")
|
||||
store.mark_status("b", "no_subtitles")
|
||||
store.mark_error("c", "boom")
|
||||
|
||||
assert store.retryable_counts("UC1") == {"error": 1, "no_subtitles": 2, "permanent": 0}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- skip reasons
|
||||
|
||||
def test_mark_status_records_the_reason(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a")])
|
||||
|
||||
store.mark_status("a", "no_subtitles", "no track matched the language policy")
|
||||
|
||||
assert "language policy" in store.get_video("a").error_msg
|
||||
|
||||
|
||||
def test_describe_distinguishes_no_captions_from_policy_rejection():
|
||||
none_at_all = describe_missing_subtitle({"subtitles": {}, "automatic_captions": {}}, {"es": "manual"})
|
||||
assert "no caption tracks published" in none_at_all
|
||||
|
||||
auto_only = describe_missing_subtitle(
|
||||
{"subtitles": {}, "automatic_captions": {"es": [{"url": "u", "ext": "json3"}]}},
|
||||
{"es": "manual"},
|
||||
)
|
||||
assert "ONLY auto-generated" in auto_only
|
||||
assert "'any' or 'auto'" in auto_only
|
||||
|
||||
|
||||
def test_describe_does_not_blame_config_when_mode_already_allows_auto():
|
||||
msg = describe_missing_subtitle(
|
||||
{"subtitles": {}, "automatic_captions": {"de": [{"url": "u", "ext": "json3"}]}},
|
||||
{"es": "any"},
|
||||
)
|
||||
assert "ONLY auto-generated" not in msg
|
||||
assert "no track matched the language policy" in msg
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- reconcile
|
||||
|
||||
_MD = """---
|
||||
video_id: "{vid}"
|
||||
title: "T"
|
||||
upload_date: "2026-01-02"
|
||||
---
|
||||
|
||||
## Transcript
|
||||
|
||||
**00:00** · hola mundo
|
||||
"""
|
||||
|
||||
|
||||
def test_reconcile_marks_done_when_the_md_is_already_on_disk(tmp_path):
|
||||
"""The exact symptom the user reported: a .md exists but the row still
|
||||
shows a failure, and nothing reconciles it without a server restart."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1")])
|
||||
store.mark_status("vid1", "no_subtitles", "stale failure")
|
||||
|
||||
md_root = tmp_path / "markdown" / "Alpha"
|
||||
md_root.mkdir(parents=True)
|
||||
(md_root / "2026-01-02_t.md").write_text(_MD.format(vid="vid1"), encoding="utf-8")
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown")
|
||||
|
||||
row = store.get_video("vid1")
|
||||
assert row.status == "done"
|
||||
assert row.markdown_path == "markdown/Alpha/2026-01-02_t.md"
|
||||
assert result["repaired_done"] == 1
|
||||
|
||||
|
||||
def test_reconcile_requeues_rows_whose_md_vanished_only_with_prune(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef("gone", "UC1", "G", "https://y/watch?v=gone"),
|
||||
VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1"),
|
||||
])
|
||||
store.mark_done("gone", "markdown/Alpha/nope.md", "es", "auto", False)
|
||||
md_root = tmp_path / "markdown" / "Alpha"
|
||||
md_root.mkdir(parents=True)
|
||||
(md_root / "2026-01-02_t.md").write_text(_MD.format(vid="vid1"), encoding="utf-8")
|
||||
|
||||
# default is non-destructive
|
||||
assert reconcile_markdown(store, tmp_path / "markdown")["missing_md"] == 0
|
||||
assert store.get_video("gone").status == "done"
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown", prune=True)
|
||||
assert result["missing_md"] == 1
|
||||
assert store.get_video("gone").status == "pending"
|
||||
|
||||
|
||||
def test_prune_refuses_to_demote_everything_when_the_root_is_empty(tmp_path):
|
||||
"""Pointed at a wrong or not-yet-populated markdown root, prune must be a
|
||||
no-op rather than wiping every finished video in the database."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("c", "UC1", "C", "https://y/watch?v=c")])
|
||||
store.mark_done("c", "markdown/Alpha/c.md", "es", "auto", False)
|
||||
(tmp_path / "markdown").mkdir()
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown", prune=True)
|
||||
|
||||
assert result["missing_md"] == 0
|
||||
assert store.get_video("c").status == "done"
|
||||
|
||||
|
||||
def test_reconcile_is_idempotent(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1")])
|
||||
md_root = tmp_path / "markdown" / "Alpha"
|
||||
md_root.mkdir(parents=True)
|
||||
(md_root / "2026-01-02_t.md").write_text(_MD.format(vid="vid1"), encoding="utf-8")
|
||||
|
||||
first = reconcile_markdown(store, tmp_path / "markdown")
|
||||
second = reconcile_markdown(store, tmp_path / "markdown")
|
||||
|
||||
assert first["repaired_done"] == 1
|
||||
assert second["repaired_done"] == 0
|
||||
assert second["missing_md"] == 0
|
||||
assert store.get_video("vid1").status == "done"
|
||||
|
||||
|
||||
def test_a_bad_encoding_does_not_abort_the_whole_scan(tmp_path):
|
||||
"""UnicodeDecodeError is a ValueError, not an OSError. Letting it escape
|
||||
aborted the loop, so one bad file silently hid every later one."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef("v1", "UC1", "A", "https://y/watch?v=v1"),
|
||||
VideoRef("v3", "UC1", "C", "https://y/watch?v=v3"),
|
||||
])
|
||||
md_root = tmp_path / "markdown" / "Alpha"
|
||||
md_root.mkdir(parents=True)
|
||||
(md_root / "a.md").write_text(_MD.format(vid="v1"), encoding="utf-8")
|
||||
(md_root / "b.md").write_bytes(b"---\nvideo_id: \xe9\xe9\xe9\n---\n")
|
||||
(md_root / "c.md").write_text(_MD.format(vid="v3"), encoding="utf-8")
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown")
|
||||
|
||||
assert store.get_video("v3").status == "done", "the file after the bad one must still import"
|
||||
assert result["repaired_done"] == 2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- re-render
|
||||
|
||||
def test_re_render_updates_markdown_path_and_removes_the_old_file(tmp_path):
|
||||
"""re-render used a raw compact upload_date while process_video uses the
|
||||
hyphenated form, so it wrote a SECOND .md and never told the DB — leaving
|
||||
the app serving the older file. Measured on real data: 94 files, 61 rows."""
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.pipeline import re_render_videos
|
||||
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("v1", "UC1", "Mi Video", "https://y/watch?v=v1", "20240519", 60)])
|
||||
store.update_video_metadata(
|
||||
"v1", view_count=1, like_count=1, tags=None, thumbnail=None, description=None,
|
||||
chapters_json="[]",
|
||||
segments_json='[{"start": 0.0, "end": 2.0, "text": "hola"}]',
|
||||
)
|
||||
md_root = tmp_path / "markdown"
|
||||
old_dir = md_root / "Alpha"
|
||||
old_dir.mkdir(parents=True)
|
||||
(old_dir / "2024-05-19_mi-video.md").write_text("stale", encoding="utf-8")
|
||||
store.mark_done("v1", "markdown/Alpha/2024-05-19_mi-video.md", "es", "auto", False)
|
||||
|
||||
cfg = Config(
|
||||
database_path=str(tmp_path / "state.db"),
|
||||
output_dir=str(md_root),
|
||||
template_path=str(Path("templates/video.md.j2").resolve()),
|
||||
)
|
||||
assert re_render_videos(store, cfg) == 1
|
||||
|
||||
files = sorted(p.name for p in md_root.rglob("*.md"))
|
||||
assert len(files) == 1, f"re-render must not leave an orphan beside it: {files}"
|
||||
row = store.get_video("v1")
|
||||
assert row.markdown_path.replace("\\", "/").endswith(files[0])
|
||||
assert (md_root.parent / row.markdown_path).exists()
|
||||
assert "stale" not in (md_root.parent / row.markdown_path).read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_re_render_is_idempotent(tmp_path):
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.pipeline import re_render_videos
|
||||
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("v1", "UC1", "Mi Video", "https://y/watch?v=v1", "20240519", 60)])
|
||||
store.update_video_metadata(
|
||||
"v1", view_count=None, like_count=None, tags=None, thumbnail=None, description=None,
|
||||
chapters_json="[]", segments_json='[{"start": 0.0, "end": 2.0, "text": "hola"}]',
|
||||
)
|
||||
store.mark_done("v1", "markdown/Alpha/whatever.md", "es", "auto", False)
|
||||
cfg = Config(
|
||||
database_path=str(tmp_path / "state.db"),
|
||||
output_dir=str(tmp_path / "markdown"),
|
||||
template_path=str(Path("templates/video.md.j2").resolve()),
|
||||
)
|
||||
|
||||
re_render_videos(store, cfg)
|
||||
first = store.get_video("v1").markdown_path
|
||||
re_render_videos(store, cfg)
|
||||
|
||||
assert store.get_video("v1").markdown_path == first
|
||||
assert len(list((tmp_path / "markdown").rglob("*.md"))) == 1
|
||||
|
||||
|
||||
def test_reconcile_reports_stale_duplicates_and_deletes_them_only_with_prune(tmp_path):
|
||||
"""The 33 leftover files the old re-render wrote under a second filename:
|
||||
the DB points at one, the other is dead weight."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1")])
|
||||
md_root = tmp_path / "markdown" / "Alpha"
|
||||
md_root.mkdir(parents=True)
|
||||
canonical = md_root / "2026-01-02_t.md"
|
||||
duplicate = md_root / "20260102_t.md"
|
||||
canonical.write_text(_MD.format(vid="vid1"), encoding="utf-8")
|
||||
duplicate.write_text(_MD.format(vid="vid1"), encoding="utf-8")
|
||||
store.mark_done("vid1", "markdown/Alpha/2026-01-02_t.md", "es", "auto", False)
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown")
|
||||
assert result["stale_dupe"] == 1
|
||||
assert duplicate.exists(), "reporting only by default"
|
||||
assert store.get_video("vid1").markdown_path == "markdown/Alpha/2026-01-02_t.md"
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown", prune=True)
|
||||
assert result["stale_dupe"] == 1
|
||||
assert not duplicate.exists()
|
||||
assert canonical.exists(), "the file the DB points at must survive"
|
||||
assert store.get_video("vid1").status == "done"
|
||||
|
||||
|
||||
def test_reconcile_counts_orphan_markdown(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
md_root = tmp_path / "markdown" / "Alpha"
|
||||
md_root.mkdir(parents=True)
|
||||
(md_root / "ghost.md").write_text(_MD.format(vid="not-in-db"), encoding="utf-8")
|
||||
|
||||
result = reconcile_markdown(store, tmp_path / "markdown")
|
||||
|
||||
assert result["orphan_md"] == 1
|
||||
assert result["repaired_done"] == 0
|
||||
@@ -0,0 +1,171 @@
|
||||
"""Guards on how many requests we spend and under what option names.
|
||||
|
||||
Two classes of regression live here, both of which actually happened:
|
||||
|
||||
1. An option name yt-dlp does not recognise. yt-dlp ignores unknown keys
|
||||
silently, so `sleep_subrequests` looked configured for the project's whole
|
||||
history while nothing ever slept between requests. `test_*_options_are_real`
|
||||
checks every key against yt-dlp's own list instead of trusting review.
|
||||
|
||||
2. An extraction that walks far more of a channel than it needs.
|
||||
`deep_channel_avatar` read one avatar URL by fully extracting every video the
|
||||
channel had ever published — 735 requests and still going when a measurement
|
||||
aborted it. The bound is asserted here because nothing else would notice.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
import yt_dlp
|
||||
|
||||
from yt_scraper import discover, extract
|
||||
|
||||
#: Every option name yt-dlp's own CLI parser produces. Anything outside this is
|
||||
#: either a typo or something yt-dlp will silently drop.
|
||||
KNOWN_YDL_OPTIONS = set(yt_dlp.parse_options([]).ydl_opts)
|
||||
|
||||
|
||||
class CapturingYDL:
|
||||
"""Stands in for yt_dlp.YoutubeDL and records the options it was built with."""
|
||||
|
||||
captured: list[dict] = []
|
||||
info: dict = {}
|
||||
|
||||
def __init__(self, options=None, *args, **kwargs):
|
||||
type(self).captured.append(dict(options or {}))
|
||||
self.options = options or {}
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc):
|
||||
return False
|
||||
|
||||
def extract_info(self, url, download=False, process=True):
|
||||
return dict(type(self).info)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def capture(monkeypatch):
|
||||
CapturingYDL.captured = []
|
||||
CapturingYDL.info = {
|
||||
"id": "UC123",
|
||||
"channel_id": "UC123",
|
||||
"channel": "Test Channel",
|
||||
"thumbnails": [{"url": "https://yt3.ggpht.com/avatar.jpg"}],
|
||||
"entries": [],
|
||||
}
|
||||
monkeypatch.setattr(yt_dlp, "YoutubeDL", CapturingYDL)
|
||||
return CapturingYDL
|
||||
|
||||
|
||||
def assert_options_are_real(opts: dict, where: str) -> None:
|
||||
unknown = sorted(set(opts) - KNOWN_YDL_OPTIONS)
|
||||
assert not unknown, (
|
||||
f"{where} passes option(s) yt-dlp does not recognise and will silently "
|
||||
f"ignore: {unknown}"
|
||||
)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ names
|
||||
|
||||
|
||||
def test_discover_channel_options_are_real(capture):
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
|
||||
assert_options_are_real(capture.captured[0], "discover_channel")
|
||||
|
||||
|
||||
def test_deep_channel_avatar_options_are_real(capture):
|
||||
discover.deep_channel_avatar("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
|
||||
assert_options_are_real(capture.captured[0], "deep_channel_avatar")
|
||||
|
||||
|
||||
def test_extract_video_options_are_real(capture):
|
||||
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
|
||||
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
|
||||
assert_options_are_real(capture.captured[0], "extract_video")
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ throttles wired
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"call",
|
||||
[
|
||||
pytest.param(
|
||||
lambda: discover.discover_channel("https://www.youtube.com/@x/videos",
|
||||
sleep_subrequests=3.25),
|
||||
id="discover_channel",
|
||||
),
|
||||
pytest.param(
|
||||
lambda: discover.deep_channel_avatar("https://www.youtube.com/@x/videos",
|
||||
sleep_subrequests=3.25),
|
||||
id="deep_channel_avatar",
|
||||
),
|
||||
pytest.param(
|
||||
lambda: extract.extract_video("https://www.youtube.com/watch?v=vid",
|
||||
{"es": "any"}, sleep_subrequests=3.25),
|
||||
id="extract_video",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_every_entry_point_forwards_the_real_sleep_option(capture, call):
|
||||
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {},
|
||||
"channel_id": "UC1", "entries": []}
|
||||
call()
|
||||
opts = capture.captured[0]
|
||||
assert opts.get("sleep_interval_requests") == 3.25
|
||||
assert opts.get("socket_timeout"), "a hung connection must not block the worker forever"
|
||||
assert "extractor_retries" in opts
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ request bounds
|
||||
|
||||
|
||||
def test_deep_avatar_does_not_walk_the_channel(capture):
|
||||
"""The 735-request bug. Both halves of the fix are asserted.
|
||||
|
||||
`extract_flat` stops yt-dlp expanding each entry into a full extraction, and
|
||||
`playlistend` stops it paginating past the first page. Either one missing
|
||||
puts the whole channel back on the wire.
|
||||
"""
|
||||
discover.deep_channel_avatar("https://www.youtube.com/@x/videos")
|
||||
opts = capture.captured[0]
|
||||
assert opts.get("extract_flat"), "must not fully extract every video"
|
||||
assert opts.get("playlistend") == 1, "must not paginate beyond the first page"
|
||||
|
||||
|
||||
def test_discover_channel_limit_becomes_playlistend(capture):
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
|
||||
assert capture.captured[0].get("playlistend") == 30
|
||||
|
||||
|
||||
def test_discover_channel_without_limit_has_no_ceiling(capture):
|
||||
"""A brand-new channel legitimately walks everything; that must stay possible."""
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos")
|
||||
assert "playlistend" not in capture.captured[0]
|
||||
|
||||
|
||||
def test_extraction_never_probes_formats(capture):
|
||||
"""`check_formats` costs one HTTP request per format and we only want captions."""
|
||||
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
|
||||
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
|
||||
assert capture.captured[0].get("check_formats") is None
|
||||
|
||||
|
||||
def test_discovery_processes_the_result_so_playlistend_applies(capture, monkeypatch):
|
||||
"""`playlistend` is silently ignored when extract_info runs with process=False.
|
||||
|
||||
Measured on a 2564-video channel: processed + playlistend=60 costs 2
|
||||
requests; unprocessed, the lazy generator ignores the limit and walking it
|
||||
costs 86. Nothing else in the codebase would catch that flip.
|
||||
"""
|
||||
seen = {}
|
||||
|
||||
def extract_info(self, url, download=False, process=True):
|
||||
seen["process"] = process
|
||||
return dict(CapturingYDL.info)
|
||||
|
||||
monkeypatch.setattr(CapturingYDL, "extract_info", extract_info)
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
|
||||
assert seen["process"] is not False
|
||||
@@ -0,0 +1,33 @@
|
||||
"""`--since` / the scrape form's date filter must not eat undated entries.
|
||||
|
||||
yt-dlp's flat channel listing does not report upload_date, so comparing a
|
||||
missing date against the cutoff used to discard everything discovery found.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from yt_scraper.cli import _apply_filters
|
||||
from yt_scraper.store import VideoRef
|
||||
|
||||
|
||||
def _ref(vid: str, upload_date: str | None) -> VideoRef:
|
||||
return VideoRef(vid, "UC1", vid, f"https://www.youtube.com/watch?v={vid}", upload_date)
|
||||
|
||||
|
||||
def test_since_keeps_entries_with_no_upload_date():
|
||||
refs = [_ref("undated", None), _ref("new", "20260715"), _ref("old", "20200101")]
|
||||
|
||||
kept = _apply_filters(refs, "2026-07-01", False, False, 0, None)
|
||||
|
||||
assert [r.video_id for r in kept] == ["undated", "new"]
|
||||
|
||||
|
||||
def test_since_still_drops_entries_that_are_provably_older():
|
||||
refs = [_ref("old", "20200101"), _ref("older", "20190101")]
|
||||
|
||||
assert _apply_filters(refs, "2026-07-01", False, False, 0, None) == []
|
||||
|
||||
|
||||
def test_without_since_nothing_is_dropped_on_date_grounds():
|
||||
refs = [_ref("undated", None), _ref("old", "20200101")]
|
||||
|
||||
assert len(_apply_filters(refs, None, False, False, 0, None)) == 2
|
||||
@@ -100,6 +100,37 @@ def test_query_videos_filters(seeded_store):
|
||||
assert rows3[0].video_id == "v2"
|
||||
|
||||
|
||||
def test_upsert_videos_reports_only_new_rows(seeded_store):
|
||||
inserted = seeded_store.upsert_videos([
|
||||
VideoRef("v1", "UC1", "Updated title", "https://y/watch?v=v1", "20240101", 120),
|
||||
VideoRef("v4", "UC1", "New video", "https://y/watch?v=v4", "20240501", 180),
|
||||
])
|
||||
|
||||
assert inserted == 1
|
||||
assert seeded_store.get_video("v1").title == "Updated title"
|
||||
assert seeded_store.get_video("v4").status == "pending"
|
||||
|
||||
|
||||
def test_upload_sort_uses_discovered_time_when_date_is_missing(seeded_store):
|
||||
seeded_store.upsert_videos([
|
||||
VideoRef("recent", "UC1", "Recent discovered", "https://y/watch?v=recent"),
|
||||
])
|
||||
|
||||
rows, _ = seeded_store.query_videos(channel_id="UC1", sort="upload_date", page=1, size=1)
|
||||
|
||||
assert rows[0].video_id == "recent"
|
||||
|
||||
|
||||
def test_mark_status_clears_stale_error_message(seeded_store):
|
||||
seeded_store.mark_error("v1", "temporary extraction failure")
|
||||
|
||||
seeded_store.mark_status("v1", "no_subtitles")
|
||||
|
||||
video = seeded_store.get_video("v1")
|
||||
assert video.status == "no_subtitles"
|
||||
assert video.error_msg is None
|
||||
|
||||
|
||||
def test_cookie_vault(store):
|
||||
from yt_scraper import cookies
|
||||
sample = (
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Store-side support for incremental sync: the known-id boundary and the
|
||||
watermark that survives a re-discovery."""
|
||||
from __future__ import annotations
|
||||
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
|
||||
|
||||
def _store(tmp_path) -> Store:
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
||||
return store
|
||||
|
||||
|
||||
def test_known_video_ids_is_scoped_to_the_channel(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
||||
store.upsert_videos([
|
||||
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
|
||||
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
|
||||
VideoRef("c", "UC2", "C", "https://y/watch?v=c"),
|
||||
])
|
||||
|
||||
assert store.known_video_ids("UC1") == {"a", "b"}
|
||||
assert store.known_video_ids("UC2") == {"c"}
|
||||
assert store.known_video_ids("nope") == set()
|
||||
|
||||
|
||||
def test_rediscovery_does_not_wipe_a_known_upload_date(tmp_path):
|
||||
"""Flat discovery reports upload_date=None; without COALESCE a routine sync
|
||||
would erase the dates learned during extraction — and with them the very
|
||||
watermark this feature is built on."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260715", 120)])
|
||||
|
||||
store.upsert_videos([VideoRef("a", "UC1", "A (renamed)", "https://y/watch?v=a", None, None)])
|
||||
|
||||
row = store.get_video("a")
|
||||
assert row.upload_date == "20260715"
|
||||
assert row.duration == 120
|
||||
assert row.title == "A (renamed)", "titles should still refresh"
|
||||
|
||||
|
||||
def test_latest_upload_date_ignores_undated_rows(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260101"),
|
||||
VideoRef("b", "UC1", "B", "https://y/watch?v=b", None),
|
||||
VideoRef("c", "UC1", "C", "https://y/watch?v=c", "20260720"),
|
||||
])
|
||||
|
||||
assert store.latest_upload_date("UC1") == "20260720"
|
||||
|
||||
|
||||
def test_mark_channel_synced_recounts_instead_of_trusting_the_window(tmp_path):
|
||||
"""An incremental pass only sees the newest slice, so video_count must come
|
||||
from the DB — otherwise an 848-video channel shrinks to the window size."""
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([
|
||||
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", "20260101")
|
||||
for i in range(40)
|
||||
])
|
||||
|
||||
marks = store.mark_channel_synced("UC1")
|
||||
|
||||
assert marks["video_count"] == 40
|
||||
channel = store.get_channel("UC1")
|
||||
assert channel["video_count"] == 40
|
||||
assert channel["last_video_date"] == "20260101"
|
||||
assert channel["last_synced_at"]
|
||||
|
||||
|
||||
def test_mark_channel_synced_keeps_the_last_date_when_nothing_is_dated(tmp_path):
|
||||
store = _store(tmp_path)
|
||||
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260101")])
|
||||
store.mark_channel_synced("UC1")
|
||||
|
||||
store.upsert_videos([VideoRef("b", "UC1", "B", "https://y/watch?v=b", None)])
|
||||
store.mark_channel_synced("UC1")
|
||||
|
||||
assert store.get_channel("UC1")["last_video_date"] == "20260101"
|
||||
@@ -0,0 +1,219 @@
|
||||
"""The transcript must be the language actually spoken, not a machine translation.
|
||||
|
||||
Found in production on the Alex Hormozi channel (English). Under
|
||||
`languages: {es: any, es-419: any, en: any}` the picker walked the config in
|
||||
order, matched `es` first, and stored a Spanish auto-translation of English
|
||||
speech — "Soy Nim Jenkinson y enseño a los aficionados a las manualidades" for
|
||||
a video whose speaker says it in English.
|
||||
|
||||
Two things conspired:
|
||||
|
||||
1. `_normalize_lang("es-orig") == "es"`, so the `-orig` suffix — the one piece
|
||||
of evidence distinguishing YouTube's real ASR track from a translation *into
|
||||
the same language* — was thrown away before comparison.
|
||||
2. Config order was treated as absolute preference, so a translation into a
|
||||
preferred language beat the original.
|
||||
|
||||
Spanish channels were unaffected only by luck: yt-dlp happens to list `es-orig`
|
||||
before `es`, so dict order gave the right answer. Every `done` row on the four
|
||||
Spanish channels recorded `transcript_lang=es-orig`; the three Hormozi rows
|
||||
recorded `es`. That asymmetry is what these tests pin down.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from yt_scraper.extract import original_language, pick_subtitle
|
||||
|
||||
CFG = {"es": "any", "es-419": "any", "en": "any"}
|
||||
|
||||
|
||||
def _track(url: str) -> list[dict]:
|
||||
return [{"ext": "json3", "url": url}]
|
||||
|
||||
|
||||
def _english_video() -> dict:
|
||||
"""An English video as YouTube actually presents it: the original ASR under
|
||||
`en-orig`, plus a long tail of translations keyed by bare language code."""
|
||||
return {
|
||||
"id": "aRVv5NLVRwE",
|
||||
"title": "My honest advice to someone who wants to get rich.",
|
||||
"subtitles": {},
|
||||
"automatic_captions": {
|
||||
"en-orig": _track("https://timedtext/en-orig"),
|
||||
"en": _track("https://timedtext/en-translated"),
|
||||
"es": _track("https://timedtext/es"),
|
||||
"es-419": _track("https://timedtext/es-419"),
|
||||
"fr": _track("https://timedtext/fr"),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _spanish_video() -> dict:
|
||||
return {
|
||||
"id": "uZH3FKH_yNw",
|
||||
"title": "Escribir codigo a mano sera irresponsable",
|
||||
"subtitles": {},
|
||||
"automatic_captions": {
|
||||
"es-orig": _track("https://timedtext/es-orig"),
|
||||
"es": _track("https://timedtext/es"),
|
||||
"en": _track("https://timedtext/en"),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def test_english_video_yields_english_not_a_spanish_translation():
|
||||
"""The production bug, verbatim."""
|
||||
pick = pick_subtitle(_english_video(), CFG, prefer_manual=True)
|
||||
assert pick is not None
|
||||
assert pick.lang == "en-orig", f"picked {pick.lang}: a translation, not the spoken language"
|
||||
|
||||
|
||||
def test_spanish_video_still_yields_the_spanish_original():
|
||||
"""The fix must not regress the four Spanish channels already in the DB."""
|
||||
pick = pick_subtitle(_spanish_video(), CFG, prefer_manual=True)
|
||||
assert pick is not None
|
||||
assert pick.lang == "es-orig"
|
||||
|
||||
|
||||
def test_original_beats_a_same_language_translation():
|
||||
"""YouTube publishes a translation *into the video's own language* too.
|
||||
|
||||
`en-orig` and `en` both normalise to "en"; only the suffix says which one is
|
||||
the real transcript, and dict order must not decide it.
|
||||
"""
|
||||
info = {
|
||||
"subtitles": {},
|
||||
# Deliberately listed translation-first to defeat insertion order.
|
||||
"automatic_captions": {
|
||||
"en": _track("https://timedtext/en-translated"),
|
||||
"en-orig": _track("https://timedtext/en-orig"),
|
||||
},
|
||||
}
|
||||
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
|
||||
assert pick.lang == "en-orig"
|
||||
assert pick.url.endswith("en-orig")
|
||||
|
||||
|
||||
def test_translation_is_a_documented_last_resort_not_a_silent_default():
|
||||
"""A German channel under a Spanish/English policy.
|
||||
|
||||
The spoken language is not one the operator asked for, so there is no
|
||||
original to give them and a translation is the only thing on offer. We do
|
||||
take it — but only on the second pass, after every untranslated option has
|
||||
been rejected, and `transcript_lang` records the bare code so the row is
|
||||
distinguishable from an `-orig` one afterwards.
|
||||
|
||||
This case is a deliberate fallback. It is NOT the behaviour that caused the
|
||||
Hormozi bug: there, `en` *was* configured and was being skipped.
|
||||
"""
|
||||
info = {
|
||||
"subtitles": {},
|
||||
"automatic_captions": {
|
||||
"de-orig": _track("https://timedtext/de-orig"),
|
||||
"es": _track("https://timedtext/de?tlang=es"),
|
||||
"en": _track("https://timedtext/de?tlang=en"),
|
||||
},
|
||||
}
|
||||
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
||||
assert pick.lang == "es"
|
||||
assert "tlang=" in pick.url, "the fallback really is a translation; nothing else was available"
|
||||
|
||||
|
||||
# ------------------------------------------------- the tlang= guard
|
||||
#
|
||||
# A translated caption URL is the base track's URL with `tlang=` appended, and
|
||||
# yt-dlp omits it when target == source. That is direct evidence, unlike the
|
||||
# `-orig` naming convention, so it catches videos whose spoken language cannot
|
||||
# be determined any other way.
|
||||
|
||||
|
||||
def test_untranslated_track_wins_even_with_no_orig_key_and_no_language_field():
|
||||
"""The gap the first version of this fix left open.
|
||||
|
||||
Without an `-orig` key and without `info["language"]`, the spoken language
|
||||
is unknown, the reorder cannot fire, and config order used to hand back the
|
||||
Spanish translation. Rejecting `tlang=` needs no such knowledge.
|
||||
"""
|
||||
info = {
|
||||
"subtitles": {},
|
||||
"automatic_captions": {
|
||||
"es": _track("https://timedtext/base?lang=en&kind=asr&tlang=es"),
|
||||
"en": _track("https://timedtext/base?lang=en&kind=asr"),
|
||||
},
|
||||
}
|
||||
assert original_language(info) is None, "precondition: spoken language is undeterminable"
|
||||
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
||||
assert pick.lang == "en"
|
||||
assert "tlang=" not in pick.url
|
||||
|
||||
|
||||
def test_the_real_hormozi_url_shape_is_recognised_as_a_translation():
|
||||
"""Verbatim parameters from the production timedtext URLs in .run/server.log."""
|
||||
info = {
|
||||
"subtitles": {},
|
||||
"automatic_captions": {
|
||||
"es": _track(
|
||||
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
|
||||
"&lang=en&kind=asr&variant=gemini&fmt=json3&tlang=es"
|
||||
),
|
||||
"en-orig": _track(
|
||||
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
|
||||
"&lang=en&kind=asr&variant=gemini&fmt=json3"
|
||||
),
|
||||
},
|
||||
}
|
||||
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
||||
assert pick.lang == "en-orig"
|
||||
assert "tlang=" not in pick.url
|
||||
|
||||
|
||||
def test_manual_captions_still_outrank_auto_for_the_same_language():
|
||||
"""Preferring the original must not override the manual/auto policy."""
|
||||
info = {
|
||||
"subtitles": {"en": _track("https://timedtext/en-manual")},
|
||||
"automatic_captions": {"en-orig": _track("https://timedtext/en-orig")},
|
||||
}
|
||||
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
|
||||
assert pick.source == "manual"
|
||||
assert pick.url.endswith("en-manual")
|
||||
|
||||
|
||||
def test_a_manual_translation_does_not_beat_the_spoken_language():
|
||||
"""Config lists es first, but the video is English with English manual subs."""
|
||||
info = {
|
||||
"subtitles": {"en": _track("https://timedtext/en-manual")},
|
||||
"automatic_captions": {
|
||||
"en-orig": _track("https://timedtext/en-orig"),
|
||||
"es": _track("https://timedtext/es"),
|
||||
},
|
||||
}
|
||||
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
||||
assert pick.lang == "en"
|
||||
assert pick.source == "manual"
|
||||
|
||||
|
||||
# ------------------------------------------------------------- detection
|
||||
|
||||
|
||||
def test_original_language_read_from_the_orig_suffix():
|
||||
assert original_language(_english_video()) == "en"
|
||||
assert original_language(_spanish_video()) == "es"
|
||||
|
||||
|
||||
def test_original_language_falls_back_to_the_info_key():
|
||||
assert original_language({"language": "pt-BR", "automatic_captions": {}}) == "pt"
|
||||
|
||||
|
||||
def test_orig_suffix_wins_over_the_info_key():
|
||||
"""`language` is metadata YouTube localises; the caption list is evidence.
|
||||
|
||||
Production showed YouTube returning English titles for Spanish videos, so
|
||||
localised metadata is not trustworthy for this decision.
|
||||
"""
|
||||
info = {"language": "es", "automatic_captions": {"en-orig": _track("u")}}
|
||||
assert original_language(info) == "en"
|
||||
|
||||
|
||||
def test_original_language_is_none_when_unknowable():
|
||||
assert original_language({"automatic_captions": {"es": _track("u")}}) is None
|
||||
assert original_language({}) is None
|
||||
@@ -0,0 +1,246 @@
|
||||
"""The circuit breaker, tested through the webapp job runner.
|
||||
|
||||
This is the regression test for the incident the whole change exists to
|
||||
prevent: a session gets rate-limited, the runner keeps going anyway, and every
|
||||
remaining video is marked failed. In production that turned one throttling
|
||||
event into 511 `no_subtitles` and 347 `error` rows on a single channel.
|
||||
|
||||
The invariant asserted everywhere below is the same: **videos the run never
|
||||
reached must still be `pending`**, because `pending` is what a later run picks
|
||||
up. A video wrongly marked `error` needs a manual reset first.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.config import Config, DelayConfig
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
from yt_scraper.webapp.jobs import JobManager
|
||||
|
||||
THROTTLED = (
|
||||
"ERROR: [youtube] {vid}: Video unavailable. This content isn't available, try "
|
||||
"again later. The current session has been rate-limited by YouTube for up to an hour."
|
||||
)
|
||||
|
||||
|
||||
def _cfg(tmp_path) -> Config:
|
||||
return Config(
|
||||
database_path=str(tmp_path / "state.db"),
|
||||
output_dir=str(tmp_path / "markdown"),
|
||||
# No real waiting in tests; the breaker's arithmetic is unit-tested
|
||||
# separately in test_ratelimit.py.
|
||||
delay=DelayConfig(
|
||||
min_seconds=0.0, max_seconds=0.0,
|
||||
backoff_base=0.001, backoff_cap=0.002, throttle_threshold=3,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _seed(store: Store, n: int) -> list[str]:
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", n)
|
||||
ids = [f"v{i:02d}" for i in range(n)]
|
||||
store.upsert_videos([
|
||||
VideoRef(v, "UC1", f"Video {i}", f"https://y/watch?v={v}") for i, v in enumerate(ids)
|
||||
])
|
||||
return ids
|
||||
|
||||
|
||||
def _always_throttled(store: Store):
|
||||
"""Stand-in for process_video that fails the way a throttled session does."""
|
||||
|
||||
def fake(row, cfg, st, *args, **kwargs):
|
||||
st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id))
|
||||
return "error"
|
||||
|
||||
return fake
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def manager(tmp_path):
|
||||
store = Store(tmp_path / "state.db")
|
||||
return JobManager(store, _cfg(tmp_path)), store
|
||||
|
||||
|
||||
def test_batch_stops_and_leaves_the_rest_pending(manager, monkeypatch):
|
||||
mgr, store = manager
|
||||
ids = _seed(store, 20)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store))
|
||||
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
|
||||
store.create_job("job", None, {"video_ids": ids})
|
||||
|
||||
mgr._run_batch("job", {"video_ids": ids})
|
||||
|
||||
touched = [v for v in ids if store.get_video(v).status != "pending"]
|
||||
assert len(touched) == 3, "must stop at the threshold, not walk all 20"
|
||||
assert all(store.get_video(v).status == "pending" for v in ids[3:])
|
||||
|
||||
job = store.get_job("job")
|
||||
assert job.status == "error"
|
||||
assert "rate-limit" in (job.last_error or "")
|
||||
|
||||
events = mgr.events_since("job", 0)
|
||||
err = [e for e in events if e["event"] == "error"]
|
||||
assert err and err[-1]["data"]["throttled"] is True
|
||||
|
||||
|
||||
def test_channel_run_stops_and_leaves_the_rest_pending(manager, monkeypatch):
|
||||
mgr, store = manager
|
||||
ids = _seed(store, 20)
|
||||
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", _always_throttled(store))
|
||||
monkeypatch.setattr(
|
||||
"yt_scraper.discover.discover_channel",
|
||||
lambda url, sleep_subrequests=2.0, limit=None: ("UC1", "Alpha", None, []),
|
||||
)
|
||||
store.create_job("job", "UC1", {})
|
||||
|
||||
mgr._run_channel("job", {})
|
||||
|
||||
assert len([v for v in ids if store.get_video(v).status != "pending"]) == 3
|
||||
assert store.get_job("job").status == "error"
|
||||
|
||||
|
||||
def test_a_healthy_run_is_untouched_by_the_breaker(manager, monkeypatch):
|
||||
"""The breaker must be invisible when nothing is throttling."""
|
||||
mgr, store = manager
|
||||
ids = _seed(store, 8)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video",
|
||||
lambda row, *a, **k: store.mark_status(row.video_id, "done") or "done")
|
||||
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
|
||||
store.create_job("job", None, {"video_ids": ids})
|
||||
|
||||
mgr._run_batch("job", {"video_ids": ids})
|
||||
|
||||
assert store.get_job("job").status == "done"
|
||||
assert all(store.get_video(v).status == "done" for v in ids)
|
||||
|
||||
|
||||
def test_ordinary_failures_do_not_stop_the_run(manager, monkeypatch):
|
||||
"""A channel with dead videos must still be processed to the end.
|
||||
|
||||
Without the throttle/permanent distinction, three members-only videos in a
|
||||
row would abort a perfectly healthy scrape.
|
||||
"""
|
||||
mgr, store = manager
|
||||
ids = _seed(store, 10)
|
||||
|
||||
def fake(row, cfg, st, *args, **kwargs):
|
||||
st.mark_error(row.video_id, "ERROR: [youtube] x: Private video. Sign in if you've been granted access")
|
||||
return "error"
|
||||
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", fake)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
|
||||
store.create_job("job", None, {"video_ids": ids})
|
||||
|
||||
mgr._run_batch("job", {"video_ids": ids})
|
||||
|
||||
assert store.get_job("job").status == "done"
|
||||
assert all(store.get_video(v).status == "error" for v in ids)
|
||||
|
||||
|
||||
def test_isolated_throttling_between_successes_does_not_stop_the_run(manager, monkeypatch):
|
||||
"""Only *consecutive* throttling means the session is banned."""
|
||||
mgr, store = manager
|
||||
ids = _seed(store, 12)
|
||||
calls = {"n": 0}
|
||||
|
||||
def fake(row, cfg, st, *args, **kwargs):
|
||||
calls["n"] += 1
|
||||
if calls["n"] % 2:
|
||||
st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id))
|
||||
return "error"
|
||||
st.mark_status(row.video_id, "done")
|
||||
return "done"
|
||||
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", fake)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
|
||||
store.create_job("job", None, {"video_ids": ids})
|
||||
|
||||
mgr._run_batch("job", {"video_ids": ids})
|
||||
|
||||
assert store.get_job("job").status == "done"
|
||||
assert calls["n"] == 12
|
||||
|
||||
|
||||
def test_a_throttled_caption_fetch_is_an_error_not_no_subtitles(tmp_path, monkeypatch):
|
||||
"""Which of the three requests YouTube refused must not decide the status.
|
||||
|
||||
A 429 on `extract_info` produced `error`; a 429 on the separate caption
|
||||
download produced `no_subtitles` — the state that means "this video
|
||||
publishes no captions", which is what the `no_subtitles` counts are read as.
|
||||
Both are retryable, but only one is honest.
|
||||
"""
|
||||
from yt_scraper.extract import VideoData
|
||||
from yt_scraper.pipeline import process_video
|
||||
|
||||
store = Store(tmp_path / "state.db")
|
||||
cfg = _cfg(tmp_path)
|
||||
_seed(store, 1)
|
||||
row = store.get_video("v00")
|
||||
|
||||
throttled = VideoData(
|
||||
info={"id": "v00", "title": "t"},
|
||||
segments=[],
|
||||
subtitle=None,
|
||||
has_chapters=False,
|
||||
skip_reason=(
|
||||
"subtitle track found (lang=en, auto) but the download failed "
|
||||
"[HTTP Error 429: HTTPError: 429 Client Error: Too Many Requests for url: ...]"
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: throttled)
|
||||
assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "error"
|
||||
|
||||
genuinely_absent = VideoData(
|
||||
info={"id": "v00", "title": "t"}, segments=[], subtitle=None,
|
||||
has_chapters=False, skip_reason="no caption tracks published for this video",
|
||||
)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: genuinely_absent)
|
||||
assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "no_subtitles"
|
||||
|
||||
|
||||
def test_metadata_survives_a_failed_caption_fetch(tmp_path, monkeypatch):
|
||||
"""The extraction already paid for this data; a retry must not re-buy it."""
|
||||
from yt_scraper.extract import VideoData
|
||||
from yt_scraper.pipeline import process_video
|
||||
|
||||
store = Store(tmp_path / "state.db")
|
||||
cfg = _cfg(tmp_path)
|
||||
_seed(store, 1)
|
||||
row = store.get_video("v00")
|
||||
|
||||
monkeypatch.setattr(
|
||||
"yt_scraper.pipeline.extract_video",
|
||||
lambda *a, **k: VideoData(
|
||||
info={"id": "v00", "title": "t", "view_count": 4321, "upload_date": "20260101",
|
||||
"thumbnail": "https://i.ytimg.com/x.jpg", "description": "hola"},
|
||||
segments=[], subtitle=None, has_chapters=False,
|
||||
skip_reason="no caption tracks published for this video",
|
||||
),
|
||||
)
|
||||
process_video(row, cfg, store, "Alpha", "UC1", "u")
|
||||
|
||||
after = store.get_video("v00")
|
||||
assert after.status == "no_subtitles"
|
||||
assert after.view_count == 4321, "metadata was discarded with the failed transcript"
|
||||
assert after.upload_date == "20260101"
|
||||
|
||||
|
||||
def test_throttled_videos_stay_retryable(manager, monkeypatch):
|
||||
"""The three videos that did fail must not be classified as permanent.
|
||||
|
||||
`Store.PERMANENT_ERROR_PATTERNS` deliberately excludes the throttling
|
||||
message; if that ever changed, a rate-limit incident would poison rows that
|
||||
a later run could have recovered.
|
||||
"""
|
||||
mgr, store = manager
|
||||
ids = _seed(store, 10)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store))
|
||||
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
|
||||
store.create_job("job", None, {"video_ids": ids})
|
||||
|
||||
mgr._run_batch("job", {"video_ids": ids})
|
||||
|
||||
counts = store.retryable_counts("UC1")
|
||||
assert counts.get("permanent", 0) == 0
|
||||
assert store.reset_videos("UC1", ("error",)) == 3
|
||||
@@ -0,0 +1,319 @@
|
||||
"""Chronological ordering of the videos list.
|
||||
|
||||
The list has to read like a YouTube channel page — newest upload first — and
|
||||
that has to hold for videos nothing has scraped yet. Those have no
|
||||
`upload_date` at all: yt-dlp's flat listing does not report one for YouTube
|
||||
entries, so the date only arrives with the extraction that produces the .md.
|
||||
`channel_seq` (the video's slot in the reverse-chronological /videos tab) is
|
||||
what keeps them in place until then.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.store import NO_DATE_SENTINEL, SCHEMA, SORT_DATE_SQL, Store, VideoRef
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def store(tmp_path):
|
||||
s = Store(tmp_path / "state.db")
|
||||
s.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
||||
return s
|
||||
|
||||
|
||||
def _ref(video_id: str, upload_date: str | None = None, channel_id: str = "UC1") -> VideoRef:
|
||||
return VideoRef(
|
||||
video_id, channel_id, video_id.upper(),
|
||||
f"https://y/watch?v={video_id}", upload_date,
|
||||
)
|
||||
|
||||
|
||||
def _ids(rows) -> list[str]:
|
||||
return [r.video_id for r in rows]
|
||||
|
||||
|
||||
# A channel as discovery hands it over: /videos-tab order, newest first, with
|
||||
# two videos nobody has extracted yet ("a" and "c") interleaved among two that
|
||||
# have been ("b" and "d").
|
||||
MIXED = [_ref("a"), _ref("b", "20240301"), _ref("c"), _ref("d", "20240101")]
|
||||
|
||||
|
||||
def test_default_order_is_newest_upload_first(store):
|
||||
store.upsert_videos([_ref("mid", "20240201"), _ref("new", "20240301"), _ref("old", "20240101")])
|
||||
|
||||
rows, total = store.query_videos()
|
||||
|
||||
assert total == 3
|
||||
assert _ids(rows) == ["new", "mid", "old"]
|
||||
|
||||
|
||||
def test_videos_without_a_markdown_keep_their_slot_in_the_channel(store):
|
||||
"""The regression this exists for: the fallback used to be `discovered_at`,
|
||||
which dates every un-scraped video "today" and floats the whole backlog
|
||||
above videos that really were uploaded this week."""
|
||||
store.upsert_videos(MIXED)
|
||||
|
||||
rows, _ = store.query_videos()
|
||||
|
||||
assert _ids(rows) == ["a", "b", "c", "d"]
|
||||
|
||||
|
||||
def test_an_undated_video_reports_the_date_it_was_sorted_by(store):
|
||||
store.upsert_videos(MIXED)
|
||||
|
||||
by_id = {r.video_id: r for r in store.query_videos()[0]}
|
||||
|
||||
assert by_id["b"].sort_date == "20240301"
|
||||
# "c" has no date of its own, so it borrows the nearest NEWER dated video's.
|
||||
assert by_id["c"].upload_date is None
|
||||
assert by_id["c"].sort_date == "20240301"
|
||||
# Nothing dated sits above "a", so the most it can claim is the newest date
|
||||
# known in its channel. Rank puts it above that video; inventing a later
|
||||
# date would let an un-scraped channel outrank every scraped one.
|
||||
assert by_id["a"].sort_date == "20240301"
|
||||
|
||||
|
||||
def test_a_channel_with_nothing_scraped_does_not_outrank_channels_that_are(store):
|
||||
"""With no dated video anywhere in the channel there is zero evidence about
|
||||
when anything was published. Measured on the real library, one such channel
|
||||
(212 videos, none extracted) took over the whole first page."""
|
||||
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
||||
store.upsert_videos([_ref("known", "20240101", channel_id="UC2")])
|
||||
store.upsert_videos([_ref("u1"), _ref("u2")]) # UC1: nothing dated at all
|
||||
|
||||
rows, _ = store.query_videos()
|
||||
|
||||
assert _ids(rows) == ["known", "u1", "u2"]
|
||||
# ...but the blind channel is still internally in listing order.
|
||||
assert {r.sort_date for r in rows if r.channel_id == "UC1"} == {NO_DATE_SENTINEL}
|
||||
|
||||
|
||||
def test_a_new_upload_ranks_above_the_backlog_it_did_not_refetch(store):
|
||||
"""An incremental sync only fetches the newest window. Everything outside it
|
||||
is older by construction, so the window is ranked above the stored rows
|
||||
instead of restarting from zero and colliding with them."""
|
||||
store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")]) # first full pass
|
||||
store.upsert_videos([_ref("n2"), _ref("n1"), _ref("v3"), _ref("v2")]) # later window
|
||||
|
||||
rows, _ = store.query_videos()
|
||||
|
||||
assert _ids(rows) == ["n2", "n1", "v3", "v2", "v1"]
|
||||
|
||||
|
||||
def test_repeated_syncs_do_not_reshuffle_a_channel(store):
|
||||
store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")])
|
||||
first = _ids(store.query_videos()[0])
|
||||
|
||||
for _ in range(3):
|
||||
store.upsert_videos([_ref("v3"), _ref("v2")]) # same window, nothing new
|
||||
|
||||
assert _ids(store.query_videos()[0]) == first
|
||||
|
||||
|
||||
@pytest.mark.parametrize("sort", ["view_count", "like_count", "duration", "", "nonsense"])
|
||||
def test_sort_keys_the_ui_sends_do_not_degrade_the_chronology(store, sort):
|
||||
"""The web UI sends `view_count`/`like_count`/`duration`; the store only
|
||||
knew `views_desc`/`duration_desc`. Every miss fell through to a raw
|
||||
`upload_date DESC` that ignored the inference and dumped undated videos at
|
||||
the bottom, whatever the channel order said."""
|
||||
store.upsert_videos(MIXED)
|
||||
|
||||
rows, _ = store.query_videos(sort=sort)
|
||||
|
||||
assert _ids(rows) == ["a", "b", "c", "d"]
|
||||
|
||||
|
||||
def test_sorting_by_views_ranks_counted_videos_and_keeps_the_rest_chronological(store):
|
||||
store.upsert_videos(MIXED)
|
||||
store.update_video_metadata("d", view_count=500)
|
||||
store.update_video_metadata("b", view_count=10)
|
||||
|
||||
rows, _ = store.query_videos(sort="view_count")
|
||||
|
||||
assert _ids(rows) == ["d", "b", "a", "c"]
|
||||
|
||||
|
||||
def test_oldest_first_reverses_the_default(store):
|
||||
store.upsert_videos(MIXED)
|
||||
|
||||
rows, _ = store.query_videos(sort="oldest")
|
||||
|
||||
assert _ids(rows) == ["d", "c", "b", "a"]
|
||||
|
||||
|
||||
def test_oldest_first_keeps_unknown_dates_at_the_end_not_the_start(store):
|
||||
""""We do not know when this is from" is not "the beginning of time".
|
||||
|
||||
NO_DATE_SENTINEL sorts below every real date, which is what keeps those rows
|
||||
off the front page under the default sort — and is exactly why they came
|
||||
FIRST under ascending. Measured: "oldest" opened with the 212 videos of a
|
||||
channel nothing had extracted, ahead of a genuine 2017 upload.
|
||||
"""
|
||||
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
||||
store.upsert_videos([_ref("ancient", "20170414", channel_id="UC2")])
|
||||
store.upsert_videos([_ref("unknown1"), _ref("unknown2")]) # UC1: no dates at all
|
||||
|
||||
rows, _ = store.query_videos(sort="oldest")
|
||||
|
||||
assert _ids(rows)[0] == "ancient"
|
||||
assert set(_ids(rows)[1:]) == {"unknown1", "unknown2"}
|
||||
|
||||
|
||||
def test_a_tie_does_not_rank_channels_by_how_big_their_catalogue_is(store):
|
||||
"""`channel_seq` counts up to a channel's video count, so comparing it
|
||||
ACROSS channels ranks by catalogue size. With 92% of adjacent pairs tying on
|
||||
sort_date, that decided most of the list: an entire channel preceded another
|
||||
purely because 575 > 476. Ties group by channel instead, so no channel's
|
||||
internal order is ever interleaved away."""
|
||||
store.upsert_channel("UC2", "@beta", "Beta", 0)
|
||||
# Same date everywhere: the tiebreak decides the whole ordering.
|
||||
store.upsert_videos([_ref(f"big{i}", "20240101") for i in range(5)])
|
||||
store.upsert_videos([_ref(f"small{i}", "20240101", channel_id="UC2") for i in range(2)])
|
||||
|
||||
ids = _ids(store.query_videos()[0])
|
||||
|
||||
big = [i for i, v in enumerate(ids) if v.startswith("big")]
|
||||
small = [i for i, v in enumerate(ids) if v.startswith("small")]
|
||||
assert max(big) < min(small) or max(small) < min(big), f"channels interleaved: {ids}"
|
||||
# and each channel is still in its own listing order
|
||||
assert [v for v in ids if v.startswith("big")] == ["big0", "big1", "big2", "big3", "big4"]
|
||||
assert [v for v in ids if v.startswith("small")] == ["small0", "small1"]
|
||||
|
||||
|
||||
def test_a_row_that_arrives_without_a_rank_is_ranked_on_the_next_open(tmp_path):
|
||||
"""A server still running the pre-column code writes rows with a NULL rank.
|
||||
Left NULL they do not just lose their place — SORT_DATE_SQL makes them
|
||||
inherit the OLDEST date in the channel, so a video discovery has only just
|
||||
found is shown last. Measured on the live database: two brand-new uploads
|
||||
displayed with sort_date 20180720, second to last of 513."""
|
||||
db = tmp_path / "state.db"
|
||||
store = Store(db)
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
|
||||
store.upsert_videos([_ref("older", "20180720"), _ref("oldest", "20180101")])
|
||||
|
||||
with sqlite3.connect(db) as conn: # what the old code path produced
|
||||
conn.execute(
|
||||
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
|
||||
" VALUES ('brand_new', 'UC1', 'Brand new', 'https://y/watch?v=brand_new',"
|
||||
" 'pending', '2026-08-08T13:25:32+00:00')"
|
||||
)
|
||||
|
||||
reopened = Store(db) # migration runs here
|
||||
|
||||
rows, _ = reopened.query_videos()
|
||||
assert _ids(rows) == ["brand_new", "older", "oldest"]
|
||||
assert rows[0].channel_seq is not None
|
||||
assert rows[0].sort_date == "20180720" # newest known in the channel, not the oldest
|
||||
|
||||
|
||||
def test_paging_neither_repeats_nor_loses_rows(store):
|
||||
store.upsert_videos([_ref(f"v{i}") for i in range(7)])
|
||||
|
||||
full = _ids(store.query_videos(size=50)[0])
|
||||
paged = [vid for page in (1, 2, 3) for vid in _ids(store.query_videos(page=page, size=3)[0])]
|
||||
|
||||
assert full == ["v0", "v1", "v2", "v3", "v4", "v5", "v6"]
|
||||
assert paged == full
|
||||
|
||||
|
||||
def test_filters_do_not_disturb_the_order(store):
|
||||
store.upsert_videos(MIXED)
|
||||
store.mark_status("a", "no_subtitles", "no captions")
|
||||
|
||||
rows, total = store.query_videos(channel_id="UC1", status="pending")
|
||||
|
||||
assert total == 3
|
||||
assert _ids(rows) == ["b", "c", "d"]
|
||||
|
||||
|
||||
def test_a_real_date_outranks_a_stale_listing_position(store):
|
||||
"""Why the date is the primary key and the rank only the tiebreak.
|
||||
|
||||
A rank that predates the column was reconstructed offline, not observed from
|
||||
YouTube, so it can be wrong. Sorting by date first means the dates we did
|
||||
pay an extraction to learn CORRECT that reconstruction instead of being
|
||||
overridden by it. Measured against the live listing on Código Espinoza:
|
||||
ordering by rank alone was strictly worse than date-then-rank.
|
||||
"""
|
||||
# The rank claims "old" is the newer of the two; its upload_date says otherwise.
|
||||
store.upsert_videos([_ref("old", "20240101"), _ref("new", "20260801")])
|
||||
|
||||
assert _ids(store.query_videos()[0]) == ["new", "old"]
|
||||
|
||||
|
||||
def test_a_discovery_pass_repairs_a_rank_the_seed_got_wrong(store):
|
||||
"""Two videos published the same day: the date cannot separate them, so the
|
||||
listing rank decides — and a wrong rank shows a wrong order. One ordinary
|
||||
sync re-observes the window from YouTube and fixes it. Measured: Código
|
||||
Espinoza went from 9 inverted pairs to 0 after a single 30-entry pass."""
|
||||
store.upsert_videos([_ref("a", "20260801"), _ref("b", "20260801")])
|
||||
assert _ids(store.query_videos()[0]) == ["a", "b"]
|
||||
|
||||
store.upsert_videos([_ref("b", "20260801"), _ref("a", "20260801")]) # real order
|
||||
|
||||
assert _ids(store.query_videos()[0]) == ["b", "a"]
|
||||
|
||||
|
||||
def test_the_ordering_lookups_stay_on_their_partial_indexes(store):
|
||||
"""Both subqueries run once per undated row, so the plan is the difference
|
||||
between a usable page and an unusable one: measured on the real library,
|
||||
576 ms per page without these indexes and 22 ms with them. SQLite drops a
|
||||
partial index the moment the query's WHERE stops implying the index's, and
|
||||
says nothing about it."""
|
||||
store.upsert_videos(MIXED)
|
||||
sql = (
|
||||
f"SELECT videos.*, {SORT_DATE_SQL} AS sort_date FROM videos "
|
||||
"ORDER BY sort_date DESC, videos.channel_seq DESC, videos.video_id DESC LIMIT 25"
|
||||
)
|
||||
|
||||
with store._cursor() as cur:
|
||||
plan = [str(row["detail"]) for row in cur.execute("EXPLAIN QUERY PLAN " + sql)]
|
||||
|
||||
assert any("COVERING INDEX idx_videos_dated_seq" in line for line in plan), plan
|
||||
assert any("COVERING INDEX idx_videos_dated" in line
|
||||
and "idx_videos_dated_seq" not in line for line in plan), plan
|
||||
|
||||
|
||||
def _legacy_db(path):
|
||||
"""A database written before `channel_seq` existed."""
|
||||
conn = sqlite3.connect(path)
|
||||
conn.executescript(SCHEMA)
|
||||
conn.execute("INSERT INTO channels (channel_id, name) VALUES ('UC1', 'Alpha')")
|
||||
# One discovery batch shares a timestamp and is inserted newest-first...
|
||||
for vid in ("b1", "b2", "b3"):
|
||||
conn.execute(
|
||||
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
|
||||
" VALUES (?, 'UC1', ?, ?, 'pending', '2026-01-01T00:00:00+00:00')",
|
||||
(vid, vid, f"https://y/watch?v={vid}"),
|
||||
)
|
||||
# ...and a later sync can only add videos newer than every one of them.
|
||||
conn.execute(
|
||||
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
|
||||
" VALUES ('later', 'UC1', 'later', 'https://y/watch?v=later', 'pending',"
|
||||
" '2026-02-01T00:00:00+00:00')"
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
|
||||
def test_migration_ranks_a_database_that_predates_the_column(tmp_path):
|
||||
db = tmp_path / "legacy.db"
|
||||
_legacy_db(db)
|
||||
|
||||
rows, _ = Store(db).query_videos()
|
||||
|
||||
assert _ids(rows) == ["later", "b1", "b2", "b3"]
|
||||
assert all(r.channel_seq is not None for r in rows)
|
||||
|
||||
|
||||
def test_migration_seeds_once_and_does_not_reshuffle_on_reopen(tmp_path):
|
||||
db = tmp_path / "legacy.db"
|
||||
_legacy_db(db)
|
||||
first = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]}
|
||||
|
||||
again = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]}
|
||||
|
||||
assert again == first
|
||||
@@ -0,0 +1,290 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import yt_dlp
|
||||
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
from yt_scraper.webapp.jobs import JobManager
|
||||
|
||||
|
||||
def test_audio_job_downloads_into_audio_directory(tmp_path, monkeypatch):
|
||||
calls: list[list[str]] = []
|
||||
|
||||
class FakeYoutubeDL:
|
||||
def __init__(self, options):
|
||||
self.options = options
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False
|
||||
|
||||
def download(self, urls):
|
||||
calls.append(urls)
|
||||
|
||||
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
|
||||
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("abc123", "UC1", "Audio test", "https://www.youtube.com/watch?v=abc123")
|
||||
])
|
||||
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_audio("audio-job", {"mode": "audio", "video_ids": ["abc123"]})
|
||||
|
||||
assert (tmp_path / "audio").is_dir()
|
||||
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
|
||||
assert store.get_job("audio-job").status == "done"
|
||||
|
||||
|
||||
def test_audio_job_with_video_ids_uses_audio_runner(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
calls: list[str] = []
|
||||
monkeypatch.setattr(manager, "_run_audio", lambda job_id, opts: calls.append("audio"))
|
||||
monkeypatch.setattr(manager, "_run_batch", lambda job_id, opts: calls.append("batch"))
|
||||
|
||||
manager._run_job("audio-job")
|
||||
|
||||
assert calls == ["audio"]
|
||||
|
||||
|
||||
def test_video_job_downloads_webm_after_markdown_exists(tmp_path, monkeypatch):
|
||||
calls: list[list[str]] = []
|
||||
captured: dict = {}
|
||||
|
||||
class FakeYoutubeDL:
|
||||
def __init__(self, options):
|
||||
captured.update(options)
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False
|
||||
|
||||
def download(self, urls):
|
||||
calls.append(urls)
|
||||
outtmpl = captured["outtmpl"].replace("%(ext)s", "webm")
|
||||
output = tmp_path / Path(outtmpl).relative_to(tmp_path)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
output.write_bytes(b"webm-test")
|
||||
|
||||
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("abc123", "UC1", "A WebM video", "https://www.youtube.com/watch?v=abc123")
|
||||
])
|
||||
md = tmp_path / "markdown" / "Alpha" / "a-webm-video.md"
|
||||
md.parent.mkdir(parents=True)
|
||||
md.write_text("# A WebM video\n", encoding="utf-8")
|
||||
store.mark_done("abc123", "markdown/Alpha/a-webm-video.md", "en", "manual", False)
|
||||
store.create_job("video-job", None, {"mode": "video", "video_ids": ["abc123"]})
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_video("video-job", {"mode": "video", "video_ids": ["abc123"]})
|
||||
|
||||
row = store.get_video("abc123")
|
||||
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
|
||||
assert captured["merge_output_format"] == "webm"
|
||||
assert captured["js_runtimes"] == {"node": {}}
|
||||
assert "protocol^=m3u8_native" in captured["format"]
|
||||
assert row.video_download_status == "done"
|
||||
assert row.video_filename == "A WebM video.webm"
|
||||
assert row.video_path == "videos/abc123/A WebM video.webm"
|
||||
assert (tmp_path / row.video_path).read_bytes() == b"webm-test"
|
||||
|
||||
|
||||
def test_discovery_job_registers_new_videos_without_processing(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60)
|
||||
])
|
||||
store.mark_status("old", "done")
|
||||
store.create_job("discover-job", "UC1", {"mode": "discover"})
|
||||
calls: list[str] = []
|
||||
|
||||
def fake_discover(url, sleep_subrequests=2.0, limit=None):
|
||||
calls.append((url, limit))
|
||||
return ("UC1", "Alpha", None, [
|
||||
VideoRef("new", "UC1", "New", "https://y/watch?v=new", "20240501", 90),
|
||||
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60),
|
||||
])
|
||||
|
||||
processed = []
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
|
||||
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", lambda *a, **k: processed.append(a))
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_job("discover-job")
|
||||
|
||||
# Windowed, not a full walk: the known "old" video ends the scan in one pass.
|
||||
assert calls == [("https://www.youtube.com/@alpha/videos", 30)]
|
||||
assert store.get_video("new").status == "pending"
|
||||
assert store.get_video("old").status == "done"
|
||||
assert processed == []
|
||||
assert store.get_job("discover-job").status == "done"
|
||||
assert any(
|
||||
event["event"] == "done" and event["data"]["new_videos"] == 1
|
||||
for event in manager.events_since("discover-job", 0)
|
||||
)
|
||||
|
||||
|
||||
def test_all_channel_discovery_continues_after_one_error(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@one", "One", 0)
|
||||
store.upsert_channel("UC2", "@two", "Two", 0)
|
||||
store.create_job("discover-all", None, {"mode": "discover"})
|
||||
|
||||
def fake_discover(url, sleep_subrequests=2.0, limit=None):
|
||||
if "@one" in url:
|
||||
raise RuntimeError("temporary failure")
|
||||
return ("UC2", "Two", None, [VideoRef("new2", "UC2", "New", "https://y/watch?v=new2")])
|
||||
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_job("discover-all")
|
||||
|
||||
assert store.get_video("new2").status == "pending"
|
||||
assert store.get_job("discover-all").status == "done"
|
||||
done = [e for e in manager.events_since("discover-all", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["errors"] == 1
|
||||
|
||||
|
||||
def _catalog_channel(monkeypatch, catalog, calls):
|
||||
def fake(url, sleep_subrequests=2.0, limit=None):
|
||||
calls.append(limit)
|
||||
return ("UC1", "Alpha", None, catalog[:limit] if limit else list(catalog))
|
||||
|
||||
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
|
||||
|
||||
|
||||
def test_discovery_keeps_the_full_video_count_after_a_windowed_pass(tmp_path, monkeypatch):
|
||||
"""The window is 30 wide but the channel has 200 videos — video_count must
|
||||
not collapse to the size of what we just looked at."""
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 200)
|
||||
catalog = [
|
||||
VideoRef(f"v{i:03d}", "UC1", f"V{i}", f"https://y/watch?v=v{i:03d}", None, 60)
|
||||
for i in range(200)
|
||||
]
|
||||
store.upsert_videos(catalog[2:]) # everything except the 2 newest
|
||||
store.create_job("disc", "UC1", {"mode": "discover"})
|
||||
calls: list = []
|
||||
_catalog_channel(monkeypatch, catalog, calls)
|
||||
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
manager._run_job("disc")
|
||||
|
||||
assert calls == [30], "must not paginate all 200"
|
||||
assert store.get_channel("UC1")["video_count"] == 200
|
||||
done = [e for e in manager.events_since("disc", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["new_videos"] == 2
|
||||
assert done["data"]["fetched"] == 30
|
||||
|
||||
|
||||
def test_discovery_full_option_walks_the_whole_channel(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 3)
|
||||
catalog = [
|
||||
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", None, 60) for i in range(3)
|
||||
]
|
||||
store.upsert_videos(catalog)
|
||||
store.create_job("disc-full", "UC1", {"mode": "discover", "full": True})
|
||||
calls: list = []
|
||||
_catalog_channel(monkeypatch, catalog, calls)
|
||||
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
manager._run_job("disc-full")
|
||||
|
||||
assert calls == [None], "full rescan must not pass a playlistend"
|
||||
done = [e for e in manager.events_since("disc-full", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["new_videos"] == 0
|
||||
|
||||
|
||||
def test_batch_job_reports_videos_without_transcripts(tmp_path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("no-captions", "UC1", "No captions", "https://y/watch?v=no-captions")
|
||||
])
|
||||
store.create_job("batch-job", None, {"video_ids": ["no-captions"]})
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *args, **kwargs: "no_subtitles")
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_batch("batch-job", {"video_ids": ["no-captions"]})
|
||||
|
||||
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["processed"] == 0
|
||||
assert done["data"]["no_subtitles"] == 1
|
||||
assert store.get_video("no-captions").markdown_path is None
|
||||
|
||||
|
||||
def test_batch_job_downloads_markdown_and_nothing_else(tmp_path, monkeypatch):
|
||||
"""The .md button downloads notes only.
|
||||
|
||||
Thumbnails are already cached by the channel-level paths (add-channel, the
|
||||
Thumbnails tool) and /api/thumbnails/{id} redirects to the CDN for whatever
|
||||
is missing, so fetching one per video here spent a request on an image the
|
||||
UI could already display.
|
||||
"""
|
||||
store = Store(tmp_path / "state.db")
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([VideoRef("vid1", "UC1", "One", "https://y/watch?v=vid1")])
|
||||
store.create_job("batch-job", None, {"video_ids": ["vid1"]})
|
||||
|
||||
thumbs: list[str] = []
|
||||
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *a, **k: "done")
|
||||
monkeypatch.setattr(
|
||||
"yt_scraper.pipeline.cache_thumbnail",
|
||||
lambda _store, video_id, _dir: bool(thumbs.append(video_id)),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"yt_scraper._yt_http.yt_get",
|
||||
lambda *a, **k: pytest.fail("a .md batch must not fetch images"),
|
||||
)
|
||||
manager = JobManager(
|
||||
store,
|
||||
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
|
||||
)
|
||||
|
||||
manager._run_batch("batch-job", {"video_ids": ["vid1"]})
|
||||
|
||||
assert thumbs == []
|
||||
assert not (tmp_path / "thumbnails").exists()
|
||||
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
|
||||
assert done["data"]["processed"] == 1
|
||||
Reference in New Issue
Block a user