wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)

This commit is contained in:
urieljareth
2026-08-22 19:37:23 -06:00
parent 4f5a68b572
commit 8a59b39c98
103 changed files with 70954 additions and 1825 deletions
+143
View File
@@ -0,0 +1,143 @@
"""Members-only / gated videos: identified from discovery, labelled, and kept
out of bulk work without becoming unreachable.
yt-dlp's flat listing reports availability="subscriber_only" for members-only
videos, so they are knowable before an extraction attempt is ever spent on them.
Measured on a real channel: 120 flat entries in one request, exactly 4 flagged,
matching exactly the 4 the DB had learned about the expensive way.
"""
from __future__ import annotations
import pytest
from yt_scraper.store import BLOCKING_AVAILABILITY, Store, VideoRef
def _store(tmp_path) -> Store:
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
return store
def _ref(vid: str, availability: str | None = None) -> VideoRef:
return VideoRef(vid, "UC1", vid, f"https://www.youtube.com/watch?v={vid}", availability=availability)
# ---------------------------------------------------------------- classification
def test_availability_from_discovery_is_stored_and_classified(tmp_path):
store = _store(tmp_path)
store.upsert_videos([_ref("gated", "subscriber_only"), _ref("open", "public")])
assert store.get_video("gated").availability == "subscriber_only"
assert store.get_video("gated").block_reason == "members_only"
assert store.get_video("open").block_reason is None
@pytest.mark.parametrize("availability,expected", [
("subscriber_only", "members_only"),
("premium_only", "premium_only"),
("private", "private"),
("needs_auth", "needs_auth"),
])
def test_every_blocking_availability_maps_to_a_reason(tmp_path, availability, expected):
store = _store(tmp_path)
store.upsert_videos([_ref("v", availability)])
assert store.get_video("v").block_reason == expected
@pytest.mark.parametrize("availability", ["public", "unlisted", None])
def test_fetchable_availability_is_never_blocked(tmp_path, availability):
"""`unlisted` downloads perfectly well — mislabelling it would hide videos
the user can actually have."""
store = _store(tmp_path)
store.upsert_videos([_ref("v", availability)])
assert store.get_video("v").block_reason is None
assert "unlisted" not in BLOCKING_AVAILABILITY
def test_block_reason_falls_back_to_the_recorded_error(tmp_path):
"""Rows burned into `error` before availability was captured must still be
identifiable without re-fetching them."""
store = _store(tmp_path)
store.upsert_videos([_ref("old")])
store.mark_error("old", "ERROR: [youtube] x: Join this channel to get access to members-only content")
assert store.get_video("old").availability is None
assert store.get_video("old").block_reason == "members_only"
def test_rate_limit_error_is_not_a_block_reason(tmp_path):
store = _store(tmp_path)
store.upsert_videos([_ref("t")])
store.mark_error("t", "ERROR: Video unavailable. The current session has been rate-limited by YouTube")
assert store.get_video("t").block_reason is None
def test_rediscovery_does_not_wipe_a_known_availability(tmp_path):
store = _store(tmp_path)
store.upsert_videos([_ref("v", "subscriber_only")])
store.upsert_videos([_ref("v", None)]) # a later flat pass omitted the field
assert store.get_video("v").availability == "subscriber_only"
def test_set_availability_records_what_extraction_learned(tmp_path):
store = _store(tmp_path)
store.upsert_videos([_ref("v")])
store.set_availability("v", "subscriber_only")
assert store.get_video("v").block_reason == "members_only"
store.set_availability("v", None) # must not clear it
assert store.get_video("v").availability == "subscriber_only"
# ---------------------------------------------------------------- queue behaviour
def test_blocked_videos_are_kept_out_of_the_pending_queue(tmp_path):
"""Bulk runs must not spend requests on videos that cannot be fetched."""
store = _store(tmp_path)
store.upsert_videos([_ref("gated", "subscriber_only"), _ref("ok"), _ref("unlisted", "unlisted")])
ids = {v.video_id for v in store.get_pending("UC1")}
assert ids == {"ok", "unlisted"}
all_ids = {v.video_id for v in store.get_pending("UC1", include_blocked=True)}
assert all_ids == {"ok", "unlisted", "gated"}
def test_a_blocked_video_is_still_reachable_by_id(tmp_path):
"""Buying the membership must not leave the video permanently stranded."""
store = _store(tmp_path)
store.upsert_videos([_ref("gated", "subscriber_only")])
assert store.get_video("gated") is not None, "explicit per-video processing still works"
# ---------------------------------------------------------------- filtering
def test_blocked_filter_finds_them_across_statuses(tmp_path):
store = _store(tmp_path)
store.upsert_videos([
_ref("gated", "subscriber_only"),
_ref("legacy"),
_ref("fine"),
])
store.mark_error("legacy", "ERROR: Join this channel to get access to members-only content")
store.mark_status("fine", "no_subtitles")
rows, total = store.query_videos(blocked=True)
assert {r.video_id for r in rows} == {"gated", "legacy"}
assert total == 2
rows, total = store.query_videos(blocked=False)
assert {r.video_id for r in rows} == {"fine"}
def test_blocked_filter_absent_means_everything(tmp_path):
store = _store(tmp_path)
store.upsert_videos([_ref("gated", "subscriber_only"), _ref("fine")])
_rows, total = store.query_videos()
assert total == 2
+75
View File
@@ -0,0 +1,75 @@
"""Tests for the YAML config loader.
Focuses on the ``languages`` and ``prefer_manual`` fields, including the
legacy/compat behaviour. Network-free, DB-free.
"""
from __future__ import annotations
import textwrap
import pytest
from yt_scraper.config import load_config, parse_languages
def _write_config(tmp_path, body: str) -> str:
p = tmp_path / "config.yaml"
p.write_text(textwrap.dedent(body), encoding="utf-8")
return str(p)
def test_load_legacy_languages_list(tmp_path) -> None:
p = _write_config(tmp_path, """
channel_url: "https://example.com/@x/videos"
languages: ["es", "en"]
prefer_manual: true
""")
cfg = load_config(p)
assert cfg.languages == {"es": "manual", "en": "manual"}
def test_load_legacy_languages_list_with_prefer_manual_false(tmp_path) -> None:
p = _write_config(tmp_path, """
languages: ["es", "en"]
prefer_manual: false
""")
cfg = load_config(p)
assert cfg.languages == {"es": "auto", "en": "auto"}
def test_load_new_dict_languages(tmp_path) -> None:
p = _write_config(tmp_path, """
languages:
en: manual
es: auto
pt: any
prefer_manual: false
""")
cfg = load_config(p)
assert cfg.languages == {"en": "manual", "es": "auto", "pt": "any"}
# prefer_manual remains accessible for fallback on `"any"` entries
assert cfg.prefer_manual is False
def test_load_unknown_mode_normalises_to_any(tmp_path) -> None:
p = _write_config(tmp_path, """
languages:
en: garbage
""")
cfg = load_config(p)
assert cfg.languages == {"en": "any"}
def test_load_missing_languages_defaults_to_any_not_manual_only(tmp_path) -> None:
"""The default must fall back to auto captions.
A manual-only default silently produces zero transcripts on the many
channels that publish only auto-generated captions, and records them as
`no_subtitles` — a terminal status that hides a purely configural failure.
"""
p = _write_config(tmp_path, """
channel_url: "https://example.com/@x/videos"
""")
cfg = load_config(p)
assert "es" in cfg.languages and "en" in cfg.languages
assert cfg.languages["es"] == "any"
+157
View File
@@ -0,0 +1,157 @@
"""Windowed channel sync: only fetch what is newer than what we already have.
The whole point is request economy against YouTube, so these tests assert on the
`limit` values handed to yt-dlp (which become `playlistend`, i.e. how many
continuation pages get requested), not just on the refs that come back.
"""
from __future__ import annotations
import pytest
from yt_scraper.discover import discover_incremental
from yt_scraper.store import VideoRef
def _ref(vid: str, upload_date: str | None = None, url: str | None = None) -> VideoRef:
return VideoRef(
video_id=vid,
channel_id="UC1",
title=vid,
url=url or f"https://www.youtube.com/watch?v={vid}",
upload_date=upload_date,
)
def _fake_channel(monkeypatch, catalog: list[VideoRef], calls: list | None = None):
"""Patch discover_channel with a channel whose /videos tab is `catalog`
(newest first) and that honours `limit` the way playlistend does."""
def fake(url, sleep_subrequests=2.0, limit=None):
if calls is not None:
calls.append(limit)
return ("UC1", "Alpha", "http://avatar", catalog[:limit] if limit else list(catalog))
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
def test_stops_at_first_run_of_known_videos(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(500)]
known = {r.video_id for r in catalog[5:]} # everything except the 5 newest
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
assert calls == [30], "one pass only — must not paginate the whole channel"
assert result.fetched == 30
assert [r.video_id for r in result.new_refs] == [f"v{i:03d}" for i in range(5)]
assert result.caught_up is True
assert result.full_scan is False
def test_widens_window_when_the_whole_window_is_new(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(500)]
known = {r.video_id for r in catalog[100:]} # 100 new uploads since last sync
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", known, window=30, overlap=3)
# 30 -> 60 -> 120: doubles only as far as needed, never the full 500.
assert calls == [30, 60, 120]
assert result.new_count == 100
assert result.caught_up is True
def test_gives_up_at_max_window_and_says_so(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(500)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental(
"https://y/@alpha/videos", {"not-in-this-channel"}, window=10, max_window=40, overlap=3
)
assert calls == [10, 20, 40]
assert result.caught_up is False, "caller must be able to tell the scan was truncated"
assert result.fetched == 40
def test_no_local_history_walks_the_whole_channel(monkeypatch):
catalog = [_ref(f"v{i:03d}") for i in range(120)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", set(), window=30)
assert calls == [None], "a first sync has no boundary to stop at"
assert result.full_scan is True
assert result.new_count == 120
def test_short_channel_is_exhausted_in_one_pass(monkeypatch):
catalog = [_ref("a"), _ref("b")]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=30)
assert calls == [30]
assert result.exhausted is True
assert result.caught_up is True
assert [r.video_id for r in result.new_refs] == ["a"]
def test_keep_filter_applies_before_the_overlap_check(monkeypatch):
"""Shorts the store never recorded must not keep the window widening."""
catalog = [
_ref("s1", url="https://www.youtube.com/shorts/s1"),
_ref("s2", url="https://www.youtube.com/shorts/s2"),
_ref("s3", url="https://www.youtube.com/shorts/s3"),
_ref("known1"),
_ref("known2"),
_ref("known3"),
] + [_ref(f"v{i}") for i in range(50)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental(
"https://y/@alpha/videos",
{"known1", "known2", "known3"},
window=6,
overlap=3,
keep=lambda r: "/shorts/" not in (r.url or ""),
)
assert calls == [6], "the three known long-form videos end the scan"
assert result.new_refs == []
def test_upload_date_cutoff_stops_the_scan_when_dates_are_available(monkeypatch):
catalog = [
_ref("n1", "20260701"),
_ref("n2", "20260630"),
_ref("o1", "20250101"),
_ref("o2", "20241231"),
] + [_ref(f"old{i}", "20200101") for i in range(50)]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental(
"https://y/@alpha/videos", {"unrelated"}, window=4, overlap=2, since="20260601"
)
assert calls == [4], "entries older than the watermark end the scan"
assert result.caught_up is True
@pytest.mark.parametrize("window,overlap", [(0, 0), (-5, -1)])
def test_degenerate_settings_are_clamped(monkeypatch, window, overlap):
catalog = [_ref("a"), _ref("b")]
calls: list = []
_fake_channel(monkeypatch, catalog, calls)
result = discover_incremental("https://y/@alpha/videos", {"b"}, window=window, overlap=overlap)
assert calls and calls[0] >= 1
assert result.fetched >= 1
+154
View File
@@ -0,0 +1,154 @@
"""Tests for per-language subtitle selection and dict-typed ``Config.languages``.
These tests are pure: no network, no yt-dlp, no DB. They cover the policy
logic that decides which subtitle track to pick for a given
``info`` dict produced by yt-dlp, plus the legacy/back-compat shims that
let older YAML configs still load.
"""
from __future__ import annotations
import pytest
from yt_scraper.config import Config, parse_languages
from yt_scraper.extract import pick_subtitle
# ---------- fixtures ----------
def _track(ext: str = "json3", url: str = "u") -> dict:
return {"ext": ext, "url": url}
def _info(manual: dict | None = None, auto: dict | None = None) -> dict:
return {"subtitles": manual or {}, "automatic_captions": auto or {}}
# ---------- pick_subtitle: mixed per-language mode ----------
def test_mixed_manual_and_auto_per_language() -> None:
"""`es: auto` only takes auto; iterating `es` first means it wins.
With ``{"es": "auto", "en": "manual"}`` the iteration matches `es`
against the auto dict and picks the auto track; for `en` only the
manual dict is consulted and the manual track is returned. The two
policies do NOT cross-pollinate between languages.
"""
info = _info(
manual={"en": [_track(url="man-en")]},
auto={"en": [_track(url="auto-en")], "es": [_track(url="auto-es")]},
)
pick = pick_subtitle(info, {"es": "auto", "en": "manual"}, prefer_manual=True)
assert pick is not None
assert pick.lang == "es"
assert pick.source == "auto"
def test_manual_only_skips_auto_even_when_present() -> None:
info = _info(manual={}, auto={"es": [_track(url="auto")]})
pick = pick_subtitle(info, {"es": "manual"}, prefer_manual=True)
assert pick is None
def test_auto_only_skips_manual_even_when_present() -> None:
info = _info(manual={"es": [_track(url="man")]}, auto={})
pick = pick_subtitle(info, {"es": "auto"}, prefer_manual=True)
assert pick is None
def test_any_defer_to_prefer_manual_default_true() -> None:
"""`any` honours the legacy prefer_manual=True default."""
info = _info(
manual={"es": [_track(url="man")]},
auto={"es": [_track(url="auto")]},
)
pick = pick_subtitle(info, {"es": "any"}, prefer_manual=True)
assert pick is not None
assert pick.source == "manual"
def test_any_defer_to_prefer_manual_default_false() -> None:
info = _info(
manual={"es": [_track(url="man")]},
auto={"es": [_track(url="auto")]},
)
pick = pick_subtitle(info, {"es": "any"}, prefer_manual=False)
assert pick is not None
assert pick.source == "auto"
def test_picks_best_format_within_track() -> None:
info = _info(manual={"en": [_track(ext="ttml", url="t"), _track(ext="json3", url="j")]})
pick = pick_subtitle(info, {"en": "manual"}, prefer_manual=True)
assert pick.url == "j"
# ---------- language normalisation ----------
def test_language_base_match() -> None:
"""`es-419` preference matches the `es` caption track."""
info = _info(manual={"es": [_track(url="u")]})
pick = pick_subtitle(info, {"es-419": "manual"}, prefer_manual=True)
assert pick is not None
assert pick.lang == "es"
# ---------- legacy list input ----------
def test_legacy_list_input_uses_prefer_manual() -> None:
info = _info(
manual={"es": [_track(url="m")]},
auto={"es": [_track(url="a")]},
)
pick = pick_subtitle(info, ["es"], prefer_manual=True)
assert pick.source == "manual"
pick = pick_subtitle(info, ["es"], prefer_manual=False)
assert pick.source == "auto"
def test_unknown_mode_falls_back_to_any() -> None:
info = _info(manual={"es": [_track()]}, auto={"es": [_track()]})
pick = pick_subtitle(info, {"es": "garbage"}, prefer_manual=True)
assert pick is not None
assert pick.source == "manual"
# ---------- parse_languages config helper ----------
def test_parse_languages_dict_pass_through() -> None:
assert parse_languages({"en": "manual", "es": "auto"}, True) == {
"en": "manual", "es": "auto",
}
def test_parse_languages_dict_unknown_mode_to_any() -> None:
assert parse_languages({"en": "garbage"}, True) == {"en": "any"}
def test_parse_languages_legacy_list_to_manual_by_default() -> None:
assert parse_languages(["es", "en"], prefer_manual=True) == {
"es": "manual", "en": "manual",
}
def test_parse_languages_legacy_list_to_auto_when_prefer_manual_false() -> None:
assert parse_languages(["es", "en"], prefer_manual=False) == {
"es": "auto", "en": "auto",
}
def test_parse_languages_none_returns_empty() -> None:
assert parse_languages(None, True) == {}
# ---------- Config dataclass default shape ----------
def test_config_languages_defaults_to_dict() -> None:
cfg = Config()
assert isinstance(cfg.languages, dict)
# "any", not "manual": the default must not exclude auto-generated captions.
assert cfg.languages == {"es": "any", "en": "any"}
def test_config_prefer_manual_defaults_true() -> None:
cfg = Config()
assert cfg.prefer_manual is True
+118
View File
@@ -0,0 +1,118 @@
"""One video must own exactly one .md, whatever happens to its title.
The filename is derived from the title, and titles are not stable: YouTube
serves them localised, so the same video came back as "La controversia de
Claude Fable 5" on one pass and "The Claude Fable controversy 5" on the next.
Creators also simply rename videos.
`re_render_videos` already deleted the superseded file; `process_video` did not,
so a re-scrape after a title change left the old file orphaned on disk. The DB
repointed, the stale file stayed, and every later scan had to wade through it —
the same shape as the incident that left 94 files for 61 rows.
The identity that matters is the video id, which never changes. These tests pin
that: the row's `markdown_path` is authoritative, and anything it used to point
at gets cleaned up.
"""
from __future__ import annotations
from pathlib import Path
import pytest
from yt_scraper.config import Config
from yt_scraper.extract import SubtitlePick, VideoData
from yt_scraper.parse import Segment
from yt_scraper.pipeline import process_video
from yt_scraper.render import build_filename_stem
from yt_scraper.store import Store, VideoRef
@pytest.fixture
def env(tmp_path):
cfg = Config(
database_path=str(tmp_path / "state.db"),
output_dir=str(tmp_path / "markdown"),
template_path="templates/video.md.j2",
)
store = Store(cfg.database_path_resolved)
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([VideoRef("vid123", "UC1", "t", "https://y/watch?v=vid123", "20260101", 60)])
return cfg, store, tmp_path
def _data(title: str) -> VideoData:
return VideoData(
info={"id": "vid123", "title": title, "upload_date": "20260101", "channel": "Alpha"},
segments=[Segment(start=0.0, end=2.0, text="hello")],
subtitle=SubtitlePick(url="u", ext="json3", lang="en-orig", source="auto"),
has_chapters=False,
)
def _md_files(root: Path) -> list[str]:
return sorted(p.name for p in root.rglob("*.md"))
def test_a_retitled_video_does_not_leave_a_second_file(env, monkeypatch):
"""The production case: the same video, title localised differently."""
cfg, store, tmp_path = env
md_root = Path(cfg.output_dir_resolved)
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
lambda *a, **k: _data("La controversia de Claude Fable 5"))
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
first = _md_files(md_root)
assert len(first) == 1
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
lambda *a, **k: _data("The Claude Fable controversy 5"))
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
after = _md_files(md_root)
assert len(after) == 1, f"one video, {len(after)} files on disk: {after}"
# And the DB points at the one that exists.
row = store.get_video("vid123")
assert (Path(cfg.output_dir_resolved).parent / row.markdown_path).exists()
assert Path(row.markdown_path).name == after[0]
def test_rescraping_an_unchanged_video_is_idempotent(env, monkeypatch):
cfg, store, tmp_path = env
md_root = Path(cfg.output_dir_resolved)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("Same Title"))
for _ in range(3):
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
assert len(_md_files(md_root)) == 1
def test_changing_the_filename_template_relocates_rather_than_duplicates(env, monkeypatch):
"""Opting into ids in the filename must not strand the old files."""
cfg, store, tmp_path = env
md_root = Path(cfg.output_dir_resolved)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("A Title"))
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
cfg.filename_template = "{upload_date}_{slug}_{video_id}"
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
files = _md_files(md_root)
assert len(files) == 1, f"template change duplicated the file: {files}"
assert "vid123" in files[0]
# ------------------------------------------------------------ template vars
def test_video_id_is_available_to_the_filename_template():
"""Lets an operator make the file self-identifying without the DB."""
stem = build_filename_stem(
"20260101", "Some Title", template="{upload_date}_{slug}_{video_id}", video_id="abc123XYZ_-"
)
assert stem == "20260101_some-title_abc123XYZ_-"
def test_default_template_is_unchanged():
"""Existing libraries keep their filenames; adding the variable is opt-in."""
assert build_filename_stem("20260101", "Some Title", video_id="abc123") == "20260101_some-title"
+246
View File
@@ -0,0 +1,246 @@
"""Unit tests for the politeness primitives.
Nothing here sleeps for real: `Pacer` takes an injectable clock and sleeper, and
`backoff_delay` returns the delay instead of consuming it. A test suite that
actually waited would be the first thing anyone deleted.
"""
from __future__ import annotations
import pytest
from yt_scraper.ratelimit import (
Pacer,
ThrottleGuard,
backoff_delay,
is_quota_exhausted,
is_rate_limited,
ydl_throttle_opts,
)
# Verbatim from the production database, where 343 of 350 `error` rows carried
# one of these. If the detector stops matching them the circuit breaker becomes
# decorative, so they are pinned here rather than paraphrased.
REAL_THROTTLE_MESSAGES = [
"ERROR: [youtube] abcdefghijk: Video unavailable. This content isn't available, "
"try again later. The current session has been rate-limited by YouTube for up to an hour.",
"ERROR: [youtube] abcdefghijk: This content isn't available, try again later. "
"The current session has been rate-limited by YouTube for up to an hour. It is recommended t",
"HTTPError: 429 Client Error: Too Many Requests for url: https://www.youtube.com/api/timedtext",
"ERROR: [youtube] xyz: Sign in to confirm you're not a bot",
"HTTP Error 429: Too Many Requests",
]
# Equally verbatim: these are permanent, must NOT trip the breaker, and must
# stay distinguishable from throttling.
REAL_PERMANENT_MESSAGES = [
"ERROR: [youtube] abcdefghijk: Join this channel to get access to members-only "
"content like this video, and other exclusive perks.",
"ERROR: [youtube] abcdefghijk: Private video. Sign in if you've been granted access to this video",
"ERROR: [youtube] abcdefghijk: This video has been removed by the uploader",
"no caption tracks published for this video",
"subtitle downloaded but parsed empty (lang=es, format=json3)",
]
@pytest.mark.parametrize("msg", REAL_THROTTLE_MESSAGES)
def test_detects_real_throttle_messages(msg):
assert is_rate_limited(msg) is True
@pytest.mark.parametrize("msg", REAL_PERMANENT_MESSAGES)
def test_ignores_permanent_failures(msg):
assert is_rate_limited(msg) is False
def test_quota_is_not_treated_as_plain_throttling():
"""Google documents quota exhaustion as daily; backing off cannot fix it."""
assert is_quota_exhausted("403 quotaExceeded") is True
assert is_quota_exhausted("dailyLimitExceeded") is True
assert is_quota_exhausted("rate-limited by YouTube") is False
def test_rate_limited_accepts_exception_objects():
assert is_rate_limited(RuntimeError("HTTP Error 429: Too Many Requests")) is True
# Both spellings occur, and they are different strings. yt-dlp raises
# "HTTP Error 429: ..."; `requests` raises "429 Client Error: ... for url: ...",
# which the project wraps as "HTTPError: 429 ...". Detection used to rely on the
# prose for the second form, so a 429 with no reason phrase — routine over
# HTTP/2 — went unnoticed and the breaker never counted it.
@pytest.mark.parametrize(
"msg",
[
"HTTP Error 429: Too Many Requests",
"HTTPError: 429 Client Error: Too Many Requests for url: https://youtube.com/api/timedtext",
"HTTP Error 429: HTTPError: 429 Client Error: for url: https://youtube.com/api/timedtext",
"429 Client Error: for url: https://www.youtube.com/api/timedtext?v=x",
"HTTP Error 408: Request Timeout",
"HTTPError: 408 Client Error: Request Timeout for url: https://youtube.com/",
],
)
def test_detects_the_status_code_without_relying_on_the_reason_phrase(msg):
assert is_rate_limited(msg) is True, f"undetected throttle: {msg}"
@pytest.mark.parametrize(
"msg",
[
"HTTP Error 404: Not Found",
"HTTPError: 403 Client Error: Forbidden for url: https://youtube.com/",
"HTTP Error 500: Internal Server Error",
"no caption tracks published for this video",
# A bare number must not be read as a status code.
"video 429 seconds long with 408 segments",
],
)
def test_does_not_treat_other_statuses_as_throttling(msg):
assert is_rate_limited(msg) is False, f"false positive: {msg}"
# ---------------------------------------------------------------- backoff
def test_backoff_grows_and_is_capped():
delays = [backoff_delay(n, base=2.0, cap=60.0) for n in range(8)]
# Jitter is < 1s so successive doublings still order strictly until the cap.
assert delays[0] < delays[1] < delays[2] < delays[3]
assert all(d <= 60.0 for d in delays)
assert delays[-1] == 60.0
def test_backoff_jitters():
"""Same attempt must not produce the same delay twice, or concurrent
clients would re-synchronise into waves — the reason Google mandates it."""
seen = {backoff_delay(2, base=2.0, cap=60.0) for _ in range(30)}
assert len(seen) > 1
def test_backoff_survives_a_runaway_counter():
assert backoff_delay(10_000, base=2.0, cap=60.0) == 60.0
assert backoff_delay(-5, base=2.0, cap=60.0) <= 3.0
# ---------------------------------------------------------------- pacer
class FakeClock:
def __init__(self):
self.now = 1000.0
self.slept: list[float] = []
def time(self) -> float:
return self.now
def sleep(self, seconds: float) -> None:
self.slept.append(seconds)
self.now += seconds
def test_pacer_spaces_calls():
clock = FakeClock()
pacer = Pacer(2.0, clock=clock.time, sleeper=clock.sleep)
assert pacer.wait() == 0.0 # first call is free
assert pacer.wait() == pytest.approx(2.0)
assert pacer.wait() == pytest.approx(2.0)
assert clock.slept == [2.0, 2.0]
def test_pacer_does_not_charge_for_time_already_spent():
"""A caller slower than the interval should never wait on top of its own work."""
clock = FakeClock()
pacer = Pacer(2.0, clock=clock.time, sleeper=clock.sleep)
pacer.wait()
clock.now += 10.0 # the request itself took 10s
assert pacer.wait() == 0.0
assert clock.slept == []
def test_pacer_disabled_by_default_interval():
clock = FakeClock()
pacer = Pacer(0.0, clock=clock.time, sleeper=clock.sleep)
assert [pacer.wait() for _ in range(5)] == [0.0] * 5
assert clock.slept == []
def test_pacer_charges_for_multi_request_callers():
"""One extract_info is two HTTP requests; billing it as one halves the budget."""
clock = FakeClock()
pacer = Pacer(2.0, clock=clock.time, sleeper=clock.sleep)
pacer.wait(cost=2) # first call still free...
assert pacer.wait() == pytest.approx(4.0) # ...but it reserved two slots
def test_penalise_pushes_the_next_slot_out():
clock = FakeClock()
pacer = Pacer(1.0, clock=clock.time, sleeper=clock.sleep)
pacer.wait()
pacer.penalise(30.0)
assert pacer.wait() == pytest.approx(30.0)
# ---------------------------------------------------------------- guard
def _guard() -> ThrottleGuard:
return ThrottleGuard(threshold=3, base=0.01, cap=0.05)
def test_guard_trips_after_consecutive_throttling():
g = _guard()
assert g.note_failure(REAL_THROTTLE_MESSAGES[0]) > 0
assert not g.tripped
g.note_failure(REAL_THROTTLE_MESSAGES[0])
assert not g.tripped
g.note_failure(REAL_THROTTLE_MESSAGES[0])
assert g.tripped
assert "consecutive" in (g.tripped_reason or "")
def test_success_resets_the_streak():
"""Isolated throttled videos between successes are noise, not a banned session."""
g = _guard()
for _ in range(10):
g.note_failure(REAL_THROTTLE_MESSAGES[0])
g.note_success()
assert not g.tripped
assert g.throttled_total == 10
def test_permanent_failures_never_trip_the_breaker():
"""A channel with a few members-only videos must not look like a ban."""
g = _guard()
for msg in REAL_PERMANENT_MESSAGES * 5:
g.note_failure(msg)
assert not g.tripped
assert g.throttled_total == 0
def test_mixed_failures_do_not_accumulate_into_a_trip():
g = _guard()
g.note_failure(REAL_THROTTLE_MESSAGES[0])
g.note_failure("Private video")
g.note_failure(REAL_THROTTLE_MESSAGES[0])
g.note_failure("no caption tracks published for this video")
g.note_failure(REAL_THROTTLE_MESSAGES[0])
assert not g.tripped
def test_quota_trips_immediately_without_backoff():
g = _guard()
assert g.note_failure("403 quotaExceeded") == 0.0
assert g.tripped
assert "quota" in (g.tripped_reason or "").lower()
# ---------------------------------------------------------------- ydl opts
def test_throttle_opts_use_names_yt_dlp_actually_reads():
opts = ydl_throttle_opts(2.5, extractor_retries=4, socket_timeout=15.0)
assert opts["sleep_interval_requests"] == 2.5
assert opts["extractor_retries"] == 4
assert opts["socket_timeout"] == 15.0
# The bug this whole module exists to prevent.
assert "sleep_subrequests" not in opts
+356
View File
@@ -0,0 +1,356 @@
"""Recovery paths: retryable statuses, recorded skip reasons, disk<->DB reconcile.
The motivating incident: 511 videos were stored as `no_subtitles` because the
language policy was manual-only while the channel publishes only auto-generated
captions. `reset_errors` could not reach them, and nothing recorded why they
were skipped, so the failure was both invisible and irreversible.
"""
from __future__ import annotations
from pathlib import Path
from yt_scraper.extract import describe_missing_subtitle
from yt_scraper.segments import reconcile_markdown
from yt_scraper.store import Store, VideoRef
def _store(tmp_path) -> Store:
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
return store
# ---------------------------------------------------------------- reset
def test_reset_reaches_no_subtitles_not_just_error(tmp_path):
store = _store(tmp_path)
store.upsert_videos([
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
VideoRef("c", "UC1", "C", "https://y/watch?v=c"),
])
store.mark_status("a", "no_subtitles", "policy rejected auto captions")
store.mark_error("b", "rate limited")
store.mark_done("c", "markdown/c.md", "es", "auto", False)
n = store.reset_videos("UC1", ("error", "no_subtitles"))
assert n == 2
assert store.get_video("a").status == "pending"
assert store.get_video("b").status == "pending"
assert store.get_video("c").status == "done", "finished work must not be re-queued"
def test_reset_never_touches_done_even_if_asked(tmp_path):
store = _store(tmp_path)
store.upsert_videos([VideoRef("c", "UC1", "C", "https://y/watch?v=c")])
store.mark_done("c", "markdown/c.md", "es", "auto", False)
assert store.reset_videos("UC1", ("done",)) == 0
assert store.get_video("c").status == "done"
def test_reset_errors_still_only_resets_errors(tmp_path):
"""The narrower legacy helper keeps its old meaning."""
store = _store(tmp_path)
store.upsert_videos([
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
])
store.mark_status("a", "no_subtitles")
store.mark_error("b", "boom")
assert store.reset_errors("UC1") == 1
assert store.get_video("a").status == "no_subtitles"
assert store.get_video("b").status == "pending"
def test_permanent_failures_are_excluded_from_retries(tmp_path):
"""Members-only videos cannot be fixed by retrying; re-running them only
spends requests the recoverable videos need."""
store = _store(tmp_path)
store.upsert_videos([
VideoRef(v, "UC1", v, f"https://y/watch?v={v}") for v in ("members", "throttled", "priv")
])
store.mark_error("members", "ERROR: [youtube] x: Join this channel to get access to members-only content")
store.mark_error("throttled", "ERROR: Video unavailable. The current session has been rate-limited by YouTube")
store.mark_error("priv", "ERROR: Private video. Sign in if you've been granted access")
counts = store.retryable_counts("UC1")
assert counts["error"] == 1, "only the throttled one is worth retrying"
assert counts["permanent"] == 2
assert store.reset_videos("UC1", ("error",)) == 1
assert store.get_video("throttled").status == "pending"
assert store.get_video("members").status == "error"
assert store.get_video("priv").status == "error"
def test_permanent_failures_can_be_reset_when_explicitly_asked(tmp_path):
store = _store(tmp_path)
store.upsert_videos([VideoRef("members", "UC1", "M", "https://y/watch?v=members")])
store.mark_error("members", "ERROR: members-only content")
assert store.reset_videos("UC1", ("error",)) == 0
assert store.reset_videos("UC1", ("error",), include_permanent=True) == 1
assert store.get_video("members").status == "pending"
def test_rate_limit_wording_is_never_treated_as_permanent(tmp_path):
"""The throttling message is the one that must stay retryable."""
store = _store(tmp_path)
store.upsert_videos([VideoRef("t", "UC1", "T", "https://y/watch?v=t")])
store.mark_error(
"t",
"ERROR: [youtube] t: Video unavailable. This content isn't available, try again later. "
"The current session has been rate-limited by YouTube for up to an hour.",
)
assert store.retryable_counts("UC1")["permanent"] == 0
assert store.reset_videos("UC1", ("error",)) == 1
def test_retryable_counts_reports_both_statuses(tmp_path):
store = _store(tmp_path)
store.upsert_videos([
VideoRef(v, "UC1", v, f"https://y/watch?v={v}") for v in ("a", "b", "c")
])
store.mark_status("a", "no_subtitles")
store.mark_status("b", "no_subtitles")
store.mark_error("c", "boom")
assert store.retryable_counts("UC1") == {"error": 1, "no_subtitles": 2, "permanent": 0}
# ---------------------------------------------------------------- skip reasons
def test_mark_status_records_the_reason(tmp_path):
store = _store(tmp_path)
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a")])
store.mark_status("a", "no_subtitles", "no track matched the language policy")
assert "language policy" in store.get_video("a").error_msg
def test_describe_distinguishes_no_captions_from_policy_rejection():
none_at_all = describe_missing_subtitle({"subtitles": {}, "automatic_captions": {}}, {"es": "manual"})
assert "no caption tracks published" in none_at_all
auto_only = describe_missing_subtitle(
{"subtitles": {}, "automatic_captions": {"es": [{"url": "u", "ext": "json3"}]}},
{"es": "manual"},
)
assert "ONLY auto-generated" in auto_only
assert "'any' or 'auto'" in auto_only
def test_describe_does_not_blame_config_when_mode_already_allows_auto():
msg = describe_missing_subtitle(
{"subtitles": {}, "automatic_captions": {"de": [{"url": "u", "ext": "json3"}]}},
{"es": "any"},
)
assert "ONLY auto-generated" not in msg
assert "no track matched the language policy" in msg
# ---------------------------------------------------------------- reconcile
_MD = """---
video_id: "{vid}"
title: "T"
upload_date: "2026-01-02"
---
## Transcript
**00:00** · hola mundo
"""
def test_reconcile_marks_done_when_the_md_is_already_on_disk(tmp_path):
"""The exact symptom the user reported: a .md exists but the row still
shows a failure, and nothing reconciles it without a server restart."""
store = _store(tmp_path)
store.upsert_videos([VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1")])
store.mark_status("vid1", "no_subtitles", "stale failure")
md_root = tmp_path / "markdown" / "Alpha"
md_root.mkdir(parents=True)
(md_root / "2026-01-02_t.md").write_text(_MD.format(vid="vid1"), encoding="utf-8")
result = reconcile_markdown(store, tmp_path / "markdown")
row = store.get_video("vid1")
assert row.status == "done"
assert row.markdown_path == "markdown/Alpha/2026-01-02_t.md"
assert result["repaired_done"] == 1
def test_reconcile_requeues_rows_whose_md_vanished_only_with_prune(tmp_path):
store = _store(tmp_path)
store.upsert_videos([
VideoRef("gone", "UC1", "G", "https://y/watch?v=gone"),
VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1"),
])
store.mark_done("gone", "markdown/Alpha/nope.md", "es", "auto", False)
md_root = tmp_path / "markdown" / "Alpha"
md_root.mkdir(parents=True)
(md_root / "2026-01-02_t.md").write_text(_MD.format(vid="vid1"), encoding="utf-8")
# default is non-destructive
assert reconcile_markdown(store, tmp_path / "markdown")["missing_md"] == 0
assert store.get_video("gone").status == "done"
result = reconcile_markdown(store, tmp_path / "markdown", prune=True)
assert result["missing_md"] == 1
assert store.get_video("gone").status == "pending"
def test_prune_refuses_to_demote_everything_when_the_root_is_empty(tmp_path):
"""Pointed at a wrong or not-yet-populated markdown root, prune must be a
no-op rather than wiping every finished video in the database."""
store = _store(tmp_path)
store.upsert_videos([VideoRef("c", "UC1", "C", "https://y/watch?v=c")])
store.mark_done("c", "markdown/Alpha/c.md", "es", "auto", False)
(tmp_path / "markdown").mkdir()
result = reconcile_markdown(store, tmp_path / "markdown", prune=True)
assert result["missing_md"] == 0
assert store.get_video("c").status == "done"
def test_reconcile_is_idempotent(tmp_path):
store = _store(tmp_path)
store.upsert_videos([VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1")])
md_root = tmp_path / "markdown" / "Alpha"
md_root.mkdir(parents=True)
(md_root / "2026-01-02_t.md").write_text(_MD.format(vid="vid1"), encoding="utf-8")
first = reconcile_markdown(store, tmp_path / "markdown")
second = reconcile_markdown(store, tmp_path / "markdown")
assert first["repaired_done"] == 1
assert second["repaired_done"] == 0
assert second["missing_md"] == 0
assert store.get_video("vid1").status == "done"
def test_a_bad_encoding_does_not_abort_the_whole_scan(tmp_path):
"""UnicodeDecodeError is a ValueError, not an OSError. Letting it escape
aborted the loop, so one bad file silently hid every later one."""
store = _store(tmp_path)
store.upsert_videos([
VideoRef("v1", "UC1", "A", "https://y/watch?v=v1"),
VideoRef("v3", "UC1", "C", "https://y/watch?v=v3"),
])
md_root = tmp_path / "markdown" / "Alpha"
md_root.mkdir(parents=True)
(md_root / "a.md").write_text(_MD.format(vid="v1"), encoding="utf-8")
(md_root / "b.md").write_bytes(b"---\nvideo_id: \xe9\xe9\xe9\n---\n")
(md_root / "c.md").write_text(_MD.format(vid="v3"), encoding="utf-8")
result = reconcile_markdown(store, tmp_path / "markdown")
assert store.get_video("v3").status == "done", "the file after the bad one must still import"
assert result["repaired_done"] == 2
# ---------------------------------------------------------------- re-render
def test_re_render_updates_markdown_path_and_removes_the_old_file(tmp_path):
"""re-render used a raw compact upload_date while process_video uses the
hyphenated form, so it wrote a SECOND .md and never told the DB — leaving
the app serving the older file. Measured on real data: 94 files, 61 rows."""
from yt_scraper.config import Config
from yt_scraper.pipeline import re_render_videos
store = _store(tmp_path)
store.upsert_videos([VideoRef("v1", "UC1", "Mi Video", "https://y/watch?v=v1", "20240519", 60)])
store.update_video_metadata(
"v1", view_count=1, like_count=1, tags=None, thumbnail=None, description=None,
chapters_json="[]",
segments_json='[{"start": 0.0, "end": 2.0, "text": "hola"}]',
)
md_root = tmp_path / "markdown"
old_dir = md_root / "Alpha"
old_dir.mkdir(parents=True)
(old_dir / "2024-05-19_mi-video.md").write_text("stale", encoding="utf-8")
store.mark_done("v1", "markdown/Alpha/2024-05-19_mi-video.md", "es", "auto", False)
cfg = Config(
database_path=str(tmp_path / "state.db"),
output_dir=str(md_root),
template_path=str(Path("templates/video.md.j2").resolve()),
)
assert re_render_videos(store, cfg) == 1
files = sorted(p.name for p in md_root.rglob("*.md"))
assert len(files) == 1, f"re-render must not leave an orphan beside it: {files}"
row = store.get_video("v1")
assert row.markdown_path.replace("\\", "/").endswith(files[0])
assert (md_root.parent / row.markdown_path).exists()
assert "stale" not in (md_root.parent / row.markdown_path).read_text(encoding="utf-8")
def test_re_render_is_idempotent(tmp_path):
from yt_scraper.config import Config
from yt_scraper.pipeline import re_render_videos
store = _store(tmp_path)
store.upsert_videos([VideoRef("v1", "UC1", "Mi Video", "https://y/watch?v=v1", "20240519", 60)])
store.update_video_metadata(
"v1", view_count=None, like_count=None, tags=None, thumbnail=None, description=None,
chapters_json="[]", segments_json='[{"start": 0.0, "end": 2.0, "text": "hola"}]',
)
store.mark_done("v1", "markdown/Alpha/whatever.md", "es", "auto", False)
cfg = Config(
database_path=str(tmp_path / "state.db"),
output_dir=str(tmp_path / "markdown"),
template_path=str(Path("templates/video.md.j2").resolve()),
)
re_render_videos(store, cfg)
first = store.get_video("v1").markdown_path
re_render_videos(store, cfg)
assert store.get_video("v1").markdown_path == first
assert len(list((tmp_path / "markdown").rglob("*.md"))) == 1
def test_reconcile_reports_stale_duplicates_and_deletes_them_only_with_prune(tmp_path):
"""The 33 leftover files the old re-render wrote under a second filename:
the DB points at one, the other is dead weight."""
store = _store(tmp_path)
store.upsert_videos([VideoRef("vid1", "UC1", "T", "https://y/watch?v=vid1")])
md_root = tmp_path / "markdown" / "Alpha"
md_root.mkdir(parents=True)
canonical = md_root / "2026-01-02_t.md"
duplicate = md_root / "20260102_t.md"
canonical.write_text(_MD.format(vid="vid1"), encoding="utf-8")
duplicate.write_text(_MD.format(vid="vid1"), encoding="utf-8")
store.mark_done("vid1", "markdown/Alpha/2026-01-02_t.md", "es", "auto", False)
result = reconcile_markdown(store, tmp_path / "markdown")
assert result["stale_dupe"] == 1
assert duplicate.exists(), "reporting only by default"
assert store.get_video("vid1").markdown_path == "markdown/Alpha/2026-01-02_t.md"
result = reconcile_markdown(store, tmp_path / "markdown", prune=True)
assert result["stale_dupe"] == 1
assert not duplicate.exists()
assert canonical.exists(), "the file the DB points at must survive"
assert store.get_video("vid1").status == "done"
def test_reconcile_counts_orphan_markdown(tmp_path):
store = _store(tmp_path)
md_root = tmp_path / "markdown" / "Alpha"
md_root.mkdir(parents=True)
(md_root / "ghost.md").write_text(_MD.format(vid="not-in-db"), encoding="utf-8")
result = reconcile_markdown(store, tmp_path / "markdown")
assert result["orphan_md"] == 1
assert result["repaired_done"] == 0
+171
View File
@@ -0,0 +1,171 @@
"""Guards on how many requests we spend and under what option names.
Two classes of regression live here, both of which actually happened:
1. An option name yt-dlp does not recognise. yt-dlp ignores unknown keys
silently, so `sleep_subrequests` looked configured for the project's whole
history while nothing ever slept between requests. `test_*_options_are_real`
checks every key against yt-dlp's own list instead of trusting review.
2. An extraction that walks far more of a channel than it needs.
`deep_channel_avatar` read one avatar URL by fully extracting every video the
channel had ever published — 735 requests and still going when a measurement
aborted it. The bound is asserted here because nothing else would notice.
"""
from __future__ import annotations
import pytest
import yt_dlp
from yt_scraper import discover, extract
#: Every option name yt-dlp's own CLI parser produces. Anything outside this is
#: either a typo or something yt-dlp will silently drop.
KNOWN_YDL_OPTIONS = set(yt_dlp.parse_options([]).ydl_opts)
class CapturingYDL:
"""Stands in for yt_dlp.YoutubeDL and records the options it was built with."""
captured: list[dict] = []
info: dict = {}
def __init__(self, options=None, *args, **kwargs):
type(self).captured.append(dict(options or {}))
self.options = options or {}
def __enter__(self):
return self
def __exit__(self, *exc):
return False
def extract_info(self, url, download=False, process=True):
return dict(type(self).info)
@pytest.fixture
def capture(monkeypatch):
CapturingYDL.captured = []
CapturingYDL.info = {
"id": "UC123",
"channel_id": "UC123",
"channel": "Test Channel",
"thumbnails": [{"url": "https://yt3.ggpht.com/avatar.jpg"}],
"entries": [],
}
monkeypatch.setattr(yt_dlp, "YoutubeDL", CapturingYDL)
return CapturingYDL
def assert_options_are_real(opts: dict, where: str) -> None:
unknown = sorted(set(opts) - KNOWN_YDL_OPTIONS)
assert not unknown, (
f"{where} passes option(s) yt-dlp does not recognise and will silently "
f"ignore: {unknown}"
)
# ------------------------------------------------------------------ names
def test_discover_channel_options_are_real(capture):
discover.discover_channel("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
assert_options_are_real(capture.captured[0], "discover_channel")
def test_deep_channel_avatar_options_are_real(capture):
discover.deep_channel_avatar("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
assert_options_are_real(capture.captured[0], "deep_channel_avatar")
def test_extract_video_options_are_real(capture):
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
assert_options_are_real(capture.captured[0], "extract_video")
# ------------------------------------------------------------------ throttles wired
@pytest.mark.parametrize(
"call",
[
pytest.param(
lambda: discover.discover_channel("https://www.youtube.com/@x/videos",
sleep_subrequests=3.25),
id="discover_channel",
),
pytest.param(
lambda: discover.deep_channel_avatar("https://www.youtube.com/@x/videos",
sleep_subrequests=3.25),
id="deep_channel_avatar",
),
pytest.param(
lambda: extract.extract_video("https://www.youtube.com/watch?v=vid",
{"es": "any"}, sleep_subrequests=3.25),
id="extract_video",
),
],
)
def test_every_entry_point_forwards_the_real_sleep_option(capture, call):
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {},
"channel_id": "UC1", "entries": []}
call()
opts = capture.captured[0]
assert opts.get("sleep_interval_requests") == 3.25
assert opts.get("socket_timeout"), "a hung connection must not block the worker forever"
assert "extractor_retries" in opts
# ------------------------------------------------------------------ request bounds
def test_deep_avatar_does_not_walk_the_channel(capture):
"""The 735-request bug. Both halves of the fix are asserted.
`extract_flat` stops yt-dlp expanding each entry into a full extraction, and
`playlistend` stops it paginating past the first page. Either one missing
puts the whole channel back on the wire.
"""
discover.deep_channel_avatar("https://www.youtube.com/@x/videos")
opts = capture.captured[0]
assert opts.get("extract_flat"), "must not fully extract every video"
assert opts.get("playlistend") == 1, "must not paginate beyond the first page"
def test_discover_channel_limit_becomes_playlistend(capture):
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
assert capture.captured[0].get("playlistend") == 30
def test_discover_channel_without_limit_has_no_ceiling(capture):
"""A brand-new channel legitimately walks everything; that must stay possible."""
discover.discover_channel("https://www.youtube.com/@x/videos")
assert "playlistend" not in capture.captured[0]
def test_extraction_never_probes_formats(capture):
"""`check_formats` costs one HTTP request per format and we only want captions."""
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
assert capture.captured[0].get("check_formats") is None
def test_discovery_processes_the_result_so_playlistend_applies(capture, monkeypatch):
"""`playlistend` is silently ignored when extract_info runs with process=False.
Measured on a 2564-video channel: processed + playlistend=60 costs 2
requests; unprocessed, the lazy generator ignores the limit and walking it
costs 86. Nothing else in the codebase would catch that flip.
"""
seen = {}
def extract_info(self, url, download=False, process=True):
seen["process"] = process
return dict(CapturingYDL.info)
monkeypatch.setattr(CapturingYDL, "extract_info", extract_info)
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
assert seen["process"] is not False
+33
View File
@@ -0,0 +1,33 @@
"""`--since` / the scrape form's date filter must not eat undated entries.
yt-dlp's flat channel listing does not report upload_date, so comparing a
missing date against the cutoff used to discard everything discovery found.
"""
from __future__ import annotations
from yt_scraper.cli import _apply_filters
from yt_scraper.store import VideoRef
def _ref(vid: str, upload_date: str | None) -> VideoRef:
return VideoRef(vid, "UC1", vid, f"https://www.youtube.com/watch?v={vid}", upload_date)
def test_since_keeps_entries_with_no_upload_date():
refs = [_ref("undated", None), _ref("new", "20260715"), _ref("old", "20200101")]
kept = _apply_filters(refs, "2026-07-01", False, False, 0, None)
assert [r.video_id for r in kept] == ["undated", "new"]
def test_since_still_drops_entries_that_are_provably_older():
refs = [_ref("old", "20200101"), _ref("older", "20190101")]
assert _apply_filters(refs, "2026-07-01", False, False, 0, None) == []
def test_without_since_nothing_is_dropped_on_date_grounds():
refs = [_ref("undated", None), _ref("old", "20200101")]
assert len(_apply_filters(refs, None, False, False, 0, None)) == 2
+31
View File
@@ -100,6 +100,37 @@ def test_query_videos_filters(seeded_store):
assert rows3[0].video_id == "v2"
def test_upsert_videos_reports_only_new_rows(seeded_store):
inserted = seeded_store.upsert_videos([
VideoRef("v1", "UC1", "Updated title", "https://y/watch?v=v1", "20240101", 120),
VideoRef("v4", "UC1", "New video", "https://y/watch?v=v4", "20240501", 180),
])
assert inserted == 1
assert seeded_store.get_video("v1").title == "Updated title"
assert seeded_store.get_video("v4").status == "pending"
def test_upload_sort_uses_discovered_time_when_date_is_missing(seeded_store):
seeded_store.upsert_videos([
VideoRef("recent", "UC1", "Recent discovered", "https://y/watch?v=recent"),
])
rows, _ = seeded_store.query_videos(channel_id="UC1", sort="upload_date", page=1, size=1)
assert rows[0].video_id == "recent"
def test_mark_status_clears_stale_error_message(seeded_store):
seeded_store.mark_error("v1", "temporary extraction failure")
seeded_store.mark_status("v1", "no_subtitles")
video = seeded_store.get_video("v1")
assert video.status == "no_subtitles"
assert video.error_msg is None
def test_cookie_vault(store):
from yt_scraper import cookies
sample = (
+80
View File
@@ -0,0 +1,80 @@
"""Store-side support for incremental sync: the known-id boundary and the
watermark that survives a re-discovery."""
from __future__ import annotations
from yt_scraper.store import Store, VideoRef
def _store(tmp_path) -> Store:
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
return store
def test_known_video_ids_is_scoped_to_the_channel(tmp_path):
store = _store(tmp_path)
store.upsert_channel("UC2", "@beta", "Beta", 0)
store.upsert_videos([
VideoRef("a", "UC1", "A", "https://y/watch?v=a"),
VideoRef("b", "UC1", "B", "https://y/watch?v=b"),
VideoRef("c", "UC2", "C", "https://y/watch?v=c"),
])
assert store.known_video_ids("UC1") == {"a", "b"}
assert store.known_video_ids("UC2") == {"c"}
assert store.known_video_ids("nope") == set()
def test_rediscovery_does_not_wipe_a_known_upload_date(tmp_path):
"""Flat discovery reports upload_date=None; without COALESCE a routine sync
would erase the dates learned during extraction — and with them the very
watermark this feature is built on."""
store = _store(tmp_path)
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260715", 120)])
store.upsert_videos([VideoRef("a", "UC1", "A (renamed)", "https://y/watch?v=a", None, None)])
row = store.get_video("a")
assert row.upload_date == "20260715"
assert row.duration == 120
assert row.title == "A (renamed)", "titles should still refresh"
def test_latest_upload_date_ignores_undated_rows(tmp_path):
store = _store(tmp_path)
store.upsert_videos([
VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260101"),
VideoRef("b", "UC1", "B", "https://y/watch?v=b", None),
VideoRef("c", "UC1", "C", "https://y/watch?v=c", "20260720"),
])
assert store.latest_upload_date("UC1") == "20260720"
def test_mark_channel_synced_recounts_instead_of_trusting_the_window(tmp_path):
"""An incremental pass only sees the newest slice, so video_count must come
from the DB — otherwise an 848-video channel shrinks to the window size."""
store = _store(tmp_path)
store.upsert_videos([
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", "20260101")
for i in range(40)
])
marks = store.mark_channel_synced("UC1")
assert marks["video_count"] == 40
channel = store.get_channel("UC1")
assert channel["video_count"] == 40
assert channel["last_video_date"] == "20260101"
assert channel["last_synced_at"]
def test_mark_channel_synced_keeps_the_last_date_when_nothing_is_dated(tmp_path):
store = _store(tmp_path)
store.upsert_videos([VideoRef("a", "UC1", "A", "https://y/watch?v=a", "20260101")])
store.mark_channel_synced("UC1")
store.upsert_videos([VideoRef("b", "UC1", "B", "https://y/watch?v=b", None)])
store.mark_channel_synced("UC1")
assert store.get_channel("UC1")["last_video_date"] == "20260101"
+219
View File
@@ -0,0 +1,219 @@
"""The transcript must be the language actually spoken, not a machine translation.
Found in production on the Alex Hormozi channel (English). Under
`languages: {es: any, es-419: any, en: any}` the picker walked the config in
order, matched `es` first, and stored a Spanish auto-translation of English
speech — "Soy Nim Jenkinson y enseño a los aficionados a las manualidades" for
a video whose speaker says it in English.
Two things conspired:
1. `_normalize_lang("es-orig") == "es"`, so the `-orig` suffix — the one piece
of evidence distinguishing YouTube's real ASR track from a translation *into
the same language* — was thrown away before comparison.
2. Config order was treated as absolute preference, so a translation into a
preferred language beat the original.
Spanish channels were unaffected only by luck: yt-dlp happens to list `es-orig`
before `es`, so dict order gave the right answer. Every `done` row on the four
Spanish channels recorded `transcript_lang=es-orig`; the three Hormozi rows
recorded `es`. That asymmetry is what these tests pin down.
"""
from __future__ import annotations
from yt_scraper.extract import original_language, pick_subtitle
CFG = {"es": "any", "es-419": "any", "en": "any"}
def _track(url: str) -> list[dict]:
return [{"ext": "json3", "url": url}]
def _english_video() -> dict:
"""An English video as YouTube actually presents it: the original ASR under
`en-orig`, plus a long tail of translations keyed by bare language code."""
return {
"id": "aRVv5NLVRwE",
"title": "My honest advice to someone who wants to get rich.",
"subtitles": {},
"automatic_captions": {
"en-orig": _track("https://timedtext/en-orig"),
"en": _track("https://timedtext/en-translated"),
"es": _track("https://timedtext/es"),
"es-419": _track("https://timedtext/es-419"),
"fr": _track("https://timedtext/fr"),
},
}
def _spanish_video() -> dict:
return {
"id": "uZH3FKH_yNw",
"title": "Escribir codigo a mano sera irresponsable",
"subtitles": {},
"automatic_captions": {
"es-orig": _track("https://timedtext/es-orig"),
"es": _track("https://timedtext/es"),
"en": _track("https://timedtext/en"),
},
}
def test_english_video_yields_english_not_a_spanish_translation():
"""The production bug, verbatim."""
pick = pick_subtitle(_english_video(), CFG, prefer_manual=True)
assert pick is not None
assert pick.lang == "en-orig", f"picked {pick.lang}: a translation, not the spoken language"
def test_spanish_video_still_yields_the_spanish_original():
"""The fix must not regress the four Spanish channels already in the DB."""
pick = pick_subtitle(_spanish_video(), CFG, prefer_manual=True)
assert pick is not None
assert pick.lang == "es-orig"
def test_original_beats_a_same_language_translation():
"""YouTube publishes a translation *into the video's own language* too.
`en-orig` and `en` both normalise to "en"; only the suffix says which one is
the real transcript, and dict order must not decide it.
"""
info = {
"subtitles": {},
# Deliberately listed translation-first to defeat insertion order.
"automatic_captions": {
"en": _track("https://timedtext/en-translated"),
"en-orig": _track("https://timedtext/en-orig"),
},
}
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
assert pick.lang == "en-orig"
assert pick.url.endswith("en-orig")
def test_translation_is_a_documented_last_resort_not_a_silent_default():
"""A German channel under a Spanish/English policy.
The spoken language is not one the operator asked for, so there is no
original to give them and a translation is the only thing on offer. We do
take it — but only on the second pass, after every untranslated option has
been rejected, and `transcript_lang` records the bare code so the row is
distinguishable from an `-orig` one afterwards.
This case is a deliberate fallback. It is NOT the behaviour that caused the
Hormozi bug: there, `en` *was* configured and was being skipped.
"""
info = {
"subtitles": {},
"automatic_captions": {
"de-orig": _track("https://timedtext/de-orig"),
"es": _track("https://timedtext/de?tlang=es"),
"en": _track("https://timedtext/de?tlang=en"),
},
}
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "es"
assert "tlang=" in pick.url, "the fallback really is a translation; nothing else was available"
# ------------------------------------------------- the tlang= guard
#
# A translated caption URL is the base track's URL with `tlang=` appended, and
# yt-dlp omits it when target == source. That is direct evidence, unlike the
# `-orig` naming convention, so it catches videos whose spoken language cannot
# be determined any other way.
def test_untranslated_track_wins_even_with_no_orig_key_and_no_language_field():
"""The gap the first version of this fix left open.
Without an `-orig` key and without `info["language"]`, the spoken language
is unknown, the reorder cannot fire, and config order used to hand back the
Spanish translation. Rejecting `tlang=` needs no such knowledge.
"""
info = {
"subtitles": {},
"automatic_captions": {
"es": _track("https://timedtext/base?lang=en&kind=asr&tlang=es"),
"en": _track("https://timedtext/base?lang=en&kind=asr"),
},
}
assert original_language(info) is None, "precondition: spoken language is undeterminable"
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "en"
assert "tlang=" not in pick.url
def test_the_real_hormozi_url_shape_is_recognised_as_a_translation():
"""Verbatim parameters from the production timedtext URLs in .run/server.log."""
info = {
"subtitles": {},
"automatic_captions": {
"es": _track(
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
"&lang=en&kind=asr&variant=gemini&fmt=json3&tlang=es"
),
"en-orig": _track(
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
"&lang=en&kind=asr&variant=gemini&fmt=json3"
),
},
}
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "en-orig"
assert "tlang=" not in pick.url
def test_manual_captions_still_outrank_auto_for_the_same_language():
"""Preferring the original must not override the manual/auto policy."""
info = {
"subtitles": {"en": _track("https://timedtext/en-manual")},
"automatic_captions": {"en-orig": _track("https://timedtext/en-orig")},
}
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
assert pick.source == "manual"
assert pick.url.endswith("en-manual")
def test_a_manual_translation_does_not_beat_the_spoken_language():
"""Config lists es first, but the video is English with English manual subs."""
info = {
"subtitles": {"en": _track("https://timedtext/en-manual")},
"automatic_captions": {
"en-orig": _track("https://timedtext/en-orig"),
"es": _track("https://timedtext/es"),
},
}
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "en"
assert pick.source == "manual"
# ------------------------------------------------------------- detection
def test_original_language_read_from_the_orig_suffix():
assert original_language(_english_video()) == "en"
assert original_language(_spanish_video()) == "es"
def test_original_language_falls_back_to_the_info_key():
assert original_language({"language": "pt-BR", "automatic_captions": {}}) == "pt"
def test_orig_suffix_wins_over_the_info_key():
"""`language` is metadata YouTube localises; the caption list is evidence.
Production showed YouTube returning English titles for Spanish videos, so
localised metadata is not trustworthy for this decision.
"""
info = {"language": "es", "automatic_captions": {"en-orig": _track("u")}}
assert original_language(info) == "en"
def test_original_language_is_none_when_unknowable():
assert original_language({"automatic_captions": {"es": _track("u")}}) is None
assert original_language({}) is None
+246
View File
@@ -0,0 +1,246 @@
"""The circuit breaker, tested through the webapp job runner.
This is the regression test for the incident the whole change exists to
prevent: a session gets rate-limited, the runner keeps going anyway, and every
remaining video is marked failed. In production that turned one throttling
event into 511 `no_subtitles` and 347 `error` rows on a single channel.
The invariant asserted everywhere below is the same: **videos the run never
reached must still be `pending`**, because `pending` is what a later run picks
up. A video wrongly marked `error` needs a manual reset first.
"""
from __future__ import annotations
import pytest
from yt_scraper.config import Config, DelayConfig
from yt_scraper.store import Store, VideoRef
from yt_scraper.webapp.jobs import JobManager
THROTTLED = (
"ERROR: [youtube] {vid}: Video unavailable. This content isn't available, try "
"again later. The current session has been rate-limited by YouTube for up to an hour."
)
def _cfg(tmp_path) -> Config:
return Config(
database_path=str(tmp_path / "state.db"),
output_dir=str(tmp_path / "markdown"),
# No real waiting in tests; the breaker's arithmetic is unit-tested
# separately in test_ratelimit.py.
delay=DelayConfig(
min_seconds=0.0, max_seconds=0.0,
backoff_base=0.001, backoff_cap=0.002, throttle_threshold=3,
),
)
def _seed(store: Store, n: int) -> list[str]:
store.upsert_channel("UC1", "@alpha", "Alpha", n)
ids = [f"v{i:02d}" for i in range(n)]
store.upsert_videos([
VideoRef(v, "UC1", f"Video {i}", f"https://y/watch?v={v}") for i, v in enumerate(ids)
])
return ids
def _always_throttled(store: Store):
"""Stand-in for process_video that fails the way a throttled session does."""
def fake(row, cfg, st, *args, **kwargs):
st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id))
return "error"
return fake
@pytest.fixture
def manager(tmp_path):
store = Store(tmp_path / "state.db")
return JobManager(store, _cfg(tmp_path)), store
def test_batch_stops_and_leaves_the_rest_pending(manager, monkeypatch):
mgr, store = manager
ids = _seed(store, 20)
monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store))
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
touched = [v for v in ids if store.get_video(v).status != "pending"]
assert len(touched) == 3, "must stop at the threshold, not walk all 20"
assert all(store.get_video(v).status == "pending" for v in ids[3:])
job = store.get_job("job")
assert job.status == "error"
assert "rate-limit" in (job.last_error or "")
events = mgr.events_since("job", 0)
err = [e for e in events if e["event"] == "error"]
assert err and err[-1]["data"]["throttled"] is True
def test_channel_run_stops_and_leaves_the_rest_pending(manager, monkeypatch):
mgr, store = manager
ids = _seed(store, 20)
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", _always_throttled(store))
monkeypatch.setattr(
"yt_scraper.discover.discover_channel",
lambda url, sleep_subrequests=2.0, limit=None: ("UC1", "Alpha", None, []),
)
store.create_job("job", "UC1", {})
mgr._run_channel("job", {})
assert len([v for v in ids if store.get_video(v).status != "pending"]) == 3
assert store.get_job("job").status == "error"
def test_a_healthy_run_is_untouched_by_the_breaker(manager, monkeypatch):
"""The breaker must be invisible when nothing is throttling."""
mgr, store = manager
ids = _seed(store, 8)
monkeypatch.setattr("yt_scraper.pipeline.process_video",
lambda row, *a, **k: store.mark_status(row.video_id, "done") or "done")
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
assert store.get_job("job").status == "done"
assert all(store.get_video(v).status == "done" for v in ids)
def test_ordinary_failures_do_not_stop_the_run(manager, monkeypatch):
"""A channel with dead videos must still be processed to the end.
Without the throttle/permanent distinction, three members-only videos in a
row would abort a perfectly healthy scrape.
"""
mgr, store = manager
ids = _seed(store, 10)
def fake(row, cfg, st, *args, **kwargs):
st.mark_error(row.video_id, "ERROR: [youtube] x: Private video. Sign in if you've been granted access")
return "error"
monkeypatch.setattr("yt_scraper.pipeline.process_video", fake)
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
assert store.get_job("job").status == "done"
assert all(store.get_video(v).status == "error" for v in ids)
def test_isolated_throttling_between_successes_does_not_stop_the_run(manager, monkeypatch):
"""Only *consecutive* throttling means the session is banned."""
mgr, store = manager
ids = _seed(store, 12)
calls = {"n": 0}
def fake(row, cfg, st, *args, **kwargs):
calls["n"] += 1
if calls["n"] % 2:
st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id))
return "error"
st.mark_status(row.video_id, "done")
return "done"
monkeypatch.setattr("yt_scraper.pipeline.process_video", fake)
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
assert store.get_job("job").status == "done"
assert calls["n"] == 12
def test_a_throttled_caption_fetch_is_an_error_not_no_subtitles(tmp_path, monkeypatch):
"""Which of the three requests YouTube refused must not decide the status.
A 429 on `extract_info` produced `error`; a 429 on the separate caption
download produced `no_subtitles` — the state that means "this video
publishes no captions", which is what the `no_subtitles` counts are read as.
Both are retryable, but only one is honest.
"""
from yt_scraper.extract import VideoData
from yt_scraper.pipeline import process_video
store = Store(tmp_path / "state.db")
cfg = _cfg(tmp_path)
_seed(store, 1)
row = store.get_video("v00")
throttled = VideoData(
info={"id": "v00", "title": "t"},
segments=[],
subtitle=None,
has_chapters=False,
skip_reason=(
"subtitle track found (lang=en, auto) but the download failed "
"[HTTP Error 429: HTTPError: 429 Client Error: Too Many Requests for url: ...]"
),
)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: throttled)
assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "error"
genuinely_absent = VideoData(
info={"id": "v00", "title": "t"}, segments=[], subtitle=None,
has_chapters=False, skip_reason="no caption tracks published for this video",
)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: genuinely_absent)
assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "no_subtitles"
def test_metadata_survives_a_failed_caption_fetch(tmp_path, monkeypatch):
"""The extraction already paid for this data; a retry must not re-buy it."""
from yt_scraper.extract import VideoData
from yt_scraper.pipeline import process_video
store = Store(tmp_path / "state.db")
cfg = _cfg(tmp_path)
_seed(store, 1)
row = store.get_video("v00")
monkeypatch.setattr(
"yt_scraper.pipeline.extract_video",
lambda *a, **k: VideoData(
info={"id": "v00", "title": "t", "view_count": 4321, "upload_date": "20260101",
"thumbnail": "https://i.ytimg.com/x.jpg", "description": "hola"},
segments=[], subtitle=None, has_chapters=False,
skip_reason="no caption tracks published for this video",
),
)
process_video(row, cfg, store, "Alpha", "UC1", "u")
after = store.get_video("v00")
assert after.status == "no_subtitles"
assert after.view_count == 4321, "metadata was discarded with the failed transcript"
assert after.upload_date == "20260101"
def test_throttled_videos_stay_retryable(manager, monkeypatch):
"""The three videos that did fail must not be classified as permanent.
`Store.PERMANENT_ERROR_PATTERNS` deliberately excludes the throttling
message; if that ever changed, a rate-limit incident would poison rows that
a later run could have recovered.
"""
mgr, store = manager
ids = _seed(store, 10)
monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store))
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
counts = store.retryable_counts("UC1")
assert counts.get("permanent", 0) == 0
assert store.reset_videos("UC1", ("error",)) == 3
+319
View File
@@ -0,0 +1,319 @@
"""Chronological ordering of the videos list.
The list has to read like a YouTube channel page — newest upload first — and
that has to hold for videos nothing has scraped yet. Those have no
`upload_date` at all: yt-dlp's flat listing does not report one for YouTube
entries, so the date only arrives with the extraction that produces the .md.
`channel_seq` (the video's slot in the reverse-chronological /videos tab) is
what keeps them in place until then.
"""
from __future__ import annotations
import sqlite3
import pytest
from yt_scraper.store import NO_DATE_SENTINEL, SCHEMA, SORT_DATE_SQL, Store, VideoRef
@pytest.fixture()
def store(tmp_path):
s = Store(tmp_path / "state.db")
s.upsert_channel("UC1", "@alpha", "Alpha", 0)
return s
def _ref(video_id: str, upload_date: str | None = None, channel_id: str = "UC1") -> VideoRef:
return VideoRef(
video_id, channel_id, video_id.upper(),
f"https://y/watch?v={video_id}", upload_date,
)
def _ids(rows) -> list[str]:
return [r.video_id for r in rows]
# A channel as discovery hands it over: /videos-tab order, newest first, with
# two videos nobody has extracted yet ("a" and "c") interleaved among two that
# have been ("b" and "d").
MIXED = [_ref("a"), _ref("b", "20240301"), _ref("c"), _ref("d", "20240101")]
def test_default_order_is_newest_upload_first(store):
store.upsert_videos([_ref("mid", "20240201"), _ref("new", "20240301"), _ref("old", "20240101")])
rows, total = store.query_videos()
assert total == 3
assert _ids(rows) == ["new", "mid", "old"]
def test_videos_without_a_markdown_keep_their_slot_in_the_channel(store):
"""The regression this exists for: the fallback used to be `discovered_at`,
which dates every un-scraped video "today" and floats the whole backlog
above videos that really were uploaded this week."""
store.upsert_videos(MIXED)
rows, _ = store.query_videos()
assert _ids(rows) == ["a", "b", "c", "d"]
def test_an_undated_video_reports_the_date_it_was_sorted_by(store):
store.upsert_videos(MIXED)
by_id = {r.video_id: r for r in store.query_videos()[0]}
assert by_id["b"].sort_date == "20240301"
# "c" has no date of its own, so it borrows the nearest NEWER dated video's.
assert by_id["c"].upload_date is None
assert by_id["c"].sort_date == "20240301"
# Nothing dated sits above "a", so the most it can claim is the newest date
# known in its channel. Rank puts it above that video; inventing a later
# date would let an un-scraped channel outrank every scraped one.
assert by_id["a"].sort_date == "20240301"
def test_a_channel_with_nothing_scraped_does_not_outrank_channels_that_are(store):
"""With no dated video anywhere in the channel there is zero evidence about
when anything was published. Measured on the real library, one such channel
(212 videos, none extracted) took over the whole first page."""
store.upsert_channel("UC2", "@beta", "Beta", 0)
store.upsert_videos([_ref("known", "20240101", channel_id="UC2")])
store.upsert_videos([_ref("u1"), _ref("u2")]) # UC1: nothing dated at all
rows, _ = store.query_videos()
assert _ids(rows) == ["known", "u1", "u2"]
# ...but the blind channel is still internally in listing order.
assert {r.sort_date for r in rows if r.channel_id == "UC1"} == {NO_DATE_SENTINEL}
def test_a_new_upload_ranks_above_the_backlog_it_did_not_refetch(store):
"""An incremental sync only fetches the newest window. Everything outside it
is older by construction, so the window is ranked above the stored rows
instead of restarting from zero and colliding with them."""
store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")]) # first full pass
store.upsert_videos([_ref("n2"), _ref("n1"), _ref("v3"), _ref("v2")]) # later window
rows, _ = store.query_videos()
assert _ids(rows) == ["n2", "n1", "v3", "v2", "v1"]
def test_repeated_syncs_do_not_reshuffle_a_channel(store):
store.upsert_videos([_ref("v3"), _ref("v2"), _ref("v1")])
first = _ids(store.query_videos()[0])
for _ in range(3):
store.upsert_videos([_ref("v3"), _ref("v2")]) # same window, nothing new
assert _ids(store.query_videos()[0]) == first
@pytest.mark.parametrize("sort", ["view_count", "like_count", "duration", "", "nonsense"])
def test_sort_keys_the_ui_sends_do_not_degrade_the_chronology(store, sort):
"""The web UI sends `view_count`/`like_count`/`duration`; the store only
knew `views_desc`/`duration_desc`. Every miss fell through to a raw
`upload_date DESC` that ignored the inference and dumped undated videos at
the bottom, whatever the channel order said."""
store.upsert_videos(MIXED)
rows, _ = store.query_videos(sort=sort)
assert _ids(rows) == ["a", "b", "c", "d"]
def test_sorting_by_views_ranks_counted_videos_and_keeps_the_rest_chronological(store):
store.upsert_videos(MIXED)
store.update_video_metadata("d", view_count=500)
store.update_video_metadata("b", view_count=10)
rows, _ = store.query_videos(sort="view_count")
assert _ids(rows) == ["d", "b", "a", "c"]
def test_oldest_first_reverses_the_default(store):
store.upsert_videos(MIXED)
rows, _ = store.query_videos(sort="oldest")
assert _ids(rows) == ["d", "c", "b", "a"]
def test_oldest_first_keeps_unknown_dates_at_the_end_not_the_start(store):
""""We do not know when this is from" is not "the beginning of time".
NO_DATE_SENTINEL sorts below every real date, which is what keeps those rows
off the front page under the default sort — and is exactly why they came
FIRST under ascending. Measured: "oldest" opened with the 212 videos of a
channel nothing had extracted, ahead of a genuine 2017 upload.
"""
store.upsert_channel("UC2", "@beta", "Beta", 0)
store.upsert_videos([_ref("ancient", "20170414", channel_id="UC2")])
store.upsert_videos([_ref("unknown1"), _ref("unknown2")]) # UC1: no dates at all
rows, _ = store.query_videos(sort="oldest")
assert _ids(rows)[0] == "ancient"
assert set(_ids(rows)[1:]) == {"unknown1", "unknown2"}
def test_a_tie_does_not_rank_channels_by_how_big_their_catalogue_is(store):
"""`channel_seq` counts up to a channel's video count, so comparing it
ACROSS channels ranks by catalogue size. With 92% of adjacent pairs tying on
sort_date, that decided most of the list: an entire channel preceded another
purely because 575 > 476. Ties group by channel instead, so no channel's
internal order is ever interleaved away."""
store.upsert_channel("UC2", "@beta", "Beta", 0)
# Same date everywhere: the tiebreak decides the whole ordering.
store.upsert_videos([_ref(f"big{i}", "20240101") for i in range(5)])
store.upsert_videos([_ref(f"small{i}", "20240101", channel_id="UC2") for i in range(2)])
ids = _ids(store.query_videos()[0])
big = [i for i, v in enumerate(ids) if v.startswith("big")]
small = [i for i, v in enumerate(ids) if v.startswith("small")]
assert max(big) < min(small) or max(small) < min(big), f"channels interleaved: {ids}"
# and each channel is still in its own listing order
assert [v for v in ids if v.startswith("big")] == ["big0", "big1", "big2", "big3", "big4"]
assert [v for v in ids if v.startswith("small")] == ["small0", "small1"]
def test_a_row_that_arrives_without_a_rank_is_ranked_on_the_next_open(tmp_path):
"""A server still running the pre-column code writes rows with a NULL rank.
Left NULL they do not just lose their place — SORT_DATE_SQL makes them
inherit the OLDEST date in the channel, so a video discovery has only just
found is shown last. Measured on the live database: two brand-new uploads
displayed with sort_date 20180720, second to last of 513."""
db = tmp_path / "state.db"
store = Store(db)
store.upsert_channel("UC1", "@alpha", "Alpha", 0)
store.upsert_videos([_ref("older", "20180720"), _ref("oldest", "20180101")])
with sqlite3.connect(db) as conn: # what the old code path produced
conn.execute(
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
" VALUES ('brand_new', 'UC1', 'Brand new', 'https://y/watch?v=brand_new',"
" 'pending', '2026-08-08T13:25:32+00:00')"
)
reopened = Store(db) # migration runs here
rows, _ = reopened.query_videos()
assert _ids(rows) == ["brand_new", "older", "oldest"]
assert rows[0].channel_seq is not None
assert rows[0].sort_date == "20180720" # newest known in the channel, not the oldest
def test_paging_neither_repeats_nor_loses_rows(store):
store.upsert_videos([_ref(f"v{i}") for i in range(7)])
full = _ids(store.query_videos(size=50)[0])
paged = [vid for page in (1, 2, 3) for vid in _ids(store.query_videos(page=page, size=3)[0])]
assert full == ["v0", "v1", "v2", "v3", "v4", "v5", "v6"]
assert paged == full
def test_filters_do_not_disturb_the_order(store):
store.upsert_videos(MIXED)
store.mark_status("a", "no_subtitles", "no captions")
rows, total = store.query_videos(channel_id="UC1", status="pending")
assert total == 3
assert _ids(rows) == ["b", "c", "d"]
def test_a_real_date_outranks_a_stale_listing_position(store):
"""Why the date is the primary key and the rank only the tiebreak.
A rank that predates the column was reconstructed offline, not observed from
YouTube, so it can be wrong. Sorting by date first means the dates we did
pay an extraction to learn CORRECT that reconstruction instead of being
overridden by it. Measured against the live listing on Código Espinoza:
ordering by rank alone was strictly worse than date-then-rank.
"""
# The rank claims "old" is the newer of the two; its upload_date says otherwise.
store.upsert_videos([_ref("old", "20240101"), _ref("new", "20260801")])
assert _ids(store.query_videos()[0]) == ["new", "old"]
def test_a_discovery_pass_repairs_a_rank_the_seed_got_wrong(store):
"""Two videos published the same day: the date cannot separate them, so the
listing rank decides — and a wrong rank shows a wrong order. One ordinary
sync re-observes the window from YouTube and fixes it. Measured: Código
Espinoza went from 9 inverted pairs to 0 after a single 30-entry pass."""
store.upsert_videos([_ref("a", "20260801"), _ref("b", "20260801")])
assert _ids(store.query_videos()[0]) == ["a", "b"]
store.upsert_videos([_ref("b", "20260801"), _ref("a", "20260801")]) # real order
assert _ids(store.query_videos()[0]) == ["b", "a"]
def test_the_ordering_lookups_stay_on_their_partial_indexes(store):
"""Both subqueries run once per undated row, so the plan is the difference
between a usable page and an unusable one: measured on the real library,
576 ms per page without these indexes and 22 ms with them. SQLite drops a
partial index the moment the query's WHERE stops implying the index's, and
says nothing about it."""
store.upsert_videos(MIXED)
sql = (
f"SELECT videos.*, {SORT_DATE_SQL} AS sort_date FROM videos "
"ORDER BY sort_date DESC, videos.channel_seq DESC, videos.video_id DESC LIMIT 25"
)
with store._cursor() as cur:
plan = [str(row["detail"]) for row in cur.execute("EXPLAIN QUERY PLAN " + sql)]
assert any("COVERING INDEX idx_videos_dated_seq" in line for line in plan), plan
assert any("COVERING INDEX idx_videos_dated" in line
and "idx_videos_dated_seq" not in line for line in plan), plan
def _legacy_db(path):
"""A database written before `channel_seq` existed."""
conn = sqlite3.connect(path)
conn.executescript(SCHEMA)
conn.execute("INSERT INTO channels (channel_id, name) VALUES ('UC1', 'Alpha')")
# One discovery batch shares a timestamp and is inserted newest-first...
for vid in ("b1", "b2", "b3"):
conn.execute(
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
" VALUES (?, 'UC1', ?, ?, 'pending', '2026-01-01T00:00:00+00:00')",
(vid, vid, f"https://y/watch?v={vid}"),
)
# ...and a later sync can only add videos newer than every one of them.
conn.execute(
"INSERT INTO videos (video_id, channel_id, title, url, status, discovered_at)"
" VALUES ('later', 'UC1', 'later', 'https://y/watch?v=later', 'pending',"
" '2026-02-01T00:00:00+00:00')"
)
conn.commit()
conn.close()
def test_migration_ranks_a_database_that_predates_the_column(tmp_path):
db = tmp_path / "legacy.db"
_legacy_db(db)
rows, _ = Store(db).query_videos()
assert _ids(rows) == ["later", "b1", "b2", "b3"]
assert all(r.channel_seq is not None for r in rows)
def test_migration_seeds_once_and_does_not_reshuffle_on_reopen(tmp_path):
db = tmp_path / "legacy.db"
_legacy_db(db)
first = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]}
again = {r.video_id: r.channel_seq for r in Store(db).query_videos()[0]}
assert again == first
+290
View File
@@ -0,0 +1,290 @@
from __future__ import annotations
from pathlib import Path
import pytest
import yt_dlp
from yt_scraper.config import Config
from yt_scraper.store import Store, VideoRef
from yt_scraper.webapp.jobs import JobManager
def test_audio_job_downloads_into_audio_directory(tmp_path, monkeypatch):
calls: list[list[str]] = []
class FakeYoutubeDL:
def __init__(self, options):
self.options = options
def __enter__(self):
return self
def __exit__(self, exc_type, exc_value, traceback):
return False
def download(self, urls):
calls.append(urls)
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("abc123", "UC1", "Audio test", "https://www.youtube.com/watch?v=abc123")
])
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_audio("audio-job", {"mode": "audio", "video_ids": ["abc123"]})
assert (tmp_path / "audio").is_dir()
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
assert store.get_job("audio-job").status == "done"
def test_audio_job_with_video_ids_uses_audio_runner(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.create_job("audio-job", None, {"mode": "audio", "video_ids": ["abc123"]})
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
calls: list[str] = []
monkeypatch.setattr(manager, "_run_audio", lambda job_id, opts: calls.append("audio"))
monkeypatch.setattr(manager, "_run_batch", lambda job_id, opts: calls.append("batch"))
manager._run_job("audio-job")
assert calls == ["audio"]
def test_video_job_downloads_webm_after_markdown_exists(tmp_path, monkeypatch):
calls: list[list[str]] = []
captured: dict = {}
class FakeYoutubeDL:
def __init__(self, options):
captured.update(options)
def __enter__(self):
return self
def __exit__(self, exc_type, exc_value, traceback):
return False
def download(self, urls):
calls.append(urls)
outtmpl = captured["outtmpl"].replace("%(ext)s", "webm")
output = tmp_path / Path(outtmpl).relative_to(tmp_path)
output.parent.mkdir(parents=True, exist_ok=True)
output.write_bytes(b"webm-test")
monkeypatch.setattr(yt_dlp, "YoutubeDL", FakeYoutubeDL)
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("abc123", "UC1", "A WebM video", "https://www.youtube.com/watch?v=abc123")
])
md = tmp_path / "markdown" / "Alpha" / "a-webm-video.md"
md.parent.mkdir(parents=True)
md.write_text("# A WebM video\n", encoding="utf-8")
store.mark_done("abc123", "markdown/Alpha/a-webm-video.md", "en", "manual", False)
store.create_job("video-job", None, {"mode": "video", "video_ids": ["abc123"]})
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_video("video-job", {"mode": "video", "video_ids": ["abc123"]})
row = store.get_video("abc123")
assert calls == [["https://www.youtube.com/watch?v=abc123"]]
assert captured["merge_output_format"] == "webm"
assert captured["js_runtimes"] == {"node": {}}
assert "protocol^=m3u8_native" in captured["format"]
assert row.video_download_status == "done"
assert row.video_filename == "A WebM video.webm"
assert row.video_path == "videos/abc123/A WebM video.webm"
assert (tmp_path / row.video_path).read_bytes() == b"webm-test"
def test_discovery_job_registers_new_videos_without_processing(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60)
])
store.mark_status("old", "done")
store.create_job("discover-job", "UC1", {"mode": "discover"})
calls: list[str] = []
def fake_discover(url, sleep_subrequests=2.0, limit=None):
calls.append((url, limit))
return ("UC1", "Alpha", None, [
VideoRef("new", "UC1", "New", "https://y/watch?v=new", "20240501", 90),
VideoRef("old", "UC1", "Old", "https://y/watch?v=old", "20240101", 60),
])
processed = []
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", lambda *a, **k: processed.append(a))
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("discover-job")
# Windowed, not a full walk: the known "old" video ends the scan in one pass.
assert calls == [("https://www.youtube.com/@alpha/videos", 30)]
assert store.get_video("new").status == "pending"
assert store.get_video("old").status == "done"
assert processed == []
assert store.get_job("discover-job").status == "done"
assert any(
event["event"] == "done" and event["data"]["new_videos"] == 1
for event in manager.events_since("discover-job", 0)
)
def test_all_channel_discovery_continues_after_one_error(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@one", "One", 0)
store.upsert_channel("UC2", "@two", "Two", 0)
store.create_job("discover-all", None, {"mode": "discover"})
def fake_discover(url, sleep_subrequests=2.0, limit=None):
if "@one" in url:
raise RuntimeError("temporary failure")
return ("UC2", "Two", None, [VideoRef("new2", "UC2", "New", "https://y/watch?v=new2")])
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake_discover)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("discover-all")
assert store.get_video("new2").status == "pending"
assert store.get_job("discover-all").status == "done"
done = [e for e in manager.events_since("discover-all", 0) if e["event"] == "done"][-1]
assert done["data"]["errors"] == 1
def _catalog_channel(monkeypatch, catalog, calls):
def fake(url, sleep_subrequests=2.0, limit=None):
calls.append(limit)
return ("UC1", "Alpha", None, catalog[:limit] if limit else list(catalog))
monkeypatch.setattr("yt_scraper.discover.discover_channel", fake)
def test_discovery_keeps_the_full_video_count_after_a_windowed_pass(tmp_path, monkeypatch):
"""The window is 30 wide but the channel has 200 videos — video_count must
not collapse to the size of what we just looked at."""
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 200)
catalog = [
VideoRef(f"v{i:03d}", "UC1", f"V{i}", f"https://y/watch?v=v{i:03d}", None, 60)
for i in range(200)
]
store.upsert_videos(catalog[2:]) # everything except the 2 newest
store.create_job("disc", "UC1", {"mode": "discover"})
calls: list = []
_catalog_channel(monkeypatch, catalog, calls)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("disc")
assert calls == [30], "must not paginate all 200"
assert store.get_channel("UC1")["video_count"] == 200
done = [e for e in manager.events_since("disc", 0) if e["event"] == "done"][-1]
assert done["data"]["new_videos"] == 2
assert done["data"]["fetched"] == 30
def test_discovery_full_option_walks_the_whole_channel(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 3)
catalog = [
VideoRef(f"v{i}", "UC1", f"V{i}", f"https://y/watch?v=v{i}", None, 60) for i in range(3)
]
store.upsert_videos(catalog)
store.create_job("disc-full", "UC1", {"mode": "discover", "full": True})
calls: list = []
_catalog_channel(monkeypatch, catalog, calls)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_job("disc-full")
assert calls == [None], "full rescan must not pass a playlistend"
done = [e for e in manager.events_since("disc-full", 0) if e["event"] == "done"][-1]
assert done["data"]["new_videos"] == 0
def test_batch_job_reports_videos_without_transcripts(tmp_path, monkeypatch):
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([
VideoRef("no-captions", "UC1", "No captions", "https://y/watch?v=no-captions")
])
store.create_job("batch-job", None, {"video_ids": ["no-captions"]})
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *args, **kwargs: "no_subtitles")
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_batch("batch-job", {"video_ids": ["no-captions"]})
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
assert done["data"]["processed"] == 0
assert done["data"]["no_subtitles"] == 1
assert store.get_video("no-captions").markdown_path is None
def test_batch_job_downloads_markdown_and_nothing_else(tmp_path, monkeypatch):
"""The .md button downloads notes only.
Thumbnails are already cached by the channel-level paths (add-channel, the
Thumbnails tool) and /api/thumbnails/{id} redirects to the CDN for whatever
is missing, so fetching one per video here spent a request on an image the
UI could already display.
"""
store = Store(tmp_path / "state.db")
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([VideoRef("vid1", "UC1", "One", "https://y/watch?v=vid1")])
store.create_job("batch-job", None, {"video_ids": ["vid1"]})
thumbs: list[str] = []
monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda *a, **k: "done")
monkeypatch.setattr(
"yt_scraper.pipeline.cache_thumbnail",
lambda _store, video_id, _dir: bool(thumbs.append(video_id)),
)
monkeypatch.setattr(
"yt_scraper._yt_http.yt_get",
lambda *a, **k: pytest.fail("a .md batch must not fetch images"),
)
manager = JobManager(
store,
Config(database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown")),
)
manager._run_batch("batch-job", {"video_ids": ["vid1"]})
assert thumbs == []
assert not (tmp_path / "thumbnails").exists()
done = [e for e in manager.events_since("batch-job", 0) if e["event"] == "done"][-1]
assert done["data"]["processed"] == 1