"""The circuit breaker, tested through the webapp job runner. This is the regression test for the incident the whole change exists to prevent: a session gets rate-limited, the runner keeps going anyway, and every remaining video is marked failed. In production that turned one throttling event into 511 `no_subtitles` and 347 `error` rows on a single channel. The invariant asserted everywhere below is the same: **videos the run never reached must still be `pending`**, because `pending` is what a later run picks up. A video wrongly marked `error` needs a manual reset first. """ from __future__ import annotations import pytest from yt_scraper.config import Config, DelayConfig from yt_scraper.store import Store, VideoRef from yt_scraper.webapp.jobs import JobManager THROTTLED = ( "ERROR: [youtube] {vid}: Video unavailable. This content isn't available, try " "again later. The current session has been rate-limited by YouTube for up to an hour." ) def _cfg(tmp_path) -> Config: return Config( database_path=str(tmp_path / "state.db"), output_dir=str(tmp_path / "markdown"), # No real waiting in tests; the breaker's arithmetic is unit-tested # separately in test_ratelimit.py. delay=DelayConfig( min_seconds=0.0, max_seconds=0.0, backoff_base=0.001, backoff_cap=0.002, throttle_threshold=3, ), ) def _seed(store: Store, n: int) -> list[str]: store.upsert_channel("UC1", "@alpha", "Alpha", n) ids = [f"v{i:02d}" for i in range(n)] store.upsert_videos([ VideoRef(v, "UC1", f"Video {i}", f"https://y/watch?v={v}") for i, v in enumerate(ids) ]) return ids def _always_throttled(store: Store): """Stand-in for process_video that fails the way a throttled session does.""" def fake(row, cfg, st, *args, **kwargs): st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id)) return "error" return fake @pytest.fixture def manager(tmp_path): store = Store(tmp_path / "state.db") return JobManager(store, _cfg(tmp_path)), store def test_batch_stops_and_leaves_the_rest_pending(manager, monkeypatch): mgr, store = manager ids = _seed(store, 20) monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store)) monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False) store.create_job("job", None, {"video_ids": ids}) mgr._run_batch("job", {"video_ids": ids}) touched = [v for v in ids if store.get_video(v).status != "pending"] assert len(touched) == 3, "must stop at the threshold, not walk all 20" assert all(store.get_video(v).status == "pending" for v in ids[3:]) job = store.get_job("job") assert job.status == "error" assert "rate-limit" in (job.last_error or "") events = mgr.events_since("job", 0) err = [e for e in events if e["event"] == "error"] assert err and err[-1]["data"]["throttled"] is True def test_channel_run_stops_and_leaves_the_rest_pending(manager, monkeypatch): mgr, store = manager ids = _seed(store, 20) monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", _always_throttled(store)) monkeypatch.setattr( "yt_scraper.discover.discover_channel", lambda url, sleep_subrequests=2.0, limit=None: ("UC1", "Alpha", None, []), ) store.create_job("job", "UC1", {}) mgr._run_channel("job", {}) assert len([v for v in ids if store.get_video(v).status != "pending"]) == 3 assert store.get_job("job").status == "error" def test_a_healthy_run_is_untouched_by_the_breaker(manager, monkeypatch): """The breaker must be invisible when nothing is throttling.""" mgr, store = manager ids = _seed(store, 8) monkeypatch.setattr("yt_scraper.pipeline.process_video", lambda row, *a, **k: store.mark_status(row.video_id, "done") or "done") monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False) store.create_job("job", None, {"video_ids": ids}) mgr._run_batch("job", {"video_ids": ids}) assert store.get_job("job").status == "done" assert all(store.get_video(v).status == "done" for v in ids) def test_ordinary_failures_do_not_stop_the_run(manager, monkeypatch): """A channel with dead videos must still be processed to the end. Without the throttle/permanent distinction, three members-only videos in a row would abort a perfectly healthy scrape. """ mgr, store = manager ids = _seed(store, 10) def fake(row, cfg, st, *args, **kwargs): st.mark_error(row.video_id, "ERROR: [youtube] x: Private video. Sign in if you've been granted access") return "error" monkeypatch.setattr("yt_scraper.pipeline.process_video", fake) monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False) store.create_job("job", None, {"video_ids": ids}) mgr._run_batch("job", {"video_ids": ids}) assert store.get_job("job").status == "done" assert all(store.get_video(v).status == "error" for v in ids) def test_isolated_throttling_between_successes_does_not_stop_the_run(manager, monkeypatch): """Only *consecutive* throttling means the session is banned.""" mgr, store = manager ids = _seed(store, 12) calls = {"n": 0} def fake(row, cfg, st, *args, **kwargs): calls["n"] += 1 if calls["n"] % 2: st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id)) return "error" st.mark_status(row.video_id, "done") return "done" monkeypatch.setattr("yt_scraper.pipeline.process_video", fake) monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False) store.create_job("job", None, {"video_ids": ids}) mgr._run_batch("job", {"video_ids": ids}) assert store.get_job("job").status == "done" assert calls["n"] == 12 def test_a_throttled_caption_fetch_is_an_error_not_no_subtitles(tmp_path, monkeypatch): """Which of the three requests YouTube refused must not decide the status. A 429 on `extract_info` produced `error`; a 429 on the separate caption download produced `no_subtitles` — the state that means "this video publishes no captions", which is what the `no_subtitles` counts are read as. Both are retryable, but only one is honest. """ from yt_scraper.extract import VideoData from yt_scraper.pipeline import process_video store = Store(tmp_path / "state.db") cfg = _cfg(tmp_path) _seed(store, 1) row = store.get_video("v00") throttled = VideoData( info={"id": "v00", "title": "t"}, segments=[], subtitle=None, has_chapters=False, skip_reason=( "subtitle track found (lang=en, auto) but the download failed " "[HTTP Error 429: HTTPError: 429 Client Error: Too Many Requests for url: ...]" ), ) monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: throttled) assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "error" genuinely_absent = VideoData( info={"id": "v00", "title": "t"}, segments=[], subtitle=None, has_chapters=False, skip_reason="no caption tracks published for this video", ) monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: genuinely_absent) assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "no_subtitles" def test_metadata_survives_a_failed_caption_fetch(tmp_path, monkeypatch): """The extraction already paid for this data; a retry must not re-buy it.""" from yt_scraper.extract import VideoData from yt_scraper.pipeline import process_video store = Store(tmp_path / "state.db") cfg = _cfg(tmp_path) _seed(store, 1) row = store.get_video("v00") monkeypatch.setattr( "yt_scraper.pipeline.extract_video", lambda *a, **k: VideoData( info={"id": "v00", "title": "t", "view_count": 4321, "upload_date": "20260101", "thumbnail": "https://i.ytimg.com/x.jpg", "description": "hola"}, segments=[], subtitle=None, has_chapters=False, skip_reason="no caption tracks published for this video", ), ) process_video(row, cfg, store, "Alpha", "UC1", "u") after = store.get_video("v00") assert after.status == "no_subtitles" assert after.view_count == 4321, "metadata was discarded with the failed transcript" assert after.upload_date == "20260101" def test_throttled_videos_stay_retryable(manager, monkeypatch): """The three videos that did fail must not be classified as permanent. `Store.PERMANENT_ERROR_PATTERNS` deliberately excludes the throttling message; if that ever changed, a rate-limit incident would poison rows that a later run could have recovered. """ mgr, store = manager ids = _seed(store, 10) monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store)) monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False) store.create_job("job", None, {"video_ids": ids}) mgr._run_batch("job", {"video_ids": ids}) counts = store.retryable_counts("UC1") assert counts.get("permanent", 0) == 0 assert store.reset_videos("UC1", ("error",)) == 3