Files
yt-channel-scraper/tests/test_throttle_breaker.py
T

247 lines
9.2 KiB
Python

"""The circuit breaker, tested through the webapp job runner.
This is the regression test for the incident the whole change exists to
prevent: a session gets rate-limited, the runner keeps going anyway, and every
remaining video is marked failed. In production that turned one throttling
event into 511 `no_subtitles` and 347 `error` rows on a single channel.
The invariant asserted everywhere below is the same: **videos the run never
reached must still be `pending`**, because `pending` is what a later run picks
up. A video wrongly marked `error` needs a manual reset first.
"""
from __future__ import annotations
import pytest
from yt_scraper.config import Config, DelayConfig
from yt_scraper.store import Store, VideoRef
from yt_scraper.webapp.jobs import JobManager
THROTTLED = (
"ERROR: [youtube] {vid}: Video unavailable. This content isn't available, try "
"again later. The current session has been rate-limited by YouTube for up to an hour."
)
def _cfg(tmp_path) -> Config:
return Config(
database_path=str(tmp_path / "state.db"),
output_dir=str(tmp_path / "markdown"),
# No real waiting in tests; the breaker's arithmetic is unit-tested
# separately in test_ratelimit.py.
delay=DelayConfig(
min_seconds=0.0, max_seconds=0.0,
backoff_base=0.001, backoff_cap=0.002, throttle_threshold=3,
),
)
def _seed(store: Store, n: int) -> list[str]:
store.upsert_channel("UC1", "@alpha", "Alpha", n)
ids = [f"v{i:02d}" for i in range(n)]
store.upsert_videos([
VideoRef(v, "UC1", f"Video {i}", f"https://y/watch?v={v}") for i, v in enumerate(ids)
])
return ids
def _always_throttled(store: Store):
"""Stand-in for process_video that fails the way a throttled session does."""
def fake(row, cfg, st, *args, **kwargs):
st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id))
return "error"
return fake
@pytest.fixture
def manager(tmp_path):
store = Store(tmp_path / "state.db")
return JobManager(store, _cfg(tmp_path)), store
def test_batch_stops_and_leaves_the_rest_pending(manager, monkeypatch):
mgr, store = manager
ids = _seed(store, 20)
monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store))
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
touched = [v for v in ids if store.get_video(v).status != "pending"]
assert len(touched) == 3, "must stop at the threshold, not walk all 20"
assert all(store.get_video(v).status == "pending" for v in ids[3:])
job = store.get_job("job")
assert job.status == "error"
assert "rate-limit" in (job.last_error or "")
events = mgr.events_since("job", 0)
err = [e for e in events if e["event"] == "error"]
assert err and err[-1]["data"]["throttled"] is True
def test_channel_run_stops_and_leaves_the_rest_pending(manager, monkeypatch):
mgr, store = manager
ids = _seed(store, 20)
monkeypatch.setattr("yt_scraper.webapp.jobs.process_video", _always_throttled(store))
monkeypatch.setattr(
"yt_scraper.discover.discover_channel",
lambda url, sleep_subrequests=2.0, limit=None: ("UC1", "Alpha", None, []),
)
store.create_job("job", "UC1", {})
mgr._run_channel("job", {})
assert len([v for v in ids if store.get_video(v).status != "pending"]) == 3
assert store.get_job("job").status == "error"
def test_a_healthy_run_is_untouched_by_the_breaker(manager, monkeypatch):
"""The breaker must be invisible when nothing is throttling."""
mgr, store = manager
ids = _seed(store, 8)
monkeypatch.setattr("yt_scraper.pipeline.process_video",
lambda row, *a, **k: store.mark_status(row.video_id, "done") or "done")
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
assert store.get_job("job").status == "done"
assert all(store.get_video(v).status == "done" for v in ids)
def test_ordinary_failures_do_not_stop_the_run(manager, monkeypatch):
"""A channel with dead videos must still be processed to the end.
Without the throttle/permanent distinction, three members-only videos in a
row would abort a perfectly healthy scrape.
"""
mgr, store = manager
ids = _seed(store, 10)
def fake(row, cfg, st, *args, **kwargs):
st.mark_error(row.video_id, "ERROR: [youtube] x: Private video. Sign in if you've been granted access")
return "error"
monkeypatch.setattr("yt_scraper.pipeline.process_video", fake)
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
assert store.get_job("job").status == "done"
assert all(store.get_video(v).status == "error" for v in ids)
def test_isolated_throttling_between_successes_does_not_stop_the_run(manager, monkeypatch):
"""Only *consecutive* throttling means the session is banned."""
mgr, store = manager
ids = _seed(store, 12)
calls = {"n": 0}
def fake(row, cfg, st, *args, **kwargs):
calls["n"] += 1
if calls["n"] % 2:
st.mark_error(row.video_id, THROTTLED.format(vid=row.video_id))
return "error"
st.mark_status(row.video_id, "done")
return "done"
monkeypatch.setattr("yt_scraper.pipeline.process_video", fake)
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
assert store.get_job("job").status == "done"
assert calls["n"] == 12
def test_a_throttled_caption_fetch_is_an_error_not_no_subtitles(tmp_path, monkeypatch):
"""Which of the three requests YouTube refused must not decide the status.
A 429 on `extract_info` produced `error`; a 429 on the separate caption
download produced `no_subtitles` — the state that means "this video
publishes no captions", which is what the `no_subtitles` counts are read as.
Both are retryable, but only one is honest.
"""
from yt_scraper.extract import VideoData
from yt_scraper.pipeline import process_video
store = Store(tmp_path / "state.db")
cfg = _cfg(tmp_path)
_seed(store, 1)
row = store.get_video("v00")
throttled = VideoData(
info={"id": "v00", "title": "t"},
segments=[],
subtitle=None,
has_chapters=False,
skip_reason=(
"subtitle track found (lang=en, auto) but the download failed "
"[HTTP Error 429: HTTPError: 429 Client Error: Too Many Requests for url: ...]"
),
)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: throttled)
assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "error"
genuinely_absent = VideoData(
info={"id": "v00", "title": "t"}, segments=[], subtitle=None,
has_chapters=False, skip_reason="no caption tracks published for this video",
)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: genuinely_absent)
assert process_video(row, cfg, store, "Alpha", "UC1", "u") == "no_subtitles"
def test_metadata_survives_a_failed_caption_fetch(tmp_path, monkeypatch):
"""The extraction already paid for this data; a retry must not re-buy it."""
from yt_scraper.extract import VideoData
from yt_scraper.pipeline import process_video
store = Store(tmp_path / "state.db")
cfg = _cfg(tmp_path)
_seed(store, 1)
row = store.get_video("v00")
monkeypatch.setattr(
"yt_scraper.pipeline.extract_video",
lambda *a, **k: VideoData(
info={"id": "v00", "title": "t", "view_count": 4321, "upload_date": "20260101",
"thumbnail": "https://i.ytimg.com/x.jpg", "description": "hola"},
segments=[], subtitle=None, has_chapters=False,
skip_reason="no caption tracks published for this video",
),
)
process_video(row, cfg, store, "Alpha", "UC1", "u")
after = store.get_video("v00")
assert after.status == "no_subtitles"
assert after.view_count == 4321, "metadata was discarded with the failed transcript"
assert after.upload_date == "20260101"
def test_throttled_videos_stay_retryable(manager, monkeypatch):
"""The three videos that did fail must not be classified as permanent.
`Store.PERMANENT_ERROR_PATTERNS` deliberately excludes the throttling
message; if that ever changed, a rate-limit incident would poison rows that
a later run could have recovered.
"""
mgr, store = manager
ids = _seed(store, 10)
monkeypatch.setattr("yt_scraper.pipeline.process_video", _always_throttled(store))
monkeypatch.setattr("yt_scraper.pipeline.cache_thumbnail", lambda *a, **k: False)
store.create_job("job", None, {"video_ids": ids})
mgr._run_batch("job", {"video_ids": ids})
counts = store.retryable_counts("UC1")
assert counts.get("permanent", 0) == 0
assert store.reset_videos("UC1", ("error",)) == 3