"""Guards on how many requests we spend and under what option names. Two classes of regression live here, both of which actually happened: 1. An option name yt-dlp does not recognise. yt-dlp ignores unknown keys silently, so `sleep_subrequests` looked configured for the project's whole history while nothing ever slept between requests. `test_*_options_are_real` checks every key against yt-dlp's own list instead of trusting review. 2. An extraction that walks far more of a channel than it needs. `deep_channel_avatar` read one avatar URL by fully extracting every video the channel had ever published — 735 requests and still going when a measurement aborted it. The bound is asserted here because nothing else would notice. """ from __future__ import annotations import pytest import yt_dlp from yt_scraper import discover, extract #: Every option name yt-dlp's own CLI parser produces. Anything outside this is #: either a typo or something yt-dlp will silently drop. KNOWN_YDL_OPTIONS = set(yt_dlp.parse_options([]).ydl_opts) class CapturingYDL: """Stands in for yt_dlp.YoutubeDL and records the options it was built with.""" captured: list[dict] = [] info: dict = {} def __init__(self, options=None, *args, **kwargs): type(self).captured.append(dict(options or {})) self.options = options or {} def __enter__(self): return self def __exit__(self, *exc): return False def extract_info(self, url, download=False, process=True): return dict(type(self).info) @pytest.fixture def capture(monkeypatch): CapturingYDL.captured = [] CapturingYDL.info = { "id": "UC123", "channel_id": "UC123", "channel": "Test Channel", "thumbnails": [{"url": "https://yt3.ggpht.com/avatar.jpg"}], "entries": [], } monkeypatch.setattr(yt_dlp, "YoutubeDL", CapturingYDL) return CapturingYDL def assert_options_are_real(opts: dict, where: str) -> None: unknown = sorted(set(opts) - KNOWN_YDL_OPTIONS) assert not unknown, ( f"{where} passes option(s) yt-dlp does not recognise and will silently " f"ignore: {unknown}" ) # ------------------------------------------------------------------ names def test_discover_channel_options_are_real(capture): discover.discover_channel("https://www.youtube.com/@x/videos", sleep_subrequests=2.0) assert_options_are_real(capture.captured[0], "discover_channel") def test_deep_channel_avatar_options_are_real(capture): discover.deep_channel_avatar("https://www.youtube.com/@x/videos", sleep_subrequests=2.0) assert_options_are_real(capture.captured[0], "deep_channel_avatar") def test_extract_video_options_are_real(capture): capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}} extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"}) assert_options_are_real(capture.captured[0], "extract_video") # ------------------------------------------------------------------ throttles wired @pytest.mark.parametrize( "call", [ pytest.param( lambda: discover.discover_channel("https://www.youtube.com/@x/videos", sleep_subrequests=3.25), id="discover_channel", ), pytest.param( lambda: discover.deep_channel_avatar("https://www.youtube.com/@x/videos", sleep_subrequests=3.25), id="deep_channel_avatar", ), pytest.param( lambda: extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"}, sleep_subrequests=3.25), id="extract_video", ), ], ) def test_every_entry_point_forwards_the_real_sleep_option(capture, call): capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}, "channel_id": "UC1", "entries": []} call() opts = capture.captured[0] assert opts.get("sleep_interval_requests") == 3.25 assert opts.get("socket_timeout"), "a hung connection must not block the worker forever" assert "extractor_retries" in opts # ------------------------------------------------------------------ request bounds def test_deep_avatar_does_not_walk_the_channel(capture): """The 735-request bug. Both halves of the fix are asserted. `extract_flat` stops yt-dlp expanding each entry into a full extraction, and `playlistend` stops it paginating past the first page. Either one missing puts the whole channel back on the wire. """ discover.deep_channel_avatar("https://www.youtube.com/@x/videos") opts = capture.captured[0] assert opts.get("extract_flat"), "must not fully extract every video" assert opts.get("playlistend") == 1, "must not paginate beyond the first page" def test_discover_channel_limit_becomes_playlistend(capture): discover.discover_channel("https://www.youtube.com/@x/videos", limit=30) assert capture.captured[0].get("playlistend") == 30 def test_discover_channel_without_limit_has_no_ceiling(capture): """A brand-new channel legitimately walks everything; that must stay possible.""" discover.discover_channel("https://www.youtube.com/@x/videos") assert "playlistend" not in capture.captured[0] def test_extraction_never_probes_formats(capture): """`check_formats` costs one HTTP request per format and we only want captions.""" capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}} extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"}) assert capture.captured[0].get("check_formats") is None def test_discovery_processes_the_result_so_playlistend_applies(capture, monkeypatch): """`playlistend` is silently ignored when extract_info runs with process=False. Measured on a 2564-video channel: processed + playlistend=60 costs 2 requests; unprocessed, the lazy generator ignores the limit and walking it costs 86. Nothing else in the codebase would catch that flip. """ seen = {} def extract_info(self, url, download=False, process=True): seen["process"] = process return dict(CapturingYDL.info) monkeypatch.setattr(CapturingYDL, "extract_info", extract_info) discover.discover_channel("https://www.youtube.com/@x/videos", limit=30) assert seen["process"] is not False