wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)
This commit is contained in:
@@ -0,0 +1,171 @@
|
||||
"""Guards on how many requests we spend and under what option names.
|
||||
|
||||
Two classes of regression live here, both of which actually happened:
|
||||
|
||||
1. An option name yt-dlp does not recognise. yt-dlp ignores unknown keys
|
||||
silently, so `sleep_subrequests` looked configured for the project's whole
|
||||
history while nothing ever slept between requests. `test_*_options_are_real`
|
||||
checks every key against yt-dlp's own list instead of trusting review.
|
||||
|
||||
2. An extraction that walks far more of a channel than it needs.
|
||||
`deep_channel_avatar` read one avatar URL by fully extracting every video the
|
||||
channel had ever published — 735 requests and still going when a measurement
|
||||
aborted it. The bound is asserted here because nothing else would notice.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
import yt_dlp
|
||||
|
||||
from yt_scraper import discover, extract
|
||||
|
||||
#: Every option name yt-dlp's own CLI parser produces. Anything outside this is
|
||||
#: either a typo or something yt-dlp will silently drop.
|
||||
KNOWN_YDL_OPTIONS = set(yt_dlp.parse_options([]).ydl_opts)
|
||||
|
||||
|
||||
class CapturingYDL:
|
||||
"""Stands in for yt_dlp.YoutubeDL and records the options it was built with."""
|
||||
|
||||
captured: list[dict] = []
|
||||
info: dict = {}
|
||||
|
||||
def __init__(self, options=None, *args, **kwargs):
|
||||
type(self).captured.append(dict(options or {}))
|
||||
self.options = options or {}
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc):
|
||||
return False
|
||||
|
||||
def extract_info(self, url, download=False, process=True):
|
||||
return dict(type(self).info)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def capture(monkeypatch):
|
||||
CapturingYDL.captured = []
|
||||
CapturingYDL.info = {
|
||||
"id": "UC123",
|
||||
"channel_id": "UC123",
|
||||
"channel": "Test Channel",
|
||||
"thumbnails": [{"url": "https://yt3.ggpht.com/avatar.jpg"}],
|
||||
"entries": [],
|
||||
}
|
||||
monkeypatch.setattr(yt_dlp, "YoutubeDL", CapturingYDL)
|
||||
return CapturingYDL
|
||||
|
||||
|
||||
def assert_options_are_real(opts: dict, where: str) -> None:
|
||||
unknown = sorted(set(opts) - KNOWN_YDL_OPTIONS)
|
||||
assert not unknown, (
|
||||
f"{where} passes option(s) yt-dlp does not recognise and will silently "
|
||||
f"ignore: {unknown}"
|
||||
)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ names
|
||||
|
||||
|
||||
def test_discover_channel_options_are_real(capture):
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
|
||||
assert_options_are_real(capture.captured[0], "discover_channel")
|
||||
|
||||
|
||||
def test_deep_channel_avatar_options_are_real(capture):
|
||||
discover.deep_channel_avatar("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
|
||||
assert_options_are_real(capture.captured[0], "deep_channel_avatar")
|
||||
|
||||
|
||||
def test_extract_video_options_are_real(capture):
|
||||
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
|
||||
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
|
||||
assert_options_are_real(capture.captured[0], "extract_video")
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ throttles wired
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"call",
|
||||
[
|
||||
pytest.param(
|
||||
lambda: discover.discover_channel("https://www.youtube.com/@x/videos",
|
||||
sleep_subrequests=3.25),
|
||||
id="discover_channel",
|
||||
),
|
||||
pytest.param(
|
||||
lambda: discover.deep_channel_avatar("https://www.youtube.com/@x/videos",
|
||||
sleep_subrequests=3.25),
|
||||
id="deep_channel_avatar",
|
||||
),
|
||||
pytest.param(
|
||||
lambda: extract.extract_video("https://www.youtube.com/watch?v=vid",
|
||||
{"es": "any"}, sleep_subrequests=3.25),
|
||||
id="extract_video",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_every_entry_point_forwards_the_real_sleep_option(capture, call):
|
||||
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {},
|
||||
"channel_id": "UC1", "entries": []}
|
||||
call()
|
||||
opts = capture.captured[0]
|
||||
assert opts.get("sleep_interval_requests") == 3.25
|
||||
assert opts.get("socket_timeout"), "a hung connection must not block the worker forever"
|
||||
assert "extractor_retries" in opts
|
||||
|
||||
|
||||
# ------------------------------------------------------------------ request bounds
|
||||
|
||||
|
||||
def test_deep_avatar_does_not_walk_the_channel(capture):
|
||||
"""The 735-request bug. Both halves of the fix are asserted.
|
||||
|
||||
`extract_flat` stops yt-dlp expanding each entry into a full extraction, and
|
||||
`playlistend` stops it paginating past the first page. Either one missing
|
||||
puts the whole channel back on the wire.
|
||||
"""
|
||||
discover.deep_channel_avatar("https://www.youtube.com/@x/videos")
|
||||
opts = capture.captured[0]
|
||||
assert opts.get("extract_flat"), "must not fully extract every video"
|
||||
assert opts.get("playlistend") == 1, "must not paginate beyond the first page"
|
||||
|
||||
|
||||
def test_discover_channel_limit_becomes_playlistend(capture):
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
|
||||
assert capture.captured[0].get("playlistend") == 30
|
||||
|
||||
|
||||
def test_discover_channel_without_limit_has_no_ceiling(capture):
|
||||
"""A brand-new channel legitimately walks everything; that must stay possible."""
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos")
|
||||
assert "playlistend" not in capture.captured[0]
|
||||
|
||||
|
||||
def test_extraction_never_probes_formats(capture):
|
||||
"""`check_formats` costs one HTTP request per format and we only want captions."""
|
||||
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
|
||||
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
|
||||
assert capture.captured[0].get("check_formats") is None
|
||||
|
||||
|
||||
def test_discovery_processes_the_result_so_playlistend_applies(capture, monkeypatch):
|
||||
"""`playlistend` is silently ignored when extract_info runs with process=False.
|
||||
|
||||
Measured on a 2564-video channel: processed + playlistend=60 costs 2
|
||||
requests; unprocessed, the lazy generator ignores the limit and walking it
|
||||
costs 86. Nothing else in the codebase would catch that flip.
|
||||
"""
|
||||
seen = {}
|
||||
|
||||
def extract_info(self, url, download=False, process=True):
|
||||
seen["process"] = process
|
||||
return dict(CapturingYDL.info)
|
||||
|
||||
monkeypatch.setattr(CapturingYDL, "extract_info", extract_info)
|
||||
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
|
||||
assert seen["process"] is not False
|
||||
Reference in New Issue
Block a user