wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)

This commit is contained in:
urieljareth
2026-08-22 19:37:23 -06:00
parent 4f5a68b572
commit 8a59b39c98
103 changed files with 70954 additions and 1825 deletions
+171
View File
@@ -0,0 +1,171 @@
"""Guards on how many requests we spend and under what option names.
Two classes of regression live here, both of which actually happened:
1. An option name yt-dlp does not recognise. yt-dlp ignores unknown keys
silently, so `sleep_subrequests` looked configured for the project's whole
history while nothing ever slept between requests. `test_*_options_are_real`
checks every key against yt-dlp's own list instead of trusting review.
2. An extraction that walks far more of a channel than it needs.
`deep_channel_avatar` read one avatar URL by fully extracting every video the
channel had ever published — 735 requests and still going when a measurement
aborted it. The bound is asserted here because nothing else would notice.
"""
from __future__ import annotations
import pytest
import yt_dlp
from yt_scraper import discover, extract
#: Every option name yt-dlp's own CLI parser produces. Anything outside this is
#: either a typo or something yt-dlp will silently drop.
KNOWN_YDL_OPTIONS = set(yt_dlp.parse_options([]).ydl_opts)
class CapturingYDL:
"""Stands in for yt_dlp.YoutubeDL and records the options it was built with."""
captured: list[dict] = []
info: dict = {}
def __init__(self, options=None, *args, **kwargs):
type(self).captured.append(dict(options or {}))
self.options = options or {}
def __enter__(self):
return self
def __exit__(self, *exc):
return False
def extract_info(self, url, download=False, process=True):
return dict(type(self).info)
@pytest.fixture
def capture(monkeypatch):
CapturingYDL.captured = []
CapturingYDL.info = {
"id": "UC123",
"channel_id": "UC123",
"channel": "Test Channel",
"thumbnails": [{"url": "https://yt3.ggpht.com/avatar.jpg"}],
"entries": [],
}
monkeypatch.setattr(yt_dlp, "YoutubeDL", CapturingYDL)
return CapturingYDL
def assert_options_are_real(opts: dict, where: str) -> None:
unknown = sorted(set(opts) - KNOWN_YDL_OPTIONS)
assert not unknown, (
f"{where} passes option(s) yt-dlp does not recognise and will silently "
f"ignore: {unknown}"
)
# ------------------------------------------------------------------ names
def test_discover_channel_options_are_real(capture):
discover.discover_channel("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
assert_options_are_real(capture.captured[0], "discover_channel")
def test_deep_channel_avatar_options_are_real(capture):
discover.deep_channel_avatar("https://www.youtube.com/@x/videos", sleep_subrequests=2.0)
assert_options_are_real(capture.captured[0], "deep_channel_avatar")
def test_extract_video_options_are_real(capture):
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
assert_options_are_real(capture.captured[0], "extract_video")
# ------------------------------------------------------------------ throttles wired
@pytest.mark.parametrize(
"call",
[
pytest.param(
lambda: discover.discover_channel("https://www.youtube.com/@x/videos",
sleep_subrequests=3.25),
id="discover_channel",
),
pytest.param(
lambda: discover.deep_channel_avatar("https://www.youtube.com/@x/videos",
sleep_subrequests=3.25),
id="deep_channel_avatar",
),
pytest.param(
lambda: extract.extract_video("https://www.youtube.com/watch?v=vid",
{"es": "any"}, sleep_subrequests=3.25),
id="extract_video",
),
],
)
def test_every_entry_point_forwards_the_real_sleep_option(capture, call):
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {},
"channel_id": "UC1", "entries": []}
call()
opts = capture.captured[0]
assert opts.get("sleep_interval_requests") == 3.25
assert opts.get("socket_timeout"), "a hung connection must not block the worker forever"
assert "extractor_retries" in opts
# ------------------------------------------------------------------ request bounds
def test_deep_avatar_does_not_walk_the_channel(capture):
"""The 735-request bug. Both halves of the fix are asserted.
`extract_flat` stops yt-dlp expanding each entry into a full extraction, and
`playlistend` stops it paginating past the first page. Either one missing
puts the whole channel back on the wire.
"""
discover.deep_channel_avatar("https://www.youtube.com/@x/videos")
opts = capture.captured[0]
assert opts.get("extract_flat"), "must not fully extract every video"
assert opts.get("playlistend") == 1, "must not paginate beyond the first page"
def test_discover_channel_limit_becomes_playlistend(capture):
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
assert capture.captured[0].get("playlistend") == 30
def test_discover_channel_without_limit_has_no_ceiling(capture):
"""A brand-new channel legitimately walks everything; that must stay possible."""
discover.discover_channel("https://www.youtube.com/@x/videos")
assert "playlistend" not in capture.captured[0]
def test_extraction_never_probes_formats(capture):
"""`check_formats` costs one HTTP request per format and we only want captions."""
capture.info = {"id": "vid", "title": "t", "subtitles": {}, "automatic_captions": {}}
extract.extract_video("https://www.youtube.com/watch?v=vid", {"es": "any"})
assert capture.captured[0].get("check_formats") is None
def test_discovery_processes_the_result_so_playlistend_applies(capture, monkeypatch):
"""`playlistend` is silently ignored when extract_info runs with process=False.
Measured on a 2564-video channel: processed + playlistend=60 costs 2
requests; unprocessed, the lazy generator ignores the limit and walking it
costs 86. Nothing else in the codebase would catch that flip.
"""
seen = {}
def extract_info(self, url, download=False, process=True):
seen["process"] = process
return dict(CapturingYDL.info)
monkeypatch.setattr(CapturingYDL, "extract_info", extract_info)
discover.discover_channel("https://www.youtube.com/@x/videos", limit=30)
assert seen["process"] is not False