feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo
Extraccion autenticada: - import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20 - extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla con sesion logueada (members-only); regex y opener cacheados - js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts - rutas de Brave multiplataforma (Windows/macOS/Linux) Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL, memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics Rendimiento: - entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota) - _rank_unranked con guarda (antes full-scan en cada arranque/import) - upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL) - thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool) - reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md), healthz expone reconcile_done - Store.transaction(): escrituras por video agrupadas (~6 commits -> 3) Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render, order_pending->store, keep_ref->config, seconds_to_ts solo en segments); re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done); fuera wrappers muertos de segments.py
This commit is contained in:
@@ -0,0 +1,88 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from yt_scraper.extract import extract_via_watch_page
|
||||
|
||||
|
||||
WATCH_PAGE = (
|
||||
"var ytInitialPlayerResponse = "
|
||||
+ json.dumps({
|
||||
"playabilityStatus": {"status": "OK"},
|
||||
"videoDetails": {
|
||||
"title": "Members only test",
|
||||
"author": "Canal",
|
||||
"lengthSeconds": "42",
|
||||
"viewCount": "7",
|
||||
"shortDescription": "desc",
|
||||
"keywords": ["a", "b"],
|
||||
},
|
||||
"microformat": {"playerMicroformatRenderer": {"publishDate": "2026-09-01"}},
|
||||
"captions": {"playerCaptionsTracklistRenderer": {"captionTracks": [
|
||||
{"languageCode": "es", "kind": "asr", "baseUrl": "https://captions.test/es"},
|
||||
]}},
|
||||
})
|
||||
+ ";"
|
||||
)
|
||||
|
||||
JSON3 = json.dumps({
|
||||
"events": [
|
||||
{"tStartMs": 0, "dDurationMs": 1000, "segs": [{"utf8": "hola "}, {"utf8": "mundo"}]},
|
||||
{"tStartMs": 1000, "dDurationMs": 500, "segs": [{"utf8": "segundo"}]},
|
||||
]
|
||||
})
|
||||
|
||||
|
||||
class _FakeResponse(io.BytesIO):
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
|
||||
def test_watch_page_fallback_builds_segments(tmp_path: Path, monkeypatch):
|
||||
from yt_scraper import extract as ex
|
||||
|
||||
def fake_urlopen(cookies_file, url, **kw):
|
||||
if url.startswith("https://www.youtube.com/"):
|
||||
return _FakeResponse(WATCH_PAGE.encode("utf-8"))
|
||||
assert "fmt=json3" in url
|
||||
return _FakeResponse(JSON3.encode("utf-8"))
|
||||
|
||||
monkeypatch.setattr(ex, "_session_urlopen", fake_urlopen)
|
||||
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
|
||||
|
||||
data = extract_via_watch_page(
|
||||
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
|
||||
)
|
||||
|
||||
assert data is not None
|
||||
assert data.subtitle is not None and data.subtitle.lang == "es"
|
||||
# parse_auto_dump merges caption events with no gap between them.
|
||||
assert len(data.segments) == 1
|
||||
assert data.segments[0].text == "hola mundo segundo"
|
||||
assert data.info["title"] == "Members only test"
|
||||
assert data.info["upload_date"] == "20260901"
|
||||
assert data.info["duration"] == 42
|
||||
assert data.skip_reason is None
|
||||
|
||||
|
||||
def test_watch_page_fallback_without_tracks_returns_none(tmp_path: Path, monkeypatch):
|
||||
from yt_scraper import extract as ex
|
||||
|
||||
page = WATCH_PAGE.replace('"captionTracks": [{', '"captionTracks": [{')
|
||||
page = "var ytInitialPlayerResponse = " + json.dumps({
|
||||
"playabilityStatus": {"status": "OK"},
|
||||
"videoDetails": {"title": "x"},
|
||||
}) + ";"
|
||||
|
||||
monkeypatch.setattr(ex, "_session_urlopen",
|
||||
lambda *a, **kw: _FakeResponse(page.encode("utf-8")))
|
||||
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
|
||||
|
||||
assert extract_via_watch_page(
|
||||
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
|
||||
) is None
|
||||
Reference in New Issue
Block a user