feat: cookies desde navegador, fallback watch-page y optimizacion integral del nucleo
Extraccion autenticada: - import_from_browser (Brave) con fallback CDP headless para cookies app-bound v20 - extract_via_watch_page: GET plano + ytInitialPlayerResponse cuando yt-dlp falla con sesion logueada (members-only); regex y opener cacheados - js_runtimes (node/deno/bun/quickjs) propagado a todos los ydl_opts - rutas de Brave multiplataforma (Windows/macOS/Linux) Webapp UX: chips de filtros removibles, skeleton loaders, estado de vista en URL, memoria de scroll, copyMd/openMd, import de cookies desde navegador, no-cache de statics Rendimiento: - entorno Jinja2 cacheado por directorio de plantilla (antes 1 por nota) - _rank_unranked con guarda (antes full-scan en cada arranque/import) - upsert_videos con executemany; dashboard sin N+1 (GROUP BY + conteo de tags en SQL) - thumbnails en paralelo (6 hilos, CDN ytimg); handlers bloqueantes -> def (threadpool) - reconcile de arranque en hilo daemon: uvicorn arriba al instante (0.95s con 1503 md), healthz expone reconcile_done - Store.transaction(): escrituras por video agrupadas (~6 commits -> 3) Refactor: helpers unicos (extract_handle->discover, safe_dirname/filename->render, order_pending->store, keep_ref->config, seconds_to_ts solo en segments); re-render del CLI delega en pipeline.re_render_videos (retira huerfanos y marca done); fuera wrappers muertos de segments.py
This commit is contained in:
@@ -0,0 +1,152 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
|
||||
from yt_scraper.cookies import (
|
||||
BrowserCookieLockedError,
|
||||
auto_import_dir,
|
||||
delete,
|
||||
import_from_browser,
|
||||
is_expired,
|
||||
parse_netscape,
|
||||
resolve_active_path,
|
||||
)
|
||||
from yt_scraper.store import CookieRow, Store
|
||||
|
||||
|
||||
SAMPLE_NETSCAPE = """# Netscape HTTP Cookie File
|
||||
.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSID\tsample_sid_token
|
||||
.youtube.com\tTRUE\t/\tTRUE\t2000000000\tLOGIN_INFO\tsample_login_info
|
||||
"""
|
||||
|
||||
|
||||
def test_parse_netscape():
|
||||
ok, info = parse_netscape(SAMPLE_NETSCAPE)
|
||||
assert ok is True
|
||||
assert info["count"] == 2
|
||||
assert info["has_session"] is True
|
||||
assert info["expires_at"] is not None
|
||||
|
||||
|
||||
def test_parse_netscape_httponly_lines_are_data():
|
||||
# SID/HSID are HttpOnly; Netscape exports prefix those lines and a
|
||||
# comment-skipping parser would silently strip the login out.
|
||||
text = (
|
||||
"# Netscape HTTP Cookie File\n"
|
||||
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSID\ttok\n"
|
||||
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tHSID\ttok\n"
|
||||
"#HttpOnly_.youtube.com\tTRUE\t/\tTRUE\t2000000000\tSSID\ttok\n"
|
||||
"# a real comment\n"
|
||||
)
|
||||
ok, info = parse_netscape(text)
|
||||
assert ok is True
|
||||
assert info["count"] == 3
|
||||
assert info["has_session"] is True
|
||||
|
||||
|
||||
def test_partial_session_export_is_not_a_session():
|
||||
# A lone __Secure-3PSID (partial extension export) is anonymous to
|
||||
# YouTube; it must not be reported as a usable session.
|
||||
text = (
|
||||
"# Netscape HTTP Cookie File\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t2000000000\t__Secure-3PSID\ttok\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t2000000000\t__Secure-3PAPISID\ttok\n"
|
||||
)
|
||||
ok, info = parse_netscape(text)
|
||||
assert ok is True
|
||||
assert info["has_session"] is False
|
||||
|
||||
|
||||
def _browser_cookie(name, value, *, domain=".youtube.com", httponly=False):
|
||||
import http.cookiejar
|
||||
return http.cookiejar.Cookie(
|
||||
version=0, name=name, value=value, port=None, port_specified=False,
|
||||
domain=domain, domain_specified=True, domain_initial_dot=domain.startswith("."),
|
||||
path="/", path_specified=True, secure=True, expires=2000000000,
|
||||
discard=False, comment=None, comment_url=None,
|
||||
rest={"HttpOnly": None} if httponly else {},
|
||||
)
|
||||
|
||||
|
||||
def test_import_from_browser_writes_vault_file(tmp_path: Path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
cookie_dir = tmp_path / "cookies"
|
||||
|
||||
def fake_extract(browser, profile=None, logger=None, **kw):
|
||||
return iter([
|
||||
_browser_cookie("SID", "sid_tok", httponly=True),
|
||||
_browser_cookie("HSID", "hsid_tok", httponly=True),
|
||||
_browser_cookie("SSID", "ssid_tok", httponly=True),
|
||||
_browser_cookie("__Secure-3PSID", "psid_tok"),
|
||||
_browser_cookie("PREF", "pref_tok", domain=".google.com"), # not youtube -> dropped
|
||||
])
|
||||
|
||||
import yt_dlp.cookies as ydl_cookies
|
||||
monkeypatch.setattr(ydl_cookies, "extract_cookies_from_browser", fake_extract)
|
||||
|
||||
cid = import_from_browser(store, browser="brave", cookie_dir=cookie_dir)
|
||||
row = store.get_cookie(cid)
|
||||
assert row is not None
|
||||
assert row.cookie_count == 4
|
||||
assert row.has_session is True
|
||||
|
||||
# The written file must round-trip as a valid Netscape file whose
|
||||
# HttpOnly session cookies survive the vault's own parser.
|
||||
from yt_scraper.cookies import parse_netscape_file
|
||||
path = cookie_dir / row.filename
|
||||
ok, info = parse_netscape_file(path)
|
||||
assert ok is True
|
||||
assert info["count"] == 4
|
||||
assert info["has_session"] is True
|
||||
|
||||
|
||||
def test_import_from_browser_maps_locked_db(tmp_path: Path, monkeypatch):
|
||||
store = Store(tmp_path / "state.db")
|
||||
|
||||
def fake_extract(browser, profile=None, logger=None, **kw):
|
||||
raise RuntimeError("Could not copy Chrome cookie database. See https://github.com/yt-dlp/yt-dlp/issues/7271 for more info")
|
||||
|
||||
import yt_dlp.cookies as ydl_cookies
|
||||
monkeypatch.setattr(ydl_cookies, "extract_cookies_from_browser", fake_extract)
|
||||
|
||||
with pytest.raises(BrowserCookieLockedError) as exc:
|
||||
import_from_browser(store, browser="brave", cookie_dir=tmp_path / "cookies")
|
||||
assert "close brave" in str(exc.value).lower()
|
||||
|
||||
|
||||
def test_auto_import_and_prune_dead_cookies(tmp_path: Path):
|
||||
db_path = tmp_path / "state.db"
|
||||
store = Store(db_path)
|
||||
cookie_dir = tmp_path / "cookies"
|
||||
cookie_dir.mkdir()
|
||||
|
||||
# Place two cookie files
|
||||
f1 = cookie_dir / "c1.txt"
|
||||
f1.write_text(SAMPLE_NETSCAPE, encoding="utf-8")
|
||||
f2 = cookie_dir / "c2.txt"
|
||||
f2.write_text(SAMPLE_NETSCAPE, encoding="utf-8")
|
||||
|
||||
# auto_import imports both and activates the first
|
||||
n = auto_import_dir(store, dir_path=cookie_dir)
|
||||
assert n == 2
|
||||
assert len(store.list_cookies()) == 2
|
||||
active = store.get_active_cookie()
|
||||
assert active is not None
|
||||
assert active.filename in ("c1.txt", "c2.txt")
|
||||
|
||||
active_path = resolve_active_path(store, cookie_dir=cookie_dir)
|
||||
assert active_path is not None
|
||||
assert Path(active_path).exists()
|
||||
|
||||
# Now simulate user deleting the active cookie file from disk
|
||||
Path(active_path).unlink()
|
||||
|
||||
# resolve_active_path should fall back to the surviving cookie file
|
||||
fallback_path = resolve_active_path(store, cookie_dir=cookie_dir)
|
||||
assert fallback_path is not None
|
||||
assert Path(fallback_path).exists()
|
||||
|
||||
# auto_import_dir should clean up the deleted cookie row from DB
|
||||
auto_import_dir(store, dir_path=cookie_dir)
|
||||
assert len(store.list_cookies()) == 1
|
||||
@@ -0,0 +1,68 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from fastapi import FastAPI
|
||||
from fastapi.testclient import TestClient
|
||||
import pytest
|
||||
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
from yt_scraper.webapp.api import build_router
|
||||
from yt_scraper.webapp.jobs import JobManager
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def api_client(tmp_path: Path, monkeypatch):
|
||||
db_path = tmp_path / "state.db"
|
||||
md_root = tmp_path / "markdown"
|
||||
md_root.mkdir()
|
||||
|
||||
store = Store(db_path)
|
||||
cfg = Config(database_path=str(db_path), output_dir=str(md_root))
|
||||
jobs = JobManager(store, cfg)
|
||||
router = build_router(store, cfg, jobs)
|
||||
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
client = TestClient(app)
|
||||
return client, store, cfg, tmp_path
|
||||
|
||||
|
||||
def test_get_and_open_markdown(api_client, monkeypatch):
|
||||
client, store, cfg, tmp_path = api_client
|
||||
store.upsert_channel("UC1", "@test", "Test Channel", 1)
|
||||
store.upsert_videos([
|
||||
VideoRef("vid1", "UC1", "Test Video", "https://www.youtube.com/watch?v=vid1")
|
||||
])
|
||||
|
||||
# 1. Not generated yet -> 404
|
||||
r_get = client.get("/api/videos/vid1/markdown")
|
||||
assert r_get.status_code == 404
|
||||
|
||||
r_open = client.post("/api/videos/vid1/open-markdown")
|
||||
assert r_open.status_code == 404
|
||||
|
||||
# 2. Create markdown file and mark done
|
||||
md_dir = tmp_path / "markdown" / "Test Channel"
|
||||
md_dir.mkdir(parents=True)
|
||||
md_file = md_dir / "2026-08-23_test-video.md"
|
||||
md_content = "# Test Video\n\nContent here"
|
||||
md_file.write_text(md_content, encoding="utf-8", newline="\n")
|
||||
|
||||
rel_path = "markdown/Test Channel/2026-08-23_test-video.md"
|
||||
store.mark_done("vid1", rel_path, "es", "auto", False)
|
||||
|
||||
# 3. GET markdown returns text
|
||||
r_get = client.get("/api/videos/vid1/markdown")
|
||||
assert r_get.status_code == 200
|
||||
assert r_get.text.replace("\r\n", "\n") == md_content
|
||||
|
||||
# 4. POST open-markdown calls _open_in_os
|
||||
opened_paths = []
|
||||
from yt_scraper.webapp import api as api_mod
|
||||
monkeypatch.setattr(api_mod, "_open_in_os", lambda p: opened_paths.append(str(p)))
|
||||
|
||||
r_open = client.post("/api/videos/vid1/open-markdown")
|
||||
assert r_open.status_code == 200
|
||||
assert len(opened_paths) == 1
|
||||
assert str(md_file.resolve()) in [str(Path(p).resolve()) for p in opened_paths]
|
||||
@@ -136,14 +136,16 @@ def test_cookie_vault(store):
|
||||
sample = (
|
||||
"# Netscape HTTP Cookie File\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSID\tabc\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSAPISID\tdef\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tHSID\tdef\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSSID\tghi\n"
|
||||
".youtube.com\tTRUE\t/\tTRUE\t1800287621\tSAPISID\tjkl\n"
|
||||
".youtube.com\tTRUE\t/\tFALSE\t1800287621\tPREF\tf1=2\n"
|
||||
)
|
||||
cid = cookies.import_text(store, sample, label="test", cookie_dir=str(Path(store.db_path).parent / "ck"))
|
||||
vault = cookies.list_vault(store)
|
||||
assert len(vault) == 1
|
||||
assert vault[0].has_session is True
|
||||
assert vault[0].cookie_count == 3
|
||||
assert vault[0].cookie_count == 5
|
||||
# set active + resolve
|
||||
cookies.set_active(store, cid)
|
||||
path = cookies.resolve_active_path(store, cookie_dir=str(Path(store.db_path).parent / "ck"))
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from yt_scraper.extract import extract_via_watch_page
|
||||
|
||||
|
||||
WATCH_PAGE = (
|
||||
"var ytInitialPlayerResponse = "
|
||||
+ json.dumps({
|
||||
"playabilityStatus": {"status": "OK"},
|
||||
"videoDetails": {
|
||||
"title": "Members only test",
|
||||
"author": "Canal",
|
||||
"lengthSeconds": "42",
|
||||
"viewCount": "7",
|
||||
"shortDescription": "desc",
|
||||
"keywords": ["a", "b"],
|
||||
},
|
||||
"microformat": {"playerMicroformatRenderer": {"publishDate": "2026-09-01"}},
|
||||
"captions": {"playerCaptionsTracklistRenderer": {"captionTracks": [
|
||||
{"languageCode": "es", "kind": "asr", "baseUrl": "https://captions.test/es"},
|
||||
]}},
|
||||
})
|
||||
+ ";"
|
||||
)
|
||||
|
||||
JSON3 = json.dumps({
|
||||
"events": [
|
||||
{"tStartMs": 0, "dDurationMs": 1000, "segs": [{"utf8": "hola "}, {"utf8": "mundo"}]},
|
||||
{"tStartMs": 1000, "dDurationMs": 500, "segs": [{"utf8": "segundo"}]},
|
||||
]
|
||||
})
|
||||
|
||||
|
||||
class _FakeResponse(io.BytesIO):
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *a):
|
||||
return False
|
||||
|
||||
|
||||
def test_watch_page_fallback_builds_segments(tmp_path: Path, monkeypatch):
|
||||
from yt_scraper import extract as ex
|
||||
|
||||
def fake_urlopen(cookies_file, url, **kw):
|
||||
if url.startswith("https://www.youtube.com/"):
|
||||
return _FakeResponse(WATCH_PAGE.encode("utf-8"))
|
||||
assert "fmt=json3" in url
|
||||
return _FakeResponse(JSON3.encode("utf-8"))
|
||||
|
||||
monkeypatch.setattr(ex, "_session_urlopen", fake_urlopen)
|
||||
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
|
||||
|
||||
data = extract_via_watch_page(
|
||||
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
|
||||
)
|
||||
|
||||
assert data is not None
|
||||
assert data.subtitle is not None and data.subtitle.lang == "es"
|
||||
# parse_auto_dump merges caption events with no gap between them.
|
||||
assert len(data.segments) == 1
|
||||
assert data.segments[0].text == "hola mundo segundo"
|
||||
assert data.info["title"] == "Members only test"
|
||||
assert data.info["upload_date"] == "20260901"
|
||||
assert data.info["duration"] == 42
|
||||
assert data.skip_reason is None
|
||||
|
||||
|
||||
def test_watch_page_fallback_without_tracks_returns_none(tmp_path: Path, monkeypatch):
|
||||
from yt_scraper import extract as ex
|
||||
|
||||
page = WATCH_PAGE.replace('"captionTracks": [{', '"captionTracks": [{')
|
||||
page = "var ytInitialPlayerResponse = " + json.dumps({
|
||||
"playabilityStatus": {"status": "OK"},
|
||||
"videoDetails": {"title": "x"},
|
||||
}) + ";"
|
||||
|
||||
monkeypatch.setattr(ex, "_session_urlopen",
|
||||
lambda *a, **kw: _FakeResponse(page.encode("utf-8")))
|
||||
monkeypatch.setattr(ex.GLOBAL_PACER, "wait", lambda **kw: None)
|
||||
|
||||
assert extract_via_watch_page(
|
||||
"https://www.youtube.com/watch?v=xyz", "cookies/fake.txt", {"es": "any"}
|
||||
) is None
|
||||
Reference in New Issue
Block a user