wip: estado de trabajo pendiente antes de la vista grid (suite 230 verde)
This commit is contained in:
@@ -0,0 +1,118 @@
|
||||
"""One video must own exactly one .md, whatever happens to its title.
|
||||
|
||||
The filename is derived from the title, and titles are not stable: YouTube
|
||||
serves them localised, so the same video came back as "La controversia de
|
||||
Claude Fable 5" on one pass and "The Claude Fable controversy 5" on the next.
|
||||
Creators also simply rename videos.
|
||||
|
||||
`re_render_videos` already deleted the superseded file; `process_video` did not,
|
||||
so a re-scrape after a title change left the old file orphaned on disk. The DB
|
||||
repointed, the stale file stayed, and every later scan had to wade through it —
|
||||
the same shape as the incident that left 94 files for 61 rows.
|
||||
|
||||
The identity that matters is the video id, which never changes. These tests pin
|
||||
that: the row's `markdown_path` is authoritative, and anything it used to point
|
||||
at gets cleaned up.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from yt_scraper.config import Config
|
||||
from yt_scraper.extract import SubtitlePick, VideoData
|
||||
from yt_scraper.parse import Segment
|
||||
from yt_scraper.pipeline import process_video
|
||||
from yt_scraper.render import build_filename_stem
|
||||
from yt_scraper.store import Store, VideoRef
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def env(tmp_path):
|
||||
cfg = Config(
|
||||
database_path=str(tmp_path / "state.db"),
|
||||
output_dir=str(tmp_path / "markdown"),
|
||||
template_path="templates/video.md.j2",
|
||||
)
|
||||
store = Store(cfg.database_path_resolved)
|
||||
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
|
||||
store.upsert_videos([VideoRef("vid123", "UC1", "t", "https://y/watch?v=vid123", "20260101", 60)])
|
||||
return cfg, store, tmp_path
|
||||
|
||||
|
||||
def _data(title: str) -> VideoData:
|
||||
return VideoData(
|
||||
info={"id": "vid123", "title": title, "upload_date": "20260101", "channel": "Alpha"},
|
||||
segments=[Segment(start=0.0, end=2.0, text="hello")],
|
||||
subtitle=SubtitlePick(url="u", ext="json3", lang="en-orig", source="auto"),
|
||||
has_chapters=False,
|
||||
)
|
||||
|
||||
|
||||
def _md_files(root: Path) -> list[str]:
|
||||
return sorted(p.name for p in root.rglob("*.md"))
|
||||
|
||||
|
||||
def test_a_retitled_video_does_not_leave_a_second_file(env, monkeypatch):
|
||||
"""The production case: the same video, title localised differently."""
|
||||
cfg, store, tmp_path = env
|
||||
md_root = Path(cfg.output_dir_resolved)
|
||||
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
|
||||
lambda *a, **k: _data("La controversia de Claude Fable 5"))
|
||||
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
|
||||
first = _md_files(md_root)
|
||||
assert len(first) == 1
|
||||
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
|
||||
lambda *a, **k: _data("The Claude Fable controversy 5"))
|
||||
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
|
||||
|
||||
after = _md_files(md_root)
|
||||
assert len(after) == 1, f"one video, {len(after)} files on disk: {after}"
|
||||
# And the DB points at the one that exists.
|
||||
row = store.get_video("vid123")
|
||||
assert (Path(cfg.output_dir_resolved).parent / row.markdown_path).exists()
|
||||
assert Path(row.markdown_path).name == after[0]
|
||||
|
||||
|
||||
def test_rescraping_an_unchanged_video_is_idempotent(env, monkeypatch):
|
||||
cfg, store, tmp_path = env
|
||||
md_root = Path(cfg.output_dir_resolved)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("Same Title"))
|
||||
for _ in range(3):
|
||||
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
|
||||
assert len(_md_files(md_root)) == 1
|
||||
|
||||
|
||||
def test_changing_the_filename_template_relocates_rather_than_duplicates(env, monkeypatch):
|
||||
"""Opting into ids in the filename must not strand the old files."""
|
||||
cfg, store, tmp_path = env
|
||||
md_root = Path(cfg.output_dir_resolved)
|
||||
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("A Title"))
|
||||
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
|
||||
|
||||
cfg.filename_template = "{upload_date}_{slug}_{video_id}"
|
||||
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
|
||||
|
||||
files = _md_files(md_root)
|
||||
assert len(files) == 1, f"template change duplicated the file: {files}"
|
||||
assert "vid123" in files[0]
|
||||
|
||||
|
||||
# ------------------------------------------------------------ template vars
|
||||
|
||||
|
||||
def test_video_id_is_available_to_the_filename_template():
|
||||
"""Lets an operator make the file self-identifying without the DB."""
|
||||
stem = build_filename_stem(
|
||||
"20260101", "Some Title", template="{upload_date}_{slug}_{video_id}", video_id="abc123XYZ_-"
|
||||
)
|
||||
assert stem == "20260101_some-title_abc123XYZ_-"
|
||||
|
||||
|
||||
def test_default_template_is_unchanged():
|
||||
"""Existing libraries keep their filenames; adding the variable is opt-in."""
|
||||
assert build_filename_stem("20260101", "Some Title", video_id="abc123") == "20260101_some-title"
|
||||
Reference in New Issue
Block a user