Files
yt-channel-scraper/tests/test_markdown_identity.py

119 lines
4.7 KiB
Python

"""One video must own exactly one .md, whatever happens to its title.
The filename is derived from the title, and titles are not stable: YouTube
serves them localised, so the same video came back as "La controversia de
Claude Fable 5" on one pass and "The Claude Fable controversy 5" on the next.
Creators also simply rename videos.
`re_render_videos` already deleted the superseded file; `process_video` did not,
so a re-scrape after a title change left the old file orphaned on disk. The DB
repointed, the stale file stayed, and every later scan had to wade through it —
the same shape as the incident that left 94 files for 61 rows.
The identity that matters is the video id, which never changes. These tests pin
that: the row's `markdown_path` is authoritative, and anything it used to point
at gets cleaned up.
"""
from __future__ import annotations
from pathlib import Path
import pytest
from yt_scraper.config import Config
from yt_scraper.extract import SubtitlePick, VideoData
from yt_scraper.parse import Segment
from yt_scraper.pipeline import process_video
from yt_scraper.render import build_filename_stem
from yt_scraper.store import Store, VideoRef
@pytest.fixture
def env(tmp_path):
cfg = Config(
database_path=str(tmp_path / "state.db"),
output_dir=str(tmp_path / "markdown"),
template_path="templates/video.md.j2",
)
store = Store(cfg.database_path_resolved)
store.upsert_channel("UC1", "@alpha", "Alpha", 1)
store.upsert_videos([VideoRef("vid123", "UC1", "t", "https://y/watch?v=vid123", "20260101", 60)])
return cfg, store, tmp_path
def _data(title: str) -> VideoData:
return VideoData(
info={"id": "vid123", "title": title, "upload_date": "20260101", "channel": "Alpha"},
segments=[Segment(start=0.0, end=2.0, text="hello")],
subtitle=SubtitlePick(url="u", ext="json3", lang="en-orig", source="auto"),
has_chapters=False,
)
def _md_files(root: Path) -> list[str]:
return sorted(p.name for p in root.rglob("*.md"))
def test_a_retitled_video_does_not_leave_a_second_file(env, monkeypatch):
"""The production case: the same video, title localised differently."""
cfg, store, tmp_path = env
md_root = Path(cfg.output_dir_resolved)
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
lambda *a, **k: _data("La controversia de Claude Fable 5"))
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
first = _md_files(md_root)
assert len(first) == 1
monkeypatch.setattr("yt_scraper.pipeline.extract_video",
lambda *a, **k: _data("The Claude Fable controversy 5"))
assert process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u") == "done"
after = _md_files(md_root)
assert len(after) == 1, f"one video, {len(after)} files on disk: {after}"
# And the DB points at the one that exists.
row = store.get_video("vid123")
assert (Path(cfg.output_dir_resolved).parent / row.markdown_path).exists()
assert Path(row.markdown_path).name == after[0]
def test_rescraping_an_unchanged_video_is_idempotent(env, monkeypatch):
cfg, store, tmp_path = env
md_root = Path(cfg.output_dir_resolved)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("Same Title"))
for _ in range(3):
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
assert len(_md_files(md_root)) == 1
def test_changing_the_filename_template_relocates_rather_than_duplicates(env, monkeypatch):
"""Opting into ids in the filename must not strand the old files."""
cfg, store, tmp_path = env
md_root = Path(cfg.output_dir_resolved)
monkeypatch.setattr("yt_scraper.pipeline.extract_video", lambda *a, **k: _data("A Title"))
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
cfg.filename_template = "{upload_date}_{slug}_{video_id}"
process_video(store.get_video("vid123"), cfg, store, "Alpha", "UC1", "u")
files = _md_files(md_root)
assert len(files) == 1, f"template change duplicated the file: {files}"
assert "vid123" in files[0]
# ------------------------------------------------------------ template vars
def test_video_id_is_available_to_the_filename_template():
"""Lets an operator make the file self-identifying without the DB."""
stem = build_filename_stem(
"20260101", "Some Title", template="{upload_date}_{slug}_{video_id}", video_id="abc123XYZ_-"
)
assert stem == "20260101_some-title_abc123XYZ_-"
def test_default_template_is_unchanged():
"""Existing libraries keep their filenames; adding the variable is opt-in."""
assert build_filename_stem("20260101", "Some Title", video_id="abc123") == "20260101_some-title"