Files
yt-channel-scraper/tests/test_subtitle_language.py
T

220 lines
8.4 KiB
Python

"""The transcript must be the language actually spoken, not a machine translation.
Found in production on the Alex Hormozi channel (English). Under
`languages: {es: any, es-419: any, en: any}` the picker walked the config in
order, matched `es` first, and stored a Spanish auto-translation of English
speech — "Soy Nim Jenkinson y enseño a los aficionados a las manualidades" for
a video whose speaker says it in English.
Two things conspired:
1. `_normalize_lang("es-orig") == "es"`, so the `-orig` suffix — the one piece
of evidence distinguishing YouTube's real ASR track from a translation *into
the same language* — was thrown away before comparison.
2. Config order was treated as absolute preference, so a translation into a
preferred language beat the original.
Spanish channels were unaffected only by luck: yt-dlp happens to list `es-orig`
before `es`, so dict order gave the right answer. Every `done` row on the four
Spanish channels recorded `transcript_lang=es-orig`; the three Hormozi rows
recorded `es`. That asymmetry is what these tests pin down.
"""
from __future__ import annotations
from yt_scraper.extract import original_language, pick_subtitle
CFG = {"es": "any", "es-419": "any", "en": "any"}
def _track(url: str) -> list[dict]:
return [{"ext": "json3", "url": url}]
def _english_video() -> dict:
"""An English video as YouTube actually presents it: the original ASR under
`en-orig`, plus a long tail of translations keyed by bare language code."""
return {
"id": "aRVv5NLVRwE",
"title": "My honest advice to someone who wants to get rich.",
"subtitles": {},
"automatic_captions": {
"en-orig": _track("https://timedtext/en-orig"),
"en": _track("https://timedtext/en-translated"),
"es": _track("https://timedtext/es"),
"es-419": _track("https://timedtext/es-419"),
"fr": _track("https://timedtext/fr"),
},
}
def _spanish_video() -> dict:
return {
"id": "uZH3FKH_yNw",
"title": "Escribir codigo a mano sera irresponsable",
"subtitles": {},
"automatic_captions": {
"es-orig": _track("https://timedtext/es-orig"),
"es": _track("https://timedtext/es"),
"en": _track("https://timedtext/en"),
},
}
def test_english_video_yields_english_not_a_spanish_translation():
"""The production bug, verbatim."""
pick = pick_subtitle(_english_video(), CFG, prefer_manual=True)
assert pick is not None
assert pick.lang == "en-orig", f"picked {pick.lang}: a translation, not the spoken language"
def test_spanish_video_still_yields_the_spanish_original():
"""The fix must not regress the four Spanish channels already in the DB."""
pick = pick_subtitle(_spanish_video(), CFG, prefer_manual=True)
assert pick is not None
assert pick.lang == "es-orig"
def test_original_beats_a_same_language_translation():
"""YouTube publishes a translation *into the video's own language* too.
`en-orig` and `en` both normalise to "en"; only the suffix says which one is
the real transcript, and dict order must not decide it.
"""
info = {
"subtitles": {},
# Deliberately listed translation-first to defeat insertion order.
"automatic_captions": {
"en": _track("https://timedtext/en-translated"),
"en-orig": _track("https://timedtext/en-orig"),
},
}
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
assert pick.lang == "en-orig"
assert pick.url.endswith("en-orig")
def test_translation_is_a_documented_last_resort_not_a_silent_default():
"""A German channel under a Spanish/English policy.
The spoken language is not one the operator asked for, so there is no
original to give them and a translation is the only thing on offer. We do
take it — but only on the second pass, after every untranslated option has
been rejected, and `transcript_lang` records the bare code so the row is
distinguishable from an `-orig` one afterwards.
This case is a deliberate fallback. It is NOT the behaviour that caused the
Hormozi bug: there, `en` *was* configured and was being skipped.
"""
info = {
"subtitles": {},
"automatic_captions": {
"de-orig": _track("https://timedtext/de-orig"),
"es": _track("https://timedtext/de?tlang=es"),
"en": _track("https://timedtext/de?tlang=en"),
},
}
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "es"
assert "tlang=" in pick.url, "the fallback really is a translation; nothing else was available"
# ------------------------------------------------- the tlang= guard
#
# A translated caption URL is the base track's URL with `tlang=` appended, and
# yt-dlp omits it when target == source. That is direct evidence, unlike the
# `-orig` naming convention, so it catches videos whose spoken language cannot
# be determined any other way.
def test_untranslated_track_wins_even_with_no_orig_key_and_no_language_field():
"""The gap the first version of this fix left open.
Without an `-orig` key and without `info["language"]`, the spoken language
is unknown, the reorder cannot fire, and config order used to hand back the
Spanish translation. Rejecting `tlang=` needs no such knowledge.
"""
info = {
"subtitles": {},
"automatic_captions": {
"es": _track("https://timedtext/base?lang=en&kind=asr&tlang=es"),
"en": _track("https://timedtext/base?lang=en&kind=asr"),
},
}
assert original_language(info) is None, "precondition: spoken language is undeterminable"
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "en"
assert "tlang=" not in pick.url
def test_the_real_hormozi_url_shape_is_recognised_as_a_translation():
"""Verbatim parameters from the production timedtext URLs in .run/server.log."""
info = {
"subtitles": {},
"automatic_captions": {
"es": _track(
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
"&lang=en&kind=asr&variant=gemini&fmt=json3&tlang=es"
),
"en-orig": _track(
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
"&lang=en&kind=asr&variant=gemini&fmt=json3"
),
},
}
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "en-orig"
assert "tlang=" not in pick.url
def test_manual_captions_still_outrank_auto_for_the_same_language():
"""Preferring the original must not override the manual/auto policy."""
info = {
"subtitles": {"en": _track("https://timedtext/en-manual")},
"automatic_captions": {"en-orig": _track("https://timedtext/en-orig")},
}
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
assert pick.source == "manual"
assert pick.url.endswith("en-manual")
def test_a_manual_translation_does_not_beat_the_spoken_language():
"""Config lists es first, but the video is English with English manual subs."""
info = {
"subtitles": {"en": _track("https://timedtext/en-manual")},
"automatic_captions": {
"en-orig": _track("https://timedtext/en-orig"),
"es": _track("https://timedtext/es"),
},
}
pick = pick_subtitle(info, CFG, prefer_manual=True)
assert pick.lang == "en"
assert pick.source == "manual"
# ------------------------------------------------------------- detection
def test_original_language_read_from_the_orig_suffix():
assert original_language(_english_video()) == "en"
assert original_language(_spanish_video()) == "es"
def test_original_language_falls_back_to_the_info_key():
assert original_language({"language": "pt-BR", "automatic_captions": {}}) == "pt"
def test_orig_suffix_wins_over_the_info_key():
"""`language` is metadata YouTube localises; the caption list is evidence.
Production showed YouTube returning English titles for Spanish videos, so
localised metadata is not trustworthy for this decision.
"""
info = {"language": "es", "automatic_captions": {"en-orig": _track("u")}}
assert original_language(info) == "en"
def test_original_language_is_none_when_unknowable():
assert original_language({"automatic_captions": {"es": _track("u")}}) is None
assert original_language({}) is None