"""The transcript must be the language actually spoken, not a machine translation. Found in production on the Alex Hormozi channel (English). Under `languages: {es: any, es-419: any, en: any}` the picker walked the config in order, matched `es` first, and stored a Spanish auto-translation of English speech — "Soy Nim Jenkinson y enseño a los aficionados a las manualidades" for a video whose speaker says it in English. Two things conspired: 1. `_normalize_lang("es-orig") == "es"`, so the `-orig` suffix — the one piece of evidence distinguishing YouTube's real ASR track from a translation *into the same language* — was thrown away before comparison. 2. Config order was treated as absolute preference, so a translation into a preferred language beat the original. Spanish channels were unaffected only by luck: yt-dlp happens to list `es-orig` before `es`, so dict order gave the right answer. Every `done` row on the four Spanish channels recorded `transcript_lang=es-orig`; the three Hormozi rows recorded `es`. That asymmetry is what these tests pin down. """ from __future__ import annotations from yt_scraper.extract import original_language, pick_subtitle CFG = {"es": "any", "es-419": "any", "en": "any"} def _track(url: str) -> list[dict]: return [{"ext": "json3", "url": url}] def _english_video() -> dict: """An English video as YouTube actually presents it: the original ASR under `en-orig`, plus a long tail of translations keyed by bare language code.""" return { "id": "aRVv5NLVRwE", "title": "My honest advice to someone who wants to get rich.", "subtitles": {}, "automatic_captions": { "en-orig": _track("https://timedtext/en-orig"), "en": _track("https://timedtext/en-translated"), "es": _track("https://timedtext/es"), "es-419": _track("https://timedtext/es-419"), "fr": _track("https://timedtext/fr"), }, } def _spanish_video() -> dict: return { "id": "uZH3FKH_yNw", "title": "Escribir codigo a mano sera irresponsable", "subtitles": {}, "automatic_captions": { "es-orig": _track("https://timedtext/es-orig"), "es": _track("https://timedtext/es"), "en": _track("https://timedtext/en"), }, } def test_english_video_yields_english_not_a_spanish_translation(): """The production bug, verbatim.""" pick = pick_subtitle(_english_video(), CFG, prefer_manual=True) assert pick is not None assert pick.lang == "en-orig", f"picked {pick.lang}: a translation, not the spoken language" def test_spanish_video_still_yields_the_spanish_original(): """The fix must not regress the four Spanish channels already in the DB.""" pick = pick_subtitle(_spanish_video(), CFG, prefer_manual=True) assert pick is not None assert pick.lang == "es-orig" def test_original_beats_a_same_language_translation(): """YouTube publishes a translation *into the video's own language* too. `en-orig` and `en` both normalise to "en"; only the suffix says which one is the real transcript, and dict order must not decide it. """ info = { "subtitles": {}, # Deliberately listed translation-first to defeat insertion order. "automatic_captions": { "en": _track("https://timedtext/en-translated"), "en-orig": _track("https://timedtext/en-orig"), }, } pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True) assert pick.lang == "en-orig" assert pick.url.endswith("en-orig") def test_translation_is_a_documented_last_resort_not_a_silent_default(): """A German channel under a Spanish/English policy. The spoken language is not one the operator asked for, so there is no original to give them and a translation is the only thing on offer. We do take it — but only on the second pass, after every untranslated option has been rejected, and `transcript_lang` records the bare code so the row is distinguishable from an `-orig` one afterwards. This case is a deliberate fallback. It is NOT the behaviour that caused the Hormozi bug: there, `en` *was* configured and was being skipped. """ info = { "subtitles": {}, "automatic_captions": { "de-orig": _track("https://timedtext/de-orig"), "es": _track("https://timedtext/de?tlang=es"), "en": _track("https://timedtext/de?tlang=en"), }, } pick = pick_subtitle(info, CFG, prefer_manual=True) assert pick.lang == "es" assert "tlang=" in pick.url, "the fallback really is a translation; nothing else was available" # ------------------------------------------------- the tlang= guard # # A translated caption URL is the base track's URL with `tlang=` appended, and # yt-dlp omits it when target == source. That is direct evidence, unlike the # `-orig` naming convention, so it catches videos whose spoken language cannot # be determined any other way. def test_untranslated_track_wins_even_with_no_orig_key_and_no_language_field(): """The gap the first version of this fix left open. Without an `-orig` key and without `info["language"]`, the spoken language is unknown, the reorder cannot fire, and config order used to hand back the Spanish translation. Rejecting `tlang=` needs no such knowledge. """ info = { "subtitles": {}, "automatic_captions": { "es": _track("https://timedtext/base?lang=en&kind=asr&tlang=es"), "en": _track("https://timedtext/base?lang=en&kind=asr"), }, } assert original_language(info) is None, "precondition: spoken language is undeterminable" pick = pick_subtitle(info, CFG, prefer_manual=True) assert pick.lang == "en" assert "tlang=" not in pick.url def test_the_real_hormozi_url_shape_is_recognised_as_a_translation(): """Verbatim parameters from the production timedtext URLs in .run/server.log.""" info = { "subtitles": {}, "automatic_captions": { "es": _track( "https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729" "&lang=en&kind=asr&variant=gemini&fmt=json3&tlang=es" ), "en-orig": _track( "https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729" "&lang=en&kind=asr&variant=gemini&fmt=json3" ), }, } pick = pick_subtitle(info, CFG, prefer_manual=True) assert pick.lang == "en-orig" assert "tlang=" not in pick.url def test_manual_captions_still_outrank_auto_for_the_same_language(): """Preferring the original must not override the manual/auto policy.""" info = { "subtitles": {"en": _track("https://timedtext/en-manual")}, "automatic_captions": {"en-orig": _track("https://timedtext/en-orig")}, } pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True) assert pick.source == "manual" assert pick.url.endswith("en-manual") def test_a_manual_translation_does_not_beat_the_spoken_language(): """Config lists es first, but the video is English with English manual subs.""" info = { "subtitles": {"en": _track("https://timedtext/en-manual")}, "automatic_captions": { "en-orig": _track("https://timedtext/en-orig"), "es": _track("https://timedtext/es"), }, } pick = pick_subtitle(info, CFG, prefer_manual=True) assert pick.lang == "en" assert pick.source == "manual" # ------------------------------------------------------------- detection def test_original_language_read_from_the_orig_suffix(): assert original_language(_english_video()) == "en" assert original_language(_spanish_video()) == "es" def test_original_language_falls_back_to_the_info_key(): assert original_language({"language": "pt-BR", "automatic_captions": {}}) == "pt" def test_orig_suffix_wins_over_the_info_key(): """`language` is metadata YouTube localises; the caption list is evidence. Production showed YouTube returning English titles for Spanish videos, so localised metadata is not trustworthy for this decision. """ info = {"language": "es", "automatic_captions": {"en-orig": _track("u")}} assert original_language(info) == "en" def test_original_language_is_none_when_unknowable(): assert original_language({"automatic_captions": {"es": _track("u")}}) is None assert original_language({}) is None