220 lines
8.4 KiB
Python
220 lines
8.4 KiB
Python
"""The transcript must be the language actually spoken, not a machine translation.
|
|
|
|
Found in production on the Alex Hormozi channel (English). Under
|
|
`languages: {es: any, es-419: any, en: any}` the picker walked the config in
|
|
order, matched `es` first, and stored a Spanish auto-translation of English
|
|
speech — "Soy Nim Jenkinson y enseño a los aficionados a las manualidades" for
|
|
a video whose speaker says it in English.
|
|
|
|
Two things conspired:
|
|
|
|
1. `_normalize_lang("es-orig") == "es"`, so the `-orig` suffix — the one piece
|
|
of evidence distinguishing YouTube's real ASR track from a translation *into
|
|
the same language* — was thrown away before comparison.
|
|
2. Config order was treated as absolute preference, so a translation into a
|
|
preferred language beat the original.
|
|
|
|
Spanish channels were unaffected only by luck: yt-dlp happens to list `es-orig`
|
|
before `es`, so dict order gave the right answer. Every `done` row on the four
|
|
Spanish channels recorded `transcript_lang=es-orig`; the three Hormozi rows
|
|
recorded `es`. That asymmetry is what these tests pin down.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from yt_scraper.extract import original_language, pick_subtitle
|
|
|
|
CFG = {"es": "any", "es-419": "any", "en": "any"}
|
|
|
|
|
|
def _track(url: str) -> list[dict]:
|
|
return [{"ext": "json3", "url": url}]
|
|
|
|
|
|
def _english_video() -> dict:
|
|
"""An English video as YouTube actually presents it: the original ASR under
|
|
`en-orig`, plus a long tail of translations keyed by bare language code."""
|
|
return {
|
|
"id": "aRVv5NLVRwE",
|
|
"title": "My honest advice to someone who wants to get rich.",
|
|
"subtitles": {},
|
|
"automatic_captions": {
|
|
"en-orig": _track("https://timedtext/en-orig"),
|
|
"en": _track("https://timedtext/en-translated"),
|
|
"es": _track("https://timedtext/es"),
|
|
"es-419": _track("https://timedtext/es-419"),
|
|
"fr": _track("https://timedtext/fr"),
|
|
},
|
|
}
|
|
|
|
|
|
def _spanish_video() -> dict:
|
|
return {
|
|
"id": "uZH3FKH_yNw",
|
|
"title": "Escribir codigo a mano sera irresponsable",
|
|
"subtitles": {},
|
|
"automatic_captions": {
|
|
"es-orig": _track("https://timedtext/es-orig"),
|
|
"es": _track("https://timedtext/es"),
|
|
"en": _track("https://timedtext/en"),
|
|
},
|
|
}
|
|
|
|
|
|
def test_english_video_yields_english_not_a_spanish_translation():
|
|
"""The production bug, verbatim."""
|
|
pick = pick_subtitle(_english_video(), CFG, prefer_manual=True)
|
|
assert pick is not None
|
|
assert pick.lang == "en-orig", f"picked {pick.lang}: a translation, not the spoken language"
|
|
|
|
|
|
def test_spanish_video_still_yields_the_spanish_original():
|
|
"""The fix must not regress the four Spanish channels already in the DB."""
|
|
pick = pick_subtitle(_spanish_video(), CFG, prefer_manual=True)
|
|
assert pick is not None
|
|
assert pick.lang == "es-orig"
|
|
|
|
|
|
def test_original_beats_a_same_language_translation():
|
|
"""YouTube publishes a translation *into the video's own language* too.
|
|
|
|
`en-orig` and `en` both normalise to "en"; only the suffix says which one is
|
|
the real transcript, and dict order must not decide it.
|
|
"""
|
|
info = {
|
|
"subtitles": {},
|
|
# Deliberately listed translation-first to defeat insertion order.
|
|
"automatic_captions": {
|
|
"en": _track("https://timedtext/en-translated"),
|
|
"en-orig": _track("https://timedtext/en-orig"),
|
|
},
|
|
}
|
|
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
|
|
assert pick.lang == "en-orig"
|
|
assert pick.url.endswith("en-orig")
|
|
|
|
|
|
def test_translation_is_a_documented_last_resort_not_a_silent_default():
|
|
"""A German channel under a Spanish/English policy.
|
|
|
|
The spoken language is not one the operator asked for, so there is no
|
|
original to give them and a translation is the only thing on offer. We do
|
|
take it — but only on the second pass, after every untranslated option has
|
|
been rejected, and `transcript_lang` records the bare code so the row is
|
|
distinguishable from an `-orig` one afterwards.
|
|
|
|
This case is a deliberate fallback. It is NOT the behaviour that caused the
|
|
Hormozi bug: there, `en` *was* configured and was being skipped.
|
|
"""
|
|
info = {
|
|
"subtitles": {},
|
|
"automatic_captions": {
|
|
"de-orig": _track("https://timedtext/de-orig"),
|
|
"es": _track("https://timedtext/de?tlang=es"),
|
|
"en": _track("https://timedtext/de?tlang=en"),
|
|
},
|
|
}
|
|
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
|
assert pick.lang == "es"
|
|
assert "tlang=" in pick.url, "the fallback really is a translation; nothing else was available"
|
|
|
|
|
|
# ------------------------------------------------- the tlang= guard
|
|
#
|
|
# A translated caption URL is the base track's URL with `tlang=` appended, and
|
|
# yt-dlp omits it when target == source. That is direct evidence, unlike the
|
|
# `-orig` naming convention, so it catches videos whose spoken language cannot
|
|
# be determined any other way.
|
|
|
|
|
|
def test_untranslated_track_wins_even_with_no_orig_key_and_no_language_field():
|
|
"""The gap the first version of this fix left open.
|
|
|
|
Without an `-orig` key and without `info["language"]`, the spoken language
|
|
is unknown, the reorder cannot fire, and config order used to hand back the
|
|
Spanish translation. Rejecting `tlang=` needs no such knowledge.
|
|
"""
|
|
info = {
|
|
"subtitles": {},
|
|
"automatic_captions": {
|
|
"es": _track("https://timedtext/base?lang=en&kind=asr&tlang=es"),
|
|
"en": _track("https://timedtext/base?lang=en&kind=asr"),
|
|
},
|
|
}
|
|
assert original_language(info) is None, "precondition: spoken language is undeterminable"
|
|
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
|
assert pick.lang == "en"
|
|
assert "tlang=" not in pick.url
|
|
|
|
|
|
def test_the_real_hormozi_url_shape_is_recognised_as_a_translation():
|
|
"""Verbatim parameters from the production timedtext URLs in .run/server.log."""
|
|
info = {
|
|
"subtitles": {},
|
|
"automatic_captions": {
|
|
"es": _track(
|
|
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
|
|
"&lang=en&kind=asr&variant=gemini&fmt=json3&tlang=es"
|
|
),
|
|
"en-orig": _track(
|
|
"https://www.youtube.com/api/timedtext?v=aRVv5NLVRwE&caps=asr&opi=112496729"
|
|
"&lang=en&kind=asr&variant=gemini&fmt=json3"
|
|
),
|
|
},
|
|
}
|
|
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
|
assert pick.lang == "en-orig"
|
|
assert "tlang=" not in pick.url
|
|
|
|
|
|
def test_manual_captions_still_outrank_auto_for_the_same_language():
|
|
"""Preferring the original must not override the manual/auto policy."""
|
|
info = {
|
|
"subtitles": {"en": _track("https://timedtext/en-manual")},
|
|
"automatic_captions": {"en-orig": _track("https://timedtext/en-orig")},
|
|
}
|
|
pick = pick_subtitle(info, {"en": "any"}, prefer_manual=True)
|
|
assert pick.source == "manual"
|
|
assert pick.url.endswith("en-manual")
|
|
|
|
|
|
def test_a_manual_translation_does_not_beat_the_spoken_language():
|
|
"""Config lists es first, but the video is English with English manual subs."""
|
|
info = {
|
|
"subtitles": {"en": _track("https://timedtext/en-manual")},
|
|
"automatic_captions": {
|
|
"en-orig": _track("https://timedtext/en-orig"),
|
|
"es": _track("https://timedtext/es"),
|
|
},
|
|
}
|
|
pick = pick_subtitle(info, CFG, prefer_manual=True)
|
|
assert pick.lang == "en"
|
|
assert pick.source == "manual"
|
|
|
|
|
|
# ------------------------------------------------------------- detection
|
|
|
|
|
|
def test_original_language_read_from_the_orig_suffix():
|
|
assert original_language(_english_video()) == "en"
|
|
assert original_language(_spanish_video()) == "es"
|
|
|
|
|
|
def test_original_language_falls_back_to_the_info_key():
|
|
assert original_language({"language": "pt-BR", "automatic_captions": {}}) == "pt"
|
|
|
|
|
|
def test_orig_suffix_wins_over_the_info_key():
|
|
"""`language` is metadata YouTube localises; the caption list is evidence.
|
|
|
|
Production showed YouTube returning English titles for Spanish videos, so
|
|
localised metadata is not trustworthy for this decision.
|
|
"""
|
|
info = {"language": "es", "automatic_captions": {"en-orig": _track("u")}}
|
|
assert original_language(info) == "en"
|
|
|
|
|
|
def test_original_language_is_none_when_unknowable():
|
|
assert original_language({"automatic_captions": {"es": _track("u")}}) is None
|
|
assert original_language({}) is None
|