channel_url: "https://www.youtube.com/@Nostal-Vlad/videos" # Per-language subtitle preference. Either a list (legacy) or a dict. # Dict form (preferred): per-lang mode. Modes are # "manual" - ONLY human-uploaded captions; a video with just auto captions is # skipped and stored as `no_subtitles` # "auto" - ONLY YouTube-auto-generated captions # "any" - try manual first (per `prefer_manual`), fall back to auto # List form (legacy): `languages: ["es", "en"]` maps EVERY entry to the hard # "manual"/"auto" mode picked by `prefer_manual` - note this means manual-ONLY # when prefer_manual is true, which yields nothing on channels that publish # only auto-generated captions. Prefer the dict form below. languages: es: any es-419: any en: any prefer_manual: true # with mode "any": try manual first, then auto include_shorts: false include_live: true min_duration_sec: 30 # Pacing and what to do when YouTube starts refusing. # # YouTube publishes no rate limits, so none of these numbers are official. The # backoff shape is: yt-dlp's own docs put a guest session at roughly 300 videos # an hour (~1000 requests) before the bot wall, and Google documents truncated # exponential backoff with jitter for its own APIs, which is what backoff_base # and backoff_cap feed. delay: min_seconds: 1.5 # randomised gap between videos / channels max_seconds: 3.5 backoff_base: 2.0 # after a rate-limit: min(base * 2**n + jitter, cap) backoff_cap: 60.0 throttle_threshold: 3 # consecutive rate-limit responses that abort the run # Floor on the gap between ANY two requests to YouTube from this process, # including the /api/tools/* endpoints that run outside the job runner. # 0 = off. Raise it if you still hit the bot wall; it is the one setting that # applies everywhere at once. min_request_interval: 0.0 audio_rate_limit: 0 # bytes/sec ceiling for audio downloads, 0 = unlimited yt_dlp: retries: 10 # download retries (only bites on the audio path) # Seconds between the individual HTTP calls inside one extraction — the watch # page, the InnerTube player call, each continuation page of a channel tab. # Forwarded to yt-dlp as `sleep_interval_requests`. sleep_subrequests: 2 extractor_retries: 3 # retries during extraction (5xx and network only: # yt-dlp's YouTube extractor never retries 403/429) socket_timeout: 30.0 # without this a hung connection blocks the worker # Incremental channel sync. Re-scanning a tracked channel reads the /videos tab # newest-first and stops once it sees `overlap` videos already in the DB, so a # routine sync costs one page instead of paginating the whole channel. If every # video in the window turns out to be new, the window doubles (up to # max_window) rather than silently missing uploads. # Set incremental: false — or pass --full / tick "Full channel rescan" — to walk # every page, which is only needed when the local catalog is incomplete. sync: incremental: true window: 30 # entries read on the first pass max_window: 300 # ceiling before reporting a truncated scan overlap: 3 # consecutive known videos that prove we caught up database_path: "data/state.db" output_dir: "data/markdown" template_path: "templates/video.md.j2" filename_template: "{upload_date}_{slug}"