Files
yt-channel-scraper/config.example.yaml
T

70 lines
3.3 KiB
YAML

channel_url: "https://www.youtube.com/@Nostal-Vlad/videos"
# Per-language subtitle preference. Either a list (legacy) or a dict.
# Dict form (preferred): per-lang mode. Modes are
# "manual" - ONLY human-uploaded captions; a video with just auto captions is
# skipped and stored as `no_subtitles`
# "auto" - ONLY YouTube-auto-generated captions
# "any" - try manual first (per `prefer_manual`), fall back to auto
# List form (legacy): `languages: ["es", "en"]` maps EVERY entry to the hard
# "manual"/"auto" mode picked by `prefer_manual` - note this means manual-ONLY
# when prefer_manual is true, which yields nothing on channels that publish
# only auto-generated captions. Prefer the dict form below.
languages:
es: any
es-419: any
en: any
prefer_manual: true # with mode "any": try manual first, then auto
include_shorts: false
include_live: true
min_duration_sec: 30
# Pacing and what to do when YouTube starts refusing.
#
# YouTube publishes no rate limits, so none of these numbers are official. The
# backoff shape is: yt-dlp's own docs put a guest session at roughly 300 videos
# an hour (~1000 requests) before the bot wall, and Google documents truncated
# exponential backoff with jitter for its own APIs, which is what backoff_base
# and backoff_cap feed.
delay:
min_seconds: 1.5 # randomised gap between videos / channels
max_seconds: 3.5
backoff_base: 2.0 # after a rate-limit: min(base * 2**n + jitter, cap)
backoff_cap: 60.0
throttle_threshold: 3 # consecutive rate-limit responses that abort the run
# Floor on the gap between ANY two requests to YouTube from this process,
# including the /api/tools/* endpoints that run outside the job runner.
# 0 = off. Raise it if you still hit the bot wall; it is the one setting that
# applies everywhere at once.
min_request_interval: 0.0
audio_rate_limit: 0 # bytes/sec ceiling for audio downloads, 0 = unlimited
yt_dlp:
retries: 10 # download retries (only bites on the audio path)
# Seconds between the individual HTTP calls inside one extraction — the watch
# page, the InnerTube player call, each continuation page of a channel tab.
# Forwarded to yt-dlp as `sleep_interval_requests`.
sleep_subrequests: 2
extractor_retries: 3 # retries during extraction (5xx and network only:
# yt-dlp's YouTube extractor never retries 403/429)
socket_timeout: 30.0 # without this a hung connection blocks the worker
# Incremental channel sync. Re-scanning a tracked channel reads the /videos tab
# newest-first and stops once it sees `overlap` videos already in the DB, so a
# routine sync costs one page instead of paginating the whole channel. If every
# video in the window turns out to be new, the window doubles (up to
# max_window) rather than silently missing uploads.
# Set incremental: false — or pass --full / tick "Full channel rescan" — to walk
# every page, which is only needed when the local catalog is incomplete.
sync:
incremental: true
window: 30 # entries read on the first pass
max_window: 300 # ceiling before reporting a truncated scan
overlap: 3 # consecutive known videos that prove we caught up
database_path: "data/state.db"
output_dir: "data/markdown"
template_path: "templates/video.md.j2"
filename_template: "{upload_date}_{slug}"