diff --git a/.env.example b/.env.example index 28a5099..90adc33 100644 --- a/.env.example +++ b/.env.example @@ -91,3 +91,20 @@ STUN_URL=stun:stun.l.google.com:19302 # the entire RFC1918 docker range. Set to an empty value (or a # different CIDR) to override; multiple values are comma-separated. # ICE_DROP_NETWORKS=172.16.0.0/12 + +# --- Encoder threading ----------------------------------------------------- +# libvpx's thread_count and cpu-used together control how aggressively +# the encoder uses CPU. Default (auto) leaves cores for the rest of the +# stack - critical when the TS6 server lives on the same host, because +# libvpx burning every core mid-frame can starve the TS6 process long +# enough for its watchdog to kill it. +# +# ENCODER_THREAD_COUNT: 0 = auto (cpu_count - 2, capped at 4). Override +# only if you've measured that your TS6 server has CPU headroom to +# spare and your encoder is the bottleneck. +# ENCODER_THREAD_COUNT=0 +# +# ENCODER_CPU_USED: -16..16. Lower = faster + lower quality. Default +# -6 mirrors aiortc; -8 or -10 if your encoder can't sustain real-time +# at the configured resolution. +# ENCODER_CPU_USED=-6 diff --git a/src/ts6_stream_bot/config.py b/src/ts6_stream_bot/config.py index 42f3f01..91a4260 100644 --- a/src/ts6_stream_bot/config.py +++ b/src/ts6_stream_bot/config.py @@ -150,6 +150,32 @@ class Settings(BaseSettings): TURN_USERNAME: str = Field(default="", description="TURN auth username.") TURN_PASSWORD: str = Field(default="", description="TURN auth password.") + # --- Encoder threading ------------------------------------------------ + # libvpx's thread_count knob is the difference between "encoder uses + # all your cores at once" (which on a colocated TS6 host can starve + # the TS6 process long enough for its watchdog to kill it) and + # "encoder leaves cores for everything else". 0 = let the broadcaster + # auto-pick a safe value (cpu_count - 2, capped at 4). + ENCODER_THREAD_COUNT: int = Field( + default=0, + description=( + "Override for libvpx's thread_count. 0 = auto (cpu_count - 2, " + "max 4). Bump only if you've measured that the TS6 server has " + "headroom to spare and your encoder is the bottleneck." + ), + ) + # cpu-used trades quality for speed. -16..16, lower = faster + lower + # quality. aiortc's default is -6; we mirror it. Drop to -8 / -10 + # if your host can't keep up at the configured resolution. + ENCODER_CPU_USED: int = Field( + default=-6, + description=( + "libvpx cpu-used (-16..16). Lower is faster + lower quality. " + "Default -6 mirrors aiortc; -8 or -10 if encoder can't " + "sustain real-time." + ), + ) + # When the bot runs in `network_mode: host`, aiortc gathers a host # candidate for *every* interface in the host namespace - that # includes Docker's bridge gateways (172.16.0.0/12) which no diff --git a/src/ts6_stream_bot/pipeline/controller.py b/src/ts6_stream_bot/pipeline/controller.py index 8621b81..9d10120 100644 --- a/src/ts6_stream_bot/pipeline/controller.py +++ b/src/ts6_stream_bot/pipeline/controller.py @@ -269,6 +269,8 @@ async def _allocate_stream(self) -> None: width=settings.SCREEN_WIDTH, height=settings.SCREEN_HEIGHT, framerate=settings.SCREEN_FPS, + cpu_used=settings.ENCODER_CPU_USED, + thread_count=settings.ENCODER_THREAD_COUNT or None, ), ) publisher = StreamPublisher( diff --git a/src/ts6_stream_bot/pipeline/video_broadcaster.py b/src/ts6_stream_bot/pipeline/video_broadcaster.py index a56159f..41d7eba 100644 --- a/src/ts6_stream_bot/pipeline/video_broadcaster.py +++ b/src/ts6_stream_bot/pipeline/video_broadcaster.py @@ -374,14 +374,29 @@ def _fanout(self, packet: Packet) -> None: def _auto_thread_count(pixels: int, cpu_count: int) -> int: - """Match aiortc's libvpx thread heuristic (vpx.number_of_threads).""" + """Pick a libvpx thread count that leaves CPU headroom for the + rest of the stack. + + aiortc's vpx encoder picks ``min(cpu_count, 8)`` outright. For us + that is too aggressive: when the bot lives on the same host as + the TS6 server (the common deployment), libvpx burning every + core mid-frame can starve the TS6 process long enough that its + own watchdog kills it. We cap at ``cpu_count - 2`` (and at most + 4 absolute) so the event loop, ffmpeg's reader thread, and the + TS6 server always have a core to make progress on. Operators + with more headroom can override via ``VideoBroadcasterConfig.thread_count``. + """ + aiortc_pick = 0 if pixels >= 1920 * 1080: - return min(cpu_count, 8) - if pixels >= 1280 * 720: - return min(cpu_count, 4) - if pixels >= 640 * 480: - return min(cpu_count, 2) - return 1 + aiortc_pick = 8 + elif pixels >= 1280 * 720: + aiortc_pick = 4 + elif pixels >= 640 * 480: + aiortc_pick = 2 + else: + return 1 + headroom_pick = max(1, cpu_count - 2) + return min(aiortc_pick, headroom_pick, cpu_count, 4) __all__ = [ diff --git a/tests/test_video_broadcaster.py b/tests/test_video_broadcaster.py index be117bd..890415b 100644 --- a/tests/test_video_broadcaster.py +++ b/tests/test_video_broadcaster.py @@ -235,6 +235,29 @@ async def test_is_alive_true_after_start() -> None: await bc.stop() +def test_auto_thread_count_caps_at_four_and_leaves_headroom() -> None: + """Regression: aiortc's vanilla heuristic returned ``min(cpu_count, 8)`` + for 1080p, which on a 6-vCPU host running TS6 next door starved the + TS6 process during keyframes long enough that its watchdog killed + it. We never use more than 4 threads and always leave at least 2 + cores for the rest of the stack.""" + from ts6_stream_bot.pipeline.video_broadcaster import _auto_thread_count + + px_1080 = 1920 * 1080 + px_720 = 1280 * 720 + + # 6-vCPU box (the live deployment): 1080p capped at 4, not 6/8. + assert _auto_thread_count(px_1080, cpu_count=6) == 4 + # 8-vCPU box: still capped at 4. + assert _auto_thread_count(px_1080, cpu_count=8) == 4 + # Tiny 2-vCPU box: leave one core free even at 1080p. + assert _auto_thread_count(px_1080, cpu_count=2) == 1 + # 720p on a beefy box: 4 threads are enough, less than 1080p's pick. + assert _auto_thread_count(px_720, cpu_count=8) == 4 + # 720p on the live 6-vCPU box: 4 threads, not 6. + assert _auto_thread_count(px_720, cpu_count=6) == 4 + + async def test_drain_to_latest_skips_backlog_keeps_newest() -> None: """The MediaPlayer queue grows unbounded if we don't keep up. We pull whatever's already buffered and keep only the newest frame