Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -91,3 +91,20 @@ STUN_URL=stun:stun.l.google.com:19302
# the entire RFC1918 docker range. Set to an empty value (or a
# different CIDR) to override; multiple values are comma-separated.
# ICE_DROP_NETWORKS=172.16.0.0/12

# --- Encoder threading -----------------------------------------------------
# libvpx's thread_count and cpu-used together control how aggressively
# the encoder uses CPU. Default (auto) leaves cores for the rest of the
# stack - critical when the TS6 server lives on the same host, because
# libvpx burning every core mid-frame can starve the TS6 process long
# enough for its watchdog to kill it.
#
# ENCODER_THREAD_COUNT: 0 = auto (cpu_count - 2, capped at 4). Override
# only if you've measured that your TS6 server has CPU headroom to
# spare and your encoder is the bottleneck.
# ENCODER_THREAD_COUNT=0
#
# ENCODER_CPU_USED: -16..16. Lower = faster + lower quality. Default
# -6 mirrors aiortc; -8 or -10 if your encoder can't sustain real-time
# at the configured resolution.
# ENCODER_CPU_USED=-6
26 changes: 26 additions & 0 deletions src/ts6_stream_bot/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -150,6 +150,32 @@ class Settings(BaseSettings):
TURN_USERNAME: str = Field(default="", description="TURN auth username.")
TURN_PASSWORD: str = Field(default="", description="TURN auth password.")

# --- Encoder threading ------------------------------------------------
# libvpx's thread_count knob is the difference between "encoder uses
# all your cores at once" (which on a colocated TS6 host can starve
# the TS6 process long enough for its watchdog to kill it) and
# "encoder leaves cores for everything else". 0 = let the broadcaster
# auto-pick a safe value (cpu_count - 2, capped at 4).
ENCODER_THREAD_COUNT: int = Field(
default=0,
description=(
"Override for libvpx's thread_count. 0 = auto (cpu_count - 2, "
"max 4). Bump only if you've measured that the TS6 server has "
"headroom to spare and your encoder is the bottleneck."
),
)
# cpu-used trades quality for speed. -16..16, lower = faster + lower
# quality. aiortc's default is -6; we mirror it. Drop to -8 / -10
# if your host can't keep up at the configured resolution.
ENCODER_CPU_USED: int = Field(
default=-6,
description=(
"libvpx cpu-used (-16..16). Lower is faster + lower quality. "
"Default -6 mirrors aiortc; -8 or -10 if encoder can't "
"sustain real-time."
),
)

# When the bot runs in `network_mode: host`, aiortc gathers a host
# candidate for *every* interface in the host namespace - that
# includes Docker's bridge gateways (172.16.0.0/12) which no
Expand Down
2 changes: 2 additions & 0 deletions src/ts6_stream_bot/pipeline/controller.py
Original file line number Diff line number Diff line change
Expand Up @@ -269,6 +269,8 @@ async def _allocate_stream(self) -> None:
width=settings.SCREEN_WIDTH,
height=settings.SCREEN_HEIGHT,
framerate=settings.SCREEN_FPS,
cpu_used=settings.ENCODER_CPU_USED,
thread_count=settings.ENCODER_THREAD_COUNT or None,
),
)
publisher = StreamPublisher(
Expand Down
29 changes: 22 additions & 7 deletions src/ts6_stream_bot/pipeline/video_broadcaster.py
Original file line number Diff line number Diff line change
Expand Up @@ -374,14 +374,29 @@ def _fanout(self, packet: Packet) -> None:


def _auto_thread_count(pixels: int, cpu_count: int) -> int:
"""Match aiortc's libvpx thread heuristic (vpx.number_of_threads)."""
"""Pick a libvpx thread count that leaves CPU headroom for the
rest of the stack.

aiortc's vpx encoder picks ``min(cpu_count, 8)`` outright. For us
that is too aggressive: when the bot lives on the same host as
the TS6 server (the common deployment), libvpx burning every
core mid-frame can starve the TS6 process long enough that its
own watchdog kills it. We cap at ``cpu_count - 2`` (and at most
4 absolute) so the event loop, ffmpeg's reader thread, and the
TS6 server always have a core to make progress on. Operators
with more headroom can override via ``VideoBroadcasterConfig.thread_count``.
"""
aiortc_pick = 0
if pixels >= 1920 * 1080:
return min(cpu_count, 8)
if pixels >= 1280 * 720:
return min(cpu_count, 4)
if pixels >= 640 * 480:
return min(cpu_count, 2)
return 1
aiortc_pick = 8
elif pixels >= 1280 * 720:
aiortc_pick = 4
elif pixels >= 640 * 480:
aiortc_pick = 2
else:
return 1
headroom_pick = max(1, cpu_count - 2)
return min(aiortc_pick, headroom_pick, cpu_count, 4)


__all__ = [
Expand Down
23 changes: 23 additions & 0 deletions tests/test_video_broadcaster.py
Original file line number Diff line number Diff line change
Expand Up @@ -235,6 +235,29 @@ async def test_is_alive_true_after_start() -> None:
await bc.stop()


def test_auto_thread_count_caps_at_four_and_leaves_headroom() -> None:
"""Regression: aiortc's vanilla heuristic returned ``min(cpu_count, 8)``
for 1080p, which on a 6-vCPU host running TS6 next door starved the
TS6 process during keyframes long enough that its watchdog killed
it. We never use more than 4 threads and always leave at least 2
cores for the rest of the stack."""
from ts6_stream_bot.pipeline.video_broadcaster import _auto_thread_count

px_1080 = 1920 * 1080
px_720 = 1280 * 720

# 6-vCPU box (the live deployment): 1080p capped at 4, not 6/8.
assert _auto_thread_count(px_1080, cpu_count=6) == 4
# 8-vCPU box: still capped at 4.
assert _auto_thread_count(px_1080, cpu_count=8) == 4
# Tiny 2-vCPU box: leave one core free even at 1080p.
assert _auto_thread_count(px_1080, cpu_count=2) == 1
# 720p on a beefy box: 4 threads are enough, less than 1080p's pick.
assert _auto_thread_count(px_720, cpu_count=8) == 4
# 720p on the live 6-vCPU box: 4 threads, not 6.
assert _auto_thread_count(px_720, cpu_count=6) == 4


async def test_drain_to_latest_skips_backlog_keeps_newest() -> None:
"""The MediaPlayer queue grows unbounded if we don't keep up. We
pull whatever's already buffered and keep only the newest frame
Expand Down
Loading