From 506cbf0fc38610b11579eaeddb1bbe96fae7469c Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 1 May 2026 19:09:08 +0000 Subject: [PATCH] Address bot crash on viewer kick: STUN, lower defaults, robust restart MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The previous live deploy showed the bot exit 137 (SIGKILL) ~6 seconds into a successful viewer connection. Operator confirmed: the kick is the SYMPTOM of the crash, not the cause - bot dies, viewer's session follows. Likely root cause is OOM from the host: aiortc + PyAV doing 1080p30 VP8 encoding for even one peer is heavyweight, and the container had no memory budget set. Four changes: 1. Add STUN to RTCPeerConnection. The first-attempt ICE timeout we saw in the trace (`ice=checking` for 20 s before the client gave up and sent cmd=reconnect) is exactly the symptom of the bot lacking a server-reflexive candidate when both peers are behind NAT. Default to stun:stun.l.google.com:19302 - same conventional value the browser-side WebRTC stacks use out of the box. 2. Lower SCREEN_WIDTH/SCREEN_HEIGHT defaults from 1920x1080 to 1280x720. 720p30 keeps the per-viewer encode comfortably under the OOM line on a 4 GB host. Bump in .env if your host has the budget; remember to bump STREAM_BITRATE alongside. 3. Robust entrypoint cleanup against SIGKILL. The trap doesn't run on SIGKILL so the pid file persists, AND a stray pulseaudio child can linger inside the same restarted container. Now we pkill -9 pulseaudio + Xvfb up front and `find ... -name pid -path '*pulse*' -delete` to catch XDG_RUNTIME_DIR variants we missed in the explicit path list. Same on shutdown. 4. Memory in heartbeat: `ts3.heartbeat ... rss_mb=…` so the next exit-137 trace shows the trajectory before the kill. ruff + mypy strict + 218 tests pass. https://claude.ai/code/session_016DuCjRJK995Tj9aDhhB9at --- docker/entrypoint.sh | 12 +++++++++++- src/ts6_stream_bot/config.py | 9 +++++++-- .../pipeline/stream_publisher.py | 19 +++++++++++++++++-- src/ts6_stream_bot/ts3lib/client.py | 17 ++++++++++++++++- 4 files changed, 51 insertions(+), 6 deletions(-) diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh index f0e9993..fcc3c4d 100644 --- a/docker/entrypoint.sh +++ b/docker/entrypoint.sh @@ -28,13 +28,23 @@ cleanup() { # "Server is already active for display N" / "Daemon already running". rm -f "$XVFB_LOCK" "$XVFB_SOCKET" 2>/dev/null || true rm -f "${PULSE_PIDS[@]}" 2>/dev/null || true + find /run /var/run /root /tmp /home -name "pid" -path "*pulse*" -delete 2>/dev/null || true } trap cleanup EXIT INT TERM # Same hygiene up-front, in case the previous run died without our trap firing -# (e.g. SIGKILL from a crash). +# (e.g. SIGKILL from a crash). When the PREVIOUS run was OOM-killed, the trap +# never ran, the pid file persists, AND a stray pulseaudio child may still be +# around inside the same restarted container - so kill any leftover process +# explicitly before file cleanup, otherwise pulseaudio's "is the pid still +# alive?" probe returns true and we get "Daemon already running". +pkill -9 -x pulseaudio 2>/dev/null || true +pkill -9 -x Xvfb 2>/dev/null || true +sleep 0.2 rm -f "$XVFB_LOCK" "$XVFB_SOCKET" 2>/dev/null || true rm -f "${PULSE_PIDS[@]}" 2>/dev/null || true +# Catch any pulse pid files in unexpected XDG_RUNTIME_DIR variants. +find /run /var/run /root /tmp /home -name "pid" -path "*pulse*" -delete 2>/dev/null || true # --- Xvfb ------------------------------------------------------------------ log "starting Xvfb on $DISPLAY (${SCREEN_WIDTH:-1920}x${SCREEN_HEIGHT:-1080}x24)" diff --git a/src/ts6_stream_bot/config.py b/src/ts6_stream_bot/config.py index aa30031..1ec7ba0 100644 --- a/src/ts6_stream_bot/config.py +++ b/src/ts6_stream_bot/config.py @@ -32,9 +32,14 @@ class Settings(BaseSettings): LOG_LEVEL: str = "INFO" # --- Display / capture ------------------------------------------------- + # 720p30 keeps the per-viewer VP8 encode out of OOM territory on + # small hosts. aiortc + PyAV at 1080p30 with even one peer can run a + # 4 GB host out of memory (live deploy hit SIGKILL after ~6 s of + # streaming). Bump SCREEN_WIDTH/SCREEN_HEIGHT in .env if your host + # has the budget; you'll also want to bump STREAM_BITRATE. DISPLAY: str = ":99" - SCREEN_WIDTH: int = 1920 - SCREEN_HEIGHT: int = 1080 + SCREEN_WIDTH: int = 1280 + SCREEN_HEIGHT: int = 720 SCREEN_FPS: int = 30 # --- Audio ------------------------------------------------------------- diff --git a/src/ts6_stream_bot/pipeline/stream_publisher.py b/src/ts6_stream_bot/pipeline/stream_publisher.py index f3ffb60..63e979b 100644 --- a/src/ts6_stream_bot/pipeline/stream_publisher.py +++ b/src/ts6_stream_bot/pipeline/stream_publisher.py @@ -27,7 +27,12 @@ from dataclasses import dataclass, field import structlog -from aiortc import RTCPeerConnection, RTCSessionDescription +from aiortc import ( + RTCConfiguration, + RTCIceServer, + RTCPeerConnection, + RTCSessionDescription, +) from aiortc.contrib.media import MediaRelay from aiortc.sdp import candidate_from_sdp @@ -257,7 +262,17 @@ async def _handle_viewer_join(self, viewer_clid: int, stream_id: str) -> None: with contextlib.suppress(Exception): await existing.pc.close() - pc = RTCPeerConnection() + # STUN gives the bot a server-reflexive (srflx) candidate so its own + # NAT/Docker-bridge address is publishable. Without it, NAT punching + # depended entirely on the viewer's side reaching us, and the live + # trace showed the first connection attempt timing out on + # ice=checking before completing on the retry. Google's public STUN + # is the conventional default. + pc = RTCPeerConnection( + configuration=RTCConfiguration( + iceServers=[RTCIceServer(urls="stun:stun.l.google.com:19302")], + ) + ) # Surface aiortc's own connection-state lifecycle so we can tell # whether a viewer disconnected because their client closed the diff --git a/src/ts6_stream_bot/ts3lib/client.py b/src/ts6_stream_bot/ts3lib/client.py index 9b1c44a..9ea6da2 100644 --- a/src/ts6_stream_bot/ts3lib/client.py +++ b/src/ts6_stream_bot/ts3lib/client.py @@ -941,7 +941,8 @@ async def _ping_loop(self) -> None: break self._send_outgoing(b"", PacketType.PING) # Every 10s, log "still alive" with last-incoming-message age - # so a complete absence of incoming traffic is unambiguous. + # plus resident memory so a creeping leak / OOM trajectory + # shows up before docker SIGKILLs us. now = time.monotonic() if now - last_heartbeat >= 10.0: last_heartbeat = now @@ -949,6 +950,7 @@ async def _ping_loop(self) -> None: "ts3.heartbeat", client_id=self._client_id, seconds_since_last_inbound=round(now - self._last_message_time, 1), + rss_mb=_rss_mb(), ) except asyncio.CancelledError: pass @@ -1011,6 +1013,19 @@ def _guard(self) -> Ts3Client._Suppressor: # --- helpers --------------------------------------------------------------- +def _rss_mb() -> int: + """Resident memory in MB for the heartbeat log. Reads /proc/self/status + so we don't add a psutil dependency just for this.""" + try: + with open("/proc/self/status", encoding="ascii") as f: + for line in f: + if line.startswith("VmRSS:"): + return int(line.split()[1]) // 1024 + except OSError: + pass + return 0 + + class _Ts3DatagramProtocol(asyncio.DatagramProtocol): """Thin asyncio glue that pumps datagrams into the parent client."""