diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh index f0e9993..fcc3c4d 100644 --- a/docker/entrypoint.sh +++ b/docker/entrypoint.sh @@ -28,13 +28,23 @@ cleanup() { # "Server is already active for display N" / "Daemon already running". rm -f "$XVFB_LOCK" "$XVFB_SOCKET" 2>/dev/null || true rm -f "${PULSE_PIDS[@]}" 2>/dev/null || true + find /run /var/run /root /tmp /home -name "pid" -path "*pulse*" -delete 2>/dev/null || true } trap cleanup EXIT INT TERM # Same hygiene up-front, in case the previous run died without our trap firing -# (e.g. SIGKILL from a crash). +# (e.g. SIGKILL from a crash). When the PREVIOUS run was OOM-killed, the trap +# never ran, the pid file persists, AND a stray pulseaudio child may still be +# around inside the same restarted container - so kill any leftover process +# explicitly before file cleanup, otherwise pulseaudio's "is the pid still +# alive?" probe returns true and we get "Daemon already running". +pkill -9 -x pulseaudio 2>/dev/null || true +pkill -9 -x Xvfb 2>/dev/null || true +sleep 0.2 rm -f "$XVFB_LOCK" "$XVFB_SOCKET" 2>/dev/null || true rm -f "${PULSE_PIDS[@]}" 2>/dev/null || true +# Catch any pulse pid files in unexpected XDG_RUNTIME_DIR variants. +find /run /var/run /root /tmp /home -name "pid" -path "*pulse*" -delete 2>/dev/null || true # --- Xvfb ------------------------------------------------------------------ log "starting Xvfb on $DISPLAY (${SCREEN_WIDTH:-1920}x${SCREEN_HEIGHT:-1080}x24)" diff --git a/src/ts6_stream_bot/config.py b/src/ts6_stream_bot/config.py index aa30031..1ec7ba0 100644 --- a/src/ts6_stream_bot/config.py +++ b/src/ts6_stream_bot/config.py @@ -32,9 +32,14 @@ class Settings(BaseSettings): LOG_LEVEL: str = "INFO" # --- Display / capture ------------------------------------------------- + # 720p30 keeps the per-viewer VP8 encode out of OOM territory on + # small hosts. aiortc + PyAV at 1080p30 with even one peer can run a + # 4 GB host out of memory (live deploy hit SIGKILL after ~6 s of + # streaming). Bump SCREEN_WIDTH/SCREEN_HEIGHT in .env if your host + # has the budget; you'll also want to bump STREAM_BITRATE. DISPLAY: str = ":99" - SCREEN_WIDTH: int = 1920 - SCREEN_HEIGHT: int = 1080 + SCREEN_WIDTH: int = 1280 + SCREEN_HEIGHT: int = 720 SCREEN_FPS: int = 30 # --- Audio ------------------------------------------------------------- diff --git a/src/ts6_stream_bot/pipeline/stream_publisher.py b/src/ts6_stream_bot/pipeline/stream_publisher.py index f3ffb60..63e979b 100644 --- a/src/ts6_stream_bot/pipeline/stream_publisher.py +++ b/src/ts6_stream_bot/pipeline/stream_publisher.py @@ -27,7 +27,12 @@ from dataclasses import dataclass, field import structlog -from aiortc import RTCPeerConnection, RTCSessionDescription +from aiortc import ( + RTCConfiguration, + RTCIceServer, + RTCPeerConnection, + RTCSessionDescription, +) from aiortc.contrib.media import MediaRelay from aiortc.sdp import candidate_from_sdp @@ -257,7 +262,17 @@ async def _handle_viewer_join(self, viewer_clid: int, stream_id: str) -> None: with contextlib.suppress(Exception): await existing.pc.close() - pc = RTCPeerConnection() + # STUN gives the bot a server-reflexive (srflx) candidate so its own + # NAT/Docker-bridge address is publishable. Without it, NAT punching + # depended entirely on the viewer's side reaching us, and the live + # trace showed the first connection attempt timing out on + # ice=checking before completing on the retry. Google's public STUN + # is the conventional default. + pc = RTCPeerConnection( + configuration=RTCConfiguration( + iceServers=[RTCIceServer(urls="stun:stun.l.google.com:19302")], + ) + ) # Surface aiortc's own connection-state lifecycle so we can tell # whether a viewer disconnected because their client closed the diff --git a/src/ts6_stream_bot/ts3lib/client.py b/src/ts6_stream_bot/ts3lib/client.py index 9b1c44a..9ea6da2 100644 --- a/src/ts6_stream_bot/ts3lib/client.py +++ b/src/ts6_stream_bot/ts3lib/client.py @@ -941,7 +941,8 @@ async def _ping_loop(self) -> None: break self._send_outgoing(b"", PacketType.PING) # Every 10s, log "still alive" with last-incoming-message age - # so a complete absence of incoming traffic is unambiguous. + # plus resident memory so a creeping leak / OOM trajectory + # shows up before docker SIGKILLs us. now = time.monotonic() if now - last_heartbeat >= 10.0: last_heartbeat = now @@ -949,6 +950,7 @@ async def _ping_loop(self) -> None: "ts3.heartbeat", client_id=self._client_id, seconds_since_last_inbound=round(now - self._last_message_time, 1), + rss_mb=_rss_mb(), ) except asyncio.CancelledError: pass @@ -1011,6 +1013,19 @@ def _guard(self) -> Ts3Client._Suppressor: # --- helpers --------------------------------------------------------------- +def _rss_mb() -> int: + """Resident memory in MB for the heartbeat log. Reads /proc/self/status + so we don't add a psutil dependency just for this.""" + try: + with open("/proc/self/status", encoding="ascii") as f: + for line in f: + if line.startswith("VmRSS:"): + return int(line.split()[1]) // 1024 + except OSError: + pass + return 0 + + class _Ts3DatagramProtocol(asyncio.DatagramProtocol): """Thin asyncio glue that pumps datagrams into the parent client."""