mcspeak/docker-compose.yml
Ryan Malloy 79bfe2a89b Fix sequential timeout stacking in shutdown budget
The same shutdown_timeout was used for both uvicorn handler drain AND
queue.stop() consumer drain — sequential phases sharing Docker's
wall-clock budget. With 30s each, worst case was 60s, exceeding
the 35s stop_grace_period and causing SIGKILL.

Fix: uvicorn gets a fixed 3s drain (handlers just re-raise), queue
gets the full shutdown_timeout. Docker grace = 3 + timeout + 5s safety.

Also adds shutdown observability: startup logs the timing chain,
queue.stop() logs remaining audio vs available budget.
2026-03-02 18:28:04 -07:00

76 lines
2.2 KiB
YAML

services:
tts-mcp:
build: .
container_name: tts-mcp
restart: unless-stopped
# Must exceed: 3s handler drain + TTS_SHUTDOWN_TIMEOUT + 5s safety margin.
# Default: 3 + 30 + 5 = 38s. Increase if TTS_SHUTDOWN_TIMEOUT > 30.
stop_grace_period: 38s
env_file: .env
environment:
# Override for Docker networking (container DNS instead of IPs)
TTS_PIPER_HOST: piper-tts
TTS_ORPHEUS_URL: http://llama-server:8081
# PipeWire client config
XDG_RUNTIME_DIR: /run/user/1000
# Force unbuffered Python output so stderr shows up in docker logs immediately
PYTHONUNBUFFERED: "1"
volumes:
# Kokoro ONNX models (read-only)
- ./models:/app/models:ro
# HuggingFace cache for SNAC model download (lazy-loaded on first Orpheus call)
- hf-cache:/home/tts/.cache/huggingface
# Persistent data (voice assignments, etc.)
- tts-data:/data
# PipeWire socket for audio playback through host speakers
- /run/user/1000/pipewire-0:/run/user/1000/pipewire-0
depends_on:
llama-server:
condition: service_healthy
networks:
- caddy
- dootie-internal
labels:
caddy: mctalkbox.l.supported.systems
caddy.reverse_proxy: "{{upstreams 8371}}"
llama-server:
build:
context: .
dockerfile: llama-server.Dockerfile
container_name: orpheus-llama-server
restart: unless-stopped
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: 1
capabilities: [gpu]
volumes:
# GGUF model file (set ORPHEUS_GGUF_PATH in .env, e.g. from Ollama blob storage)
- ${ORPHEUS_GGUF_PATH}:/models/orpheus.gguf:ro
command: >-
--host 0.0.0.0 --port 8081
--model /models/orpheus.gguf
--n-gpu-layers 999 --ctx-size 4096
--flash-attn --cont-batching
networks:
- dootie-internal
healthcheck:
test: ["CMD", "curl", "-sf", "http://127.0.0.1:8081/health"]
interval: 15s
timeout: 5s
start_period: 120s
retries: 5
volumes:
hf-cache:
tts-data:
networks:
caddy:
external: true
dootie-internal:
external: true