speak() now blocks until playback finishes, reporting progress via SSE (5% entry tone → 30% synthesis → 35-99% playing → 100% done). Entry tone fires immediately on call to cover synthesis latency. Queue shutdown waits for current speech to finish (configurable timeout, default 30s) before draining pending items — no more mid-sentence cutoffs on container restart. Cancellation via cancel_speech() tool or MCP notifications/cancelled kills pw-play and plays a vinyl scratch tone. Consumer continues to next item after cancel. Progress tracking uses a background ticker task instead of asyncio.wait_for polling — the latter causes stale CancelledError propagation to the consumer under Python 3.13.
74 lines
2.0 KiB
YAML
74 lines
2.0 KiB
YAML
services:
|
|
tts-mcp:
|
|
build: .
|
|
container_name: tts-mcp
|
|
restart: unless-stopped
|
|
stop_grace_period: 35s
|
|
env_file: .env
|
|
environment:
|
|
# Override for Docker networking (container DNS instead of IPs)
|
|
TTS_PIPER_HOST: piper-tts
|
|
TTS_ORPHEUS_URL: http://llama-server:8081
|
|
# PipeWire client config
|
|
XDG_RUNTIME_DIR: /run/user/1000
|
|
# Force unbuffered Python output so stderr shows up in docker logs immediately
|
|
PYTHONUNBUFFERED: "1"
|
|
volumes:
|
|
# Kokoro ONNX models (read-only)
|
|
- ./models:/app/models:ro
|
|
# HuggingFace cache for SNAC model download (lazy-loaded on first Orpheus call)
|
|
- hf-cache:/home/tts/.cache/huggingface
|
|
# Persistent data (voice assignments, etc.)
|
|
- tts-data:/data
|
|
# PipeWire socket for audio playback through host speakers
|
|
- /run/user/1000/pipewire-0:/run/user/1000/pipewire-0
|
|
depends_on:
|
|
llama-server:
|
|
condition: service_healthy
|
|
networks:
|
|
- caddy
|
|
- dootie-internal
|
|
labels:
|
|
caddy: mctalkbox.l.supported.systems
|
|
caddy.reverse_proxy: "{{upstreams 8371}}"
|
|
|
|
llama-server:
|
|
build:
|
|
context: .
|
|
dockerfile: llama-server.Dockerfile
|
|
container_name: orpheus-llama-server
|
|
restart: unless-stopped
|
|
deploy:
|
|
resources:
|
|
reservations:
|
|
devices:
|
|
- driver: nvidia
|
|
count: 1
|
|
capabilities: [gpu]
|
|
volumes:
|
|
# GGUF model file (set ORPHEUS_GGUF_PATH in .env, e.g. from Ollama blob storage)
|
|
- ${ORPHEUS_GGUF_PATH}:/models/orpheus.gguf:ro
|
|
command: >-
|
|
--host 0.0.0.0 --port 8081
|
|
--model /models/orpheus.gguf
|
|
--n-gpu-layers 999 --ctx-size 4096
|
|
--flash-attn --cont-batching
|
|
networks:
|
|
- dootie-internal
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-sf", "http://127.0.0.1:8081/health"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
start_period: 120s
|
|
retries: 5
|
|
|
|
volumes:
|
|
hf-cache:
|
|
tts-data:
|
|
|
|
networks:
|
|
caddy:
|
|
external: true
|
|
dootie-internal:
|
|
external: true
|