From eaff3e88615230c47b19ab277f451681ed7108ec Mon Sep 17 00:00:00 2001 From: Ryan Malloy Date: Tue, 24 Feb 2026 11:58:10 -0700 Subject: [PATCH] Add project-aware voice identity with round-robin assignment Each project gets a distinct voice from a curated English pool, assigned via round-robin with gender/accent interleaving for maximum perceptual contrast between consecutive projects. - New voice_identity.py: pool filtering, interleaving, persistence - Round-robin replaces SHA-256 hashing (no collisions until pool exhaustion at 22 voices) - Assignments persist to /data/voice-assignments.json across restarts - speak() and generate_audio() accept optional project= parameter - MCP roots fallback with 2s timeout for future bidirectional clients - English-only pool (af_/am_/bf_/bm_/ef_/em_ prefixes) - af_nicole excluded from auto-assign (whispery), still explicit-ok - Fix voice blacklist to use full identifiers (am_adam, af_jessica) --- CLAUDE.md | 33 +++++ Dockerfile | 4 +- docker-compose.yml | 3 + src/tts_mcp/server.py | 70 +++++++++-- src/tts_mcp/settings.py | 12 +- src/tts_mcp/voice_identity.py | 223 ++++++++++++++++++++++++++++++++++ 6 files changed, 331 insertions(+), 14 deletions(-) create mode 100644 src/tts_mcp/voice_identity.py diff --git a/CLAUDE.md b/CLAUDE.md index 2669c55..40c7b8b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -44,6 +44,39 @@ The standby tone is always the built-in ascending blip. It plays instead of the Tones are generated programmatically at startup (48kHz, 16-bit PCM, -3 dB headroom) in `tones.py` using numpy. No bundled audio assets. +## Voice Identity (Project-Aware Voices) + +When multiple Claude Code sessions connect simultaneously, voice identity gives each project a distinct voice via round-robin assignment from a curated English voice pool. + +### How It Works + +1. Client calls `speak()` or `generate_audio()` without specifying `voice=` +2. Project is identified via the `project` tool parameter, or falls back to MCP Roots (`list_roots()` with 2s timeout) +3. The next unused voice is assigned from the interleaved pool (alternating gender and accent for maximum contrast) +4. Assignment is persisted to `/data/voice-assignments.json` — survives server restarts + +Explicit `voice=` parameter always overrides auto-assignment. Voice pools are cached for 5 minutes (picks up blacklist/engine changes). + +**Note:** MCP Roots require stateful Streamable HTTP. With `stateless_http=True` (current default), roots will timeout — the `project` parameter is the primary identification method. + +### Configuration + +```env +TTS_VOICE_IDENTITY=true # Enable project-aware voice assignment +TTS_VOICE_IDENTITY_PREFIXES=af_,am_,bf_,bm_,ef_,em_ # English voice prefixes +TTS_VOICE_IDENTITY_EXCLUDE=af_nicole # Available explicitly, excluded from auto-assign (whispery) +TTS_VOICE_IDENTITY_FILE=/data/voice-assignments.json # Persist across restarts +TTS_ANNOUNCE_PROJECT=false # Prefix speech with project name +``` + +### Pool Interleave Order + +Voices are interleaved for perceptual diversity: American female → British male → European female → American male → British female → European male. First 6 projects get maximally distinct voices. + +### Files + +- `voice_identity.py` — Pool filtering, interleaving, round-robin assignment, JSON persistence + ## Architecture - `server.py` — FastMCP lifespan, tool definitions, engine setup diff --git a/Dockerfile b/Dockerfile index 1052701..ee1cc14 100644 --- a/Dockerfile +++ b/Dockerfile @@ -28,8 +28,8 @@ RUN --mount=type=cache,target=/root/.cache/uv \ # Non-root user matching host uid for PipeWire socket access RUN useradd -u 1000 -m tts \ - && mkdir -p /home/tts/.cache/huggingface \ - && chown -R tts:tts /home/tts/.cache + && mkdir -p /home/tts/.cache/huggingface /data \ + && chown -R tts:tts /home/tts/.cache /data USER tts ENV PATH="/app/.venv/bin:$PATH" diff --git a/docker-compose.yml b/docker-compose.yml index 08b07b5..7fff276 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -15,6 +15,8 @@ services: - ./models:/app/models:ro # HuggingFace cache for SNAC model download (lazy-loaded on first Orpheus call) - hf-cache:/home/tts/.cache/huggingface + # Persistent data (voice assignments, etc.) + - tts-data:/data # PipeWire socket for audio playback through host speakers - /run/user/1000/pipewire-0:/run/user/1000/pipewire-0 depends_on: @@ -59,6 +61,7 @@ services: volumes: hf-cache: + tts-data: networks: caddy: diff --git a/src/tts_mcp/server.py b/src/tts_mcp/server.py index 18dc125..5a511e0 100644 --- a/src/tts_mcp/server.py +++ b/src/tts_mcp/server.py @@ -16,6 +16,7 @@ from .engines.piper import PiperEngine from .queue import Priority, SpeechQueue from .settings import settings from .tones import generate_tones, resolve_tone +from .voice_identity import VoiceIdentityCache, get_project_name, resolve_voice ENGINE_NAMES = Literal["piper", "kokoro", "orpheus"] @@ -70,8 +71,17 @@ async def app_lifespan(server: FastMCP): file=sys.stderr, ) + voice_cache = VoiceIdentityCache(persist_path=settings.voice_identity_file) + + identity_label = "on" if settings.voice_identity else "off" + announce_label = "+announce" if settings.announce_project else "" + print( + f" Voice identity: {identity_label}{announce_label}", + file=sys.stderr, + ) + try: - yield {"engines": engines, "queue": queue} + yield {"engines": engines, "queue": queue, "voice_cache": voice_cache} finally: print("TTS MCP server shutting down", file=sys.stderr) await queue.stop() @@ -91,16 +101,43 @@ mcp = FastMCP( "through the host speakers (queued so agents don't talk over each other). " "Use 'generate_audio' to synthesize without playing. " "Engines: kokoro (fast ONNX, ~50 voices), piper (Wyoming/Docker), " - "orpheus (LLM via llama-server, supports etc.)." + "orpheus (LLM via llama-server, supports etc.). " + "Always pass project= with the current project directory name (last path component) " + "so each project gets a consistent, distinct voice automatically." ), lifespan=app_lifespan, ) -def _get_state(ctx: Context) -> tuple[dict[str, TTSEngine], SpeechQueue]: - """Extract engines and queue from lifespan context.""" +def _get_state(ctx: Context) -> tuple[dict[str, TTSEngine], SpeechQueue, VoiceIdentityCache]: + """Extract engines, queue, and voice cache from lifespan context.""" state = ctx.lifespan_context - return state["engines"], state["queue"] + return state["engines"], state["queue"], state["voice_cache"] + + +async def _resolve_project_voice( + ctx: Context, + engine_name: str, + eng: TTSEngine, + voice: str | None, + text: str, + voice_cache: VoiceIdentityCache, + project: str | None = None, +) -> tuple[str | None, str]: + """Auto-assign voice from project hint or MCP roots, optionally prefix text.""" + if voice is not None or not settings.voice_identity: + return voice, text + + # Explicit project param takes priority, then try MCP roots + project_name = project or await get_project_name(ctx) + if not project_name: + return voice, text + + voice = await resolve_voice(engine_name, eng, project_name, voice_cache) + if settings.announce_project and voice is not None: + text = f"{project_name}. {text}" + + return voice, text # --------------------------------------------------------------------------- @@ -113,6 +150,7 @@ async def speak( engine: ENGINE_NAMES = "kokoro", voice: str | None = None, urgent: bool = False, + project: str | None = None, ctx: Context = CurrentContext(), ) -> dict: """Synthesize text and play it through the host speakers. @@ -124,15 +162,19 @@ async def speak( Args: text: Text to speak. Orpheus supports emotion tags like , , etc. engine: TTS engine to use. kokoro is fastest, orpheus is most expressive. - voice: Voice name (use list_voices to see options). None = engine default. + voice: Voice name (use list_voices to see options). None = auto-assigned by project. urgent: If True, this message jumps ahead of normal-priority items. + project: Project name for voice identity (auto-detected from MCP roots if omitted). """ - engines, queue = _get_state(ctx) + engines, queue, voice_cache = _get_state(ctx) if engine not in engines: return {"error": f"Unknown engine: {engine}. Available: {list(engines.keys())}"} eng = engines[engine] + voice, text = await _resolve_project_voice( + ctx, engine, eng, voice, text, voice_cache, project, + ) # Synthesize audio (not queued — multiple agents can synthesize simultaneously) await ctx.info(f"Synthesizing with {engine}...") @@ -151,6 +193,7 @@ async def generate_audio( text: str, engine: ENGINE_NAMES = "kokoro", voice: str | None = None, + project: str | None = None, ctx: Context = CurrentContext(), ) -> dict: """Synthesize text to a WAV file without playing it. @@ -161,14 +204,19 @@ async def generate_audio( Args: text: Text to synthesize. engine: TTS engine to use. - voice: Voice name (use list_voices to see options). None = engine default. + voice: Voice name (use list_voices to see options). None = auto-assigned by project. + project: Project name for voice identity (auto-detected from MCP roots if omitted). """ - engines, _ = _get_state(ctx) + engines, _, voice_cache = _get_state(ctx) if engine not in engines: return {"error": f"Unknown engine: {engine}. Available: {list(engines.keys())}"} eng = engines[engine] + voice, text = await _resolve_project_voice( + ctx, engine, eng, voice, text, voice_cache, project, + ) + await ctx.info(f"Generating audio with {engine}...") result = await eng.synthesize(text, voice) @@ -194,7 +242,7 @@ async def list_voices( Args: engine: Which engine to list voices for. """ - engines, _ = _get_state(ctx) + engines, _, _ = _get_state(ctx) if engine not in engines: return [] @@ -210,7 +258,7 @@ async def list_engines( Returns engine name, default voice, and health check results. """ - engines, queue = _get_state(ctx) + engines, queue, _ = _get_state(ctx) results = [] for name, eng in engines.items(): diff --git a/src/tts_mcp/settings.py b/src/tts_mcp/settings.py index 08df587..bc15262 100644 --- a/src/tts_mcp/settings.py +++ b/src/tts_mcp/settings.py @@ -24,11 +24,21 @@ class Settings(BaseSettings): orpheus_url: str = "http://127.0.0.1:8081" # Voice filtering - voice_blacklist: str = "amy,jess,zoe,adam" + voice_blacklist: str = "am_adam,af_jessica" # Audio output (empty = system temp dir) audio_dir: str = "" + # Voice identity (project-aware voice assignment) + voice_identity: bool = True + announce_project: bool = False + # Comma-separated prefixes for auto-assignment pool (English voices) + voice_identity_prefixes: str = "af_,am_,bf_,bm_,ef_,em_" + # Voices available explicitly but excluded from auto-assignment (e.g. whispery) + voice_identity_exclude: str = "af_nicole" + # JSON file to persist project->voice assignments across restarts + voice_identity_file: str = "" + # Entry tone before speech: "chirp", "apollo", "none", or path to custom WAV entry_tone: str = "chirp" diff --git a/src/tts_mcp/voice_identity.py b/src/tts_mcp/voice_identity.py new file mode 100644 index 0000000..b7a2c67 --- /dev/null +++ b/src/tts_mcp/voice_identity.py @@ -0,0 +1,223 @@ +"""Project-aware voice assignment via round-robin. + +Each project is assigned the next unused voice from a curated English +voice pool. Assignments persist to a JSON file so projects keep their +voice across server restarts. +""" + +import asyncio +import json +import sys +import time +from pathlib import Path +from urllib.parse import unquote, urlparse + +from fastmcp import Context + +from .engines.base import TTSEngine +from .settings import settings + + +def extract_project_name(roots: list) -> str | None: + """Extract project name from the first file:// root URI. + + Takes the last path component of the first file:// URI found. + ``file:///home/rpm/fix-tts`` -> ``"fix-tts"`` + """ + for root in roots: + uri = str(root.uri) if hasattr(root, "uri") else str(root) + parsed = urlparse(uri) + if parsed.scheme != "file": + continue + path = unquote(parsed.path).rstrip("/") + if not path: + continue + return path.rsplit("/", 1)[-1] + return None + + +def _build_identity_pool(voices: list[str]) -> list[str]: + """Filter a voice list down to the auto-assignment pool. + + Keeps only voices matching configured prefixes, removes excluded + voices, then interleaves by gender and accent so consecutive + assignments sound maximally distinct. + """ + prefixes = tuple( + p.strip() for p in settings.voice_identity_prefixes.split(",") if p.strip() + ) + excludes = { + v.strip().lower() + for v in settings.voice_identity_exclude.split(",") + if v.strip() + } + eligible = [ + v for v in voices + if v.startswith(prefixes) and v.lower() not in excludes + ] + return _interleave_voices(eligible) + + +def _interleave_voices(voices: list[str]) -> list[str]: + """Interleave voices by category for maximum perceptual diversity. + + Groups by prefix (af_=American female, am_=American male, bf_=British + female, etc.) then round-robins across groups so consecutive voices + alternate gender and accent. + """ + buckets: dict[str, list[str]] = {} + for v in voices: + prefix = v[:3] if len(v) >= 3 else "xx_" + buckets.setdefault(prefix, []).append(v) + + # Order buckets for max contrast: female/male alternating, US/UK/EU + bucket_order = ["af_", "bm_", "ef_", "am_", "bf_", "em_"] + ordered = [k for k in bucket_order if k in buckets] + # Append any remaining prefixes not in the predefined order + ordered += [k for k in sorted(buckets) if k not in ordered] + + result: list[str] = [] + while any(buckets[k] for k in ordered): + for k in ordered: + if buckets[k]: + result.append(buckets[k].pop(0)) + return result + + +class VoiceIdentityCache: + """Server-level cache with round-robin assignment and optional persistence. + + Instead of hashing, projects are assigned voices sequentially — first + project gets voice 0, second gets voice 1, etc. This guarantees + maximum diversity until the pool is exhausted, then wraps around. + + Assignments are persisted to a JSON file (if configured) so projects + keep the same voice across server restarts. + """ + + POOL_TTL = 300 # 5 minutes + + def __init__(self, persist_path: str = "") -> None: + self._voice_pools: dict[str, tuple[list[str], float]] = {} + self._assignments: dict[str, dict[str, str]] = {} # engine -> {project -> voice} + self._persist_path: Path | None = Path(persist_path) if persist_path else None + self._load() + + # -- Persistence ---------------------------------------------------------- + + def _load(self) -> None: + if not self._persist_path or not self._persist_path.exists(): + return + try: + data = json.loads(self._persist_path.read_text()) + self._assignments = data.get("assignments", {}) + count = sum(len(v) for v in self._assignments.values()) + print(f" Voice identity: loaded {count} assignments from {self._persist_path}", file=sys.stderr) + except Exception as exc: + print(f" Voice identity: failed to load {self._persist_path} ({exc})", file=sys.stderr) + + def _save(self) -> None: + if not self._persist_path: + return + try: + self._persist_path.parent.mkdir(parents=True, exist_ok=True) + self._persist_path.write_text(json.dumps({"assignments": self._assignments}, indent=2)) + except Exception as exc: + print(f"Voice identity: failed to save ({exc})", file=sys.stderr) + + # -- Pool cache ----------------------------------------------------------- + + def get_pool(self, engine_name: str) -> list[str] | None: + entry = self._voice_pools.get(engine_name) + if entry is None: + return None + pool, cached_at = entry + if time.monotonic() - cached_at > self.POOL_TTL: + del self._voice_pools[engine_name] + return None + return list(pool) + + def put_pool(self, engine_name: str, voices: list[str]) -> None: + self._voice_pools[engine_name] = (list(voices), time.monotonic()) + + # -- Assignment ----------------------------------------------------------- + + def get(self, engine_name: str, project_name: str) -> str | None: + return self._assignments.get(engine_name, {}).get(project_name) + + def assign_next(self, engine_name: str, project_name: str, pool: list[str]) -> str: + """Assign the next unused voice from the pool via round-robin.""" + engine_assignments = self._assignments.setdefault(engine_name, {}) + + # Check if already assigned + existing = engine_assignments.get(project_name) + if existing and existing in pool: + return existing + + # Find voices already taken + used = set(engine_assignments.values()) + + # Pick the first unused voice + for voice in pool: + if voice not in used: + engine_assignments[project_name] = voice + self._save() + return voice + + # Pool exhausted — wrap around, pick least-used + usage = {} + for v in pool: + usage[v] = sum(1 for assigned in used if assigned == v) + voice = min(pool, key=lambda v: usage.get(v, 0)) + engine_assignments[project_name] = voice + self._save() + return voice + + +async def get_project_name(ctx: Context) -> str | None: + """Fetch project name from MCP roots (server->client round-trip). + + Uses a 2-second timeout because list_roots() hangs indefinitely when + the client doesn't support roots or the transport is stateless HTTP + with no bidirectional channel. + """ + try: + roots = await asyncio.wait_for(ctx.list_roots(), timeout=2.0) + if not roots: + return None + return extract_project_name(roots) + except TimeoutError: + return None + except Exception as exc: + exc_type = type(exc).__name__ + print( + f"Voice identity: roots unavailable ({exc_type}: {exc})", + file=sys.stderr, + ) + return None + + +async def resolve_voice( + engine_name: str, + engine: TTSEngine, + project_name: str, + cache: VoiceIdentityCache, +) -> str | None: + """Resolve voice for a project using round-robin assignment.""" + # Check cache first + cached = cache.get(engine_name, project_name) + if cached is not None: + return cached + + # Build or fetch the identity pool + pool = cache.get_pool(engine_name) + if pool is None: + all_voices = await engine.list_voices() + pool = _build_identity_pool(all_voices) + if not pool: + return None + cache.put_pool(engine_name, pool) + + voice = cache.assign_next(engine_name, project_name, pool) + print(f"Voice assigned: {project_name} -> {engine_name}/{voice}", file=sys.stderr) + return voice