Add transcribe tool — Parakeet STT via gpu.supported.systems

mcspeak gains a `transcribe` MCP tool that accepts an audio file path
and returns text. Hits the Whisper-API-shaped Parakeet endpoint at
mcspeak.gpu.supported.systems/v1/audio/transcriptions with the shared
bearer key from TTS_PARAKEET_KEY. The mcspeak.* subdomain flows
through to Langfuse as user=mcspeak for tenant attribution.

Input audio_path validates against /output/ and /tmp/mcspeak/ —
symmetric with generate_audio's _resolve_output_path discipline. That
covers the primary round-trip use case (transcribing audio mcspeak
just generated) without extra mounts. Path canonicalization rejects
both absolute paths outside the allowlist and ../-traversal escapes.

The tool exposes forward-compat params (timestamp_granularities,
diarize, num_speakers, vad, punctuation, min_confidence) that the
current phonescribe gateway silently ignores. When the gpu-stack
agent lands word-level alignment or speaker diarization, the same
calls start producing richer responses with no client change. The
shaped-now-instead-of-later approach saves a breaking change later.

response_format dispatch: json/verbose_json parse as JSON dict
directly; text/srt/vtt wrap as {"text": <body>, "format": <fmt>}.
The gateway today rejects srt/vtt with HTTP 400 "use json | text |
verbose_json" — that's a clean error users see, not silent garbage.

transcribe.py: shared httpx.AsyncClient mirrored from orpheus.py's
pattern (explicit timeouts, ConnectError/ReadError/TimeoutException
catches). Closed via lifespan finalizer.

httpx 0.28 gotcha: passing `data=` as list-of-tuples silently routes
the value to `content=`, which wraps as SyncByteStream and crashes
the AsyncClient with "Attempted to send a sync request." `data` must
be a Mapping; list values inside the dict become repeated form
fields automatically.
This commit is contained in:
Ryan Malloy 2026-06-16 07:18:13 -06:00
parent 27b4c4ae29
commit f00d0e27a2
4 changed files with 265 additions and 0 deletions

View File

@ -10,6 +10,14 @@ COMPOSE_PROJECT=mcspeak
# https://github.com/rhasspy/piper/blob/master/VOICES.md
# TTS_PIPER_VOICE=es_MX-ald-medium
# Parakeet speech-to-text via the gpu.supported.systems gateway.
# Without TTS_PARAKEET_KEY set, the transcribe() tool returns
# "Parakeet not configured". The shared bearer is documented in
# ~/.claude/rules/gpu.md. The mcspeak.* subdomain flows through to
# Langfuse as user=mcspeak for tenant attribution.
# TTS_PARAKEET_URL=https://mcspeak.gpu.supported.systems/v1/audio/transcriptions
# TTS_PARAKEET_KEY=sk-gpu-lb-master-key-2026
# Path to Orpheus GGUF model file (required by llama-server service)
# Tip: if you've already pulled it via Ollama, find the blob with:
# ollama show --modelfile orpheus | grep FROM

View File

@ -13,6 +13,7 @@ from fastmcp.server.dependencies import CurrentContext
from mcp.types import ToolAnnotations
from .audio import ConversionError, SUPPORTED_FORMATS, convert_audio, play_audio
from .transcribe import TranscriptionError, transcribe_audio
from .engines.base import TTSEngine
from .engines.kokoro import KokoroEngine
from .engines.orpheus import OrpheusEngine
@ -30,6 +31,42 @@ FORMAT_NAMES = Literal["wav", "mp3", "ogg", "flac", "m4a"]
# Host path is configured via TTS_OUTPUT_HOST_DIR (default ~/mcspeak-out).
OUTPUT_DIR_CONTAINER = Path("/output")
# Allowed input dirs for transcribe(). /output is the host-visible bind mount,
# /tmp/mcspeak is the in-container scratch dir for engine-synthesized WAVs.
# Keeping the input scope tight (vs. accepting any container path) mirrors
# the discipline _resolve_output_path uses for writes.
_TRANSCRIBE_INPUT_DIRS = (Path("/output"), Path("/tmp/mcspeak"))
def _validate_readable_audio_path(audio_path: str) -> Path:
"""Resolve audio_path against the allowed input dirs and check existence.
Same canonicalization trick as _resolve_output_path Path.resolve() kills
any ".." traversal before relative_to() checks containment.
"""
p = Path(audio_path).resolve()
bases = [b.resolve() for b in _TRANSCRIBE_INPUT_DIRS]
if not any(_is_under(p, base) for base in bases):
raise ValueError(
f"audio_path must resolve under /output/ or /tmp/mcspeak/ "
f"(got {audio_path!r}). /output/ is bind-mounted from the host's "
f"TTS_OUTPUT_HOST_DIR; /tmp/mcspeak/ holds mcspeak's own "
f"engine-synthesized WAVs."
)
if not p.exists():
raise FileNotFoundError(f"Audio file not found: {p}")
if not p.is_file():
raise ValueError(f"Audio path is not a regular file: {p}")
return p
def _is_under(p: Path, base: Path) -> bool:
try:
p.relative_to(base)
return True
except ValueError:
return False
def _resolve_output_path(output_path: str | None, fmt: str) -> Path:
"""Resolve a user-supplied output_path to an absolute path under /output/.
@ -167,6 +204,9 @@ async def app_lifespan(server: FastMCP):
for eng in engines.values():
if hasattr(eng, "close"):
await eng.close()
# Close the shared Parakeet httpx client (no-op if never used).
from .transcribe import close as _close_transcribe_client
await _close_transcribe_client()
# ---------------------------------------------------------------------------
@ -684,6 +724,66 @@ async def generate_audio(
}
@mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True))
async def transcribe(
audio_path: str,
response_format: Literal["json", "text", "verbose_json", "srt", "vtt"] = "json",
# Forward-compat params (may be no-ops on the current phonescribe gateway).
# Shaped now so the client API doesn't need a breaking change when the
# gateway lands these features.
timestamp_granularities: list[Literal["word", "segment"]] | None = None,
diarize: bool = False,
num_speakers: int | None = None,
vad: bool | None = None,
punctuation: bool | None = None,
min_confidence: float | None = None,
ctx: Context = CurrentContext(),
) -> dict:
"""Transcribe an audio file to text via Parakeet (English-only).
Hits the gpu.supported.systems Whisper-API-shaped endpoint. The
audio_path must resolve under /output/ (host's TTS_OUTPUT_HOST_DIR
bind mount) or /tmp/mcspeak/ (in-container scratch dir for
mcspeak's own synthesized WAVs). Any container ffmpeg can decode
is accepted (WAV, MP3, M4A, FLAC, OGG, ...).
Args:
audio_path: Path to the audio file.
response_format: 'json' (default, {"text": ...}), 'verbose_json'
(adds duration + segments[]), 'text' (raw body in {"text": ...}),
'srt' or 'vtt' (raw subtitle body in {"text": ..., "format": ...}).
timestamp_granularities: ["word"] or ["segment"]. May be a no-op
today; will surface when phonescribe adds word-level alignment.
diarize: Request speaker labels in segments[]. No-op until gpu-stack
adds diarization.
num_speakers: Hint for diarization quality. Ignored without diarize.
vad: Voice-activity-detection preprocessing toggle.
punctuation: Force punctuation/capitalization post-process.
min_confidence: Drop segments below this confidence (0.0-1.0).
"""
try:
path = _validate_readable_audio_path(audio_path)
except (ValueError, FileNotFoundError) as e:
return {"error": str(e)}
await ctx.info(f"Transcribing {path.name}...")
try:
result = await transcribe_audio(
path,
response_format=response_format,
timestamp_granularities=timestamp_granularities,
diarize=diarize,
num_speakers=num_speakers,
vad=vad,
punctuation=punctuation,
min_confidence=min_confidence,
)
except TranscriptionError as e:
return {"error": str(e)}
return result
@mcp.tool(annotations=ToolAnnotations(readOnlyHint=True))
async def list_voices(
engine: ENGINE_NAMES,

View File

@ -27,6 +27,13 @@ class Settings(BaseSettings):
# Orpheus (llama-server + SNAC)
orpheus_url: str = "http://127.0.0.1:8081"
# Parakeet speech-to-text via the gpu.supported.systems gateway.
# The mcspeak.* subdomain flows through to Langfuse as user=mcspeak
# for tenant attribution. The bearer key is required — see
# ~/.claude/rules/gpu.md for the shared value.
parakeet_url: str = "https://mcspeak.gpu.supported.systems/v1/audio/transcriptions"
parakeet_key: str = ""
# Voice filtering
voice_blacklist: str = "am_adam,af_jessica"

150
src/mcspeak/transcribe.py Normal file
View File

@ -0,0 +1,150 @@
"""Speech-to-text via Parakeet through the gpu.supported.systems gateway.
The gateway exposes a Whisper-API-shaped `/v1/audio/transcriptions` endpoint
backed by NVIDIA Parakeet-TDT running on the Mac via MLX (see phonescribe).
English-only. Accepts any container that ffmpeg/pydub can decode.
The forward-compat params (diarize, timestamp_granularities, vad, etc.) are
silently passed through as multipart form fields. The current gateway ignores
unknown fields; when phonescribe adds support, these calls start producing
richer responses with no client change.
"""
from pathlib import Path
import httpx
from .settings import settings
class TranscriptionError(RuntimeError):
"""Raised when transcription fails or is misconfigured."""
# Module-level client — reused across calls, mirrors orpheus.py:120-127.
# Read timeout is generous (90s) because Parakeet runs ~realtime on the Mac
# and the gateway serializes inference on a single pool slot, so a queued
# call can wait its turn before processing begins.
_client: httpx.AsyncClient | None = None
def _get_client() -> httpx.AsyncClient:
global _client
if _client is None:
_client = httpx.AsyncClient(
timeout=httpx.Timeout(connect=10.0, read=90.0, write=30.0, pool=30.0),
limits=httpx.Limits(max_connections=4, max_keepalive_connections=2),
)
return _client
async def close() -> None:
"""Close the shared httpx client. Called from server.py lifespan."""
global _client
if _client is not None:
await _client.aclose()
_client = None
async def transcribe_audio(
path: Path,
response_format: str = "json",
*,
timestamp_granularities: list[str] | None = None,
diarize: bool = False,
num_speakers: int | None = None,
vad: bool | None = None,
punctuation: bool | None = None,
min_confidence: float | None = None,
) -> dict:
"""POST an audio file to the gateway and return parsed results.
Args:
path: Audio file. Any container ffmpeg can decode (WAV/MP3/M4A/...).
response_format: 'json' | 'text' | 'verbose_json' | 'srt' | 'vtt'.
Forward-compat (may be no-ops on the current phonescribe gateway):
timestamp_granularities: e.g. ["word"] or ["word", "segment"].
diarize: speaker labels in segments[].speaker when supported.
num_speakers: optional hint for diarization quality.
vad: voice-activity-detection preprocessing.
punctuation: insert punctuation/capitalization post-process.
min_confidence: drop segments below this confidence.
Returns:
- response_format in {json, verbose_json}: parsed JSON dict.
- response_format in {text, srt, vtt}: {"text": <body>, "format": <fmt>}.
Raises TranscriptionError on any HTTP non-200, network failure, or
missing/misconfigured bearer key.
"""
if not settings.parakeet_key:
raise TranscriptionError(
"Parakeet not configured. Set TTS_PARAKEET_KEY in .env "
"(shared bearer key documented in ~/.claude/rules/gpu.md)."
)
if not path.exists():
raise TranscriptionError(f"Audio file not found: {path}")
if not path.is_file():
raise TranscriptionError(f"Audio path is not a regular file: {path}")
# Read the whole file into memory. Transcription audio is typically
# under 100 MB even for long meetings, so this is fine. If we ever
# need to handle multi-hour audio, switch to a streaming upload.
audio_bytes = path.read_bytes()
# Multipart form: file + model + response_format + any non-None
# forward-compat fields. `data` MUST be a dict (Mapping) — httpx 0.28
# silently routes non-Mapping `data` into `content=` which builds a
# SyncByteStream and crashes the async client. List values get emitted
# as repeated form fields by httpx's multipart encoder.
data: dict[str, str | list[str]] = {
"model": "parakeet",
"response_format": response_format,
}
if timestamp_granularities:
data["timestamp_granularities"] = list(timestamp_granularities)
if diarize:
data["diarize"] = "true"
if num_speakers is not None:
data["num_speakers"] = str(num_speakers)
if vad is not None:
data["vad"] = "true" if vad else "false"
if punctuation is not None:
data["punctuation"] = "true" if punctuation else "false"
if min_confidence is not None:
data["min_confidence"] = str(min_confidence)
files = {"file": (path.name, audio_bytes, "application/octet-stream")}
headers = {"Authorization": f"Bearer {settings.parakeet_key}"}
client = _get_client()
try:
resp = await client.post(
settings.parakeet_url,
files=files,
data=data,
headers=headers,
)
except httpx.ConnectError as e:
raise TranscriptionError(
f"Cannot reach Parakeet gateway at {settings.parakeet_url}: {e}. "
f"Check the URL and that this host's outbound IP is allowlisted."
)
except (httpx.ReadError, httpx.RemoteProtocolError) as e:
raise TranscriptionError(f"Connection lost mid-transcription: {e}")
except httpx.TimeoutException as e:
raise TranscriptionError(f"Parakeet request timed out: {e}")
if resp.status_code != 200:
raise TranscriptionError(
f"Parakeet returned HTTP {resp.status_code}: "
f"{resp.text[:500] or '(empty body)'}"
)
# Response format dispatch. json/verbose_json are JSON-shaped; text/srt/vtt
# are plain bodies that we wrap so the MCP tool always returns a dict.
if response_format in ("json", "verbose_json"):
return resp.json()
return {"text": resp.text, "format": response_format}