"""Speech-to-text through macOS 26's SpeechAnalyzer, via a Swift helper. `SpeechTranscriber` is the accurate recogniser on this OS, and it is Swift-only — built on actors and AsyncSequence, with no Objective-C surface — so pyobjc cannot reach it. `swift/speech-helper` is a small binary that does, and this drives it as a subprocess: one process per utterance, JSON on stdout. Two modules sit behind the analyzer and the choice between them is not obvious, so it was measured on eight sentences of technical speech: | module | WER | |-----------------------------------|-------| | SpeechTranscriber | 18.8% | | SpeechTranscriber + vocabulary | 18.8% | | DictationTranscriber | 25.9% | | DictationTranscriber + vocabulary | 17.6% | `SpeechTranscriber` has the better acoustic model but **ignores** `AnalysisContext.contextualStrings` — output is byte-identical with and without terms. `DictationTranscriber` is weaker bare yet consumes them, and biasing matters more than the model on jargon, so it is the default here. Add repair rules on top and it reaches 12.9%, the best measured configuration. Both avoid the session limit that forces the transcript stitching in `apple_stt.py`: 40 of 40 sentences recovered from 83 seconds of audio, with no reassembly. The subprocess costs a few tens of milliseconds per utterance. Against a recogniser that runs far faster than real time, that is a fair trade. """ import asyncio import json import os import subprocess import sys import tempfile from collections.abc import AsyncGenerator from pathlib import Path from loguru import logger from pipecat.frames.frames import ErrorFrame, Frame, TranscriptionFrame from pipecat.services.settings import STTSettings from pipecat.services.stt_service import SegmentedSTTService from pipecat.transcriptions.language import Language from pipecat.utils.time import time_now_iso8601 def get_helper_path(binary_name: str) -> Path: if getattr(sys, "frozen", False) and hasattr(sys, "_MEIPASS"): p1 = Path(sys._MEIPASS) / binary_name if p1.exists(): return p1 p2 = Path(sys._MEIPASS) / "swift" / binary_name if p2.exists(): return p2 if getattr(sys, "frozen", False): res_dir = Path(sys.executable).parent.parent / "Resources" p1 = res_dir / binary_name if p1.exists(): return p1 p2 = res_dir / "swift" / binary_name if p2.exists(): return p2 return Path(__file__).parent / "swift" / binary_name HELPER = get_helper_path("speech-helper") # Generous: the helper may be downloading the on-device model on first use. _FIRST_RUN_TIMEOUT = 300.0 _TIMEOUT_BASE = 15.0 _TIMEOUT_PER_AUDIO_SECOND = 0.5 # A locally built binary that has not been through this Mac's approval process # is SIGKILLed on launch rather than failing in any legible way. _KILLED = -9 def _needs_approval(returncode: int) -> bool: return returncode in (_KILLED, 137) async def _run_helper(args: list[str], timeout: float) -> tuple[int, str, str]: process = await asyncio.create_subprocess_exec( str(HELPER), *args, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE, ) try: stdout, stderr = await asyncio.wait_for(process.communicate(), timeout) except asyncio.TimeoutError: process.kill() raise return process.returncode, stdout.decode(errors="replace"), stderr.decode(errors="replace") def probe(locale: str = "en-US") -> tuple[bool, str]: """Check whether the helper exists, is allowed to run, and has its model. Blocking on purpose: this runs while the pipeline is being assembled, which may already be inside an event loop, so it cannot start one of its own. """ if not HELPER.exists(): return False, f"helper not built — run {HELPER.parent}/build.sh" try: result = subprocess.run( [str(HELPER), "--check", "--locale", locale], capture_output=True, text=True, timeout=_FIRST_RUN_TIMEOUT, ) except subprocess.TimeoutExpired: return False, "helper timed out during its availability check" except OSError as e: return False, str(e) returncode, stdout, stderr = result.returncode, result.stdout, result.stderr if _needs_approval(returncode): return False, ( "the helper was killed on launch, which on this managed Mac means the " f"binary is still awaiting approval. Request approval for {HELPER}, " "then try again." ) if returncode != 0: return False, (stderr.strip() or stdout.strip() or f"helper exited {returncode}") try: payload = json.loads(stdout) except json.JSONDecodeError: return False, f"unreadable helper output: {stdout[:120]!r}" if not payload.get("available"): return False, f"no SpeechTranscriber model for {locale}" return True, f"SpeechAnalyzer ready, assets {payload.get('assets', 'unknown')}" class SpeechAnalyzerSTTService(SegmentedSTTService): """Transcribe VAD-delimited segments with SpeechTranscriber.""" def __init__( self, *, locale: str = "en-US", language: Language = Language.EN_US, vocabulary=None, want_alternatives: bool = False, module: str = "dictation", **kwargs, ): super().__init__(settings=STTSettings(model=None, language=locale), **kwargs) self._locale = locale self._language = language self._vocabulary = vocabulary self._want_alternatives = want_alternatives # "dictation" consumes the vocabulary; "transcriber" has the stronger # acoustic model but ignores it. See the module docstring. self._module = module def can_generate_metrics(self) -> bool: return True async def run_stt(self, audio: bytes) -> AsyncGenerator[Frame, None]: """Transcribe one speech segment. Args: audio: The segment as a WAV container, per ``wants_wav_segments``. """ await self.start_processing_metrics() with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f: f.write(audio) path = f.name audio_seconds = max(0.0, (len(audio) - 44) / 2 / self.sample_rate) args = [path, "--locale", self._locale] if self._vocabulary and (terms := self._vocabulary.terms()): args += ["--terms", ",".join(terms)] if self._want_alternatives: args.append("--alternatives") if self._module == "dictation": args.append("--dictation") try: returncode, stdout, stderr = await _run_helper( args, _TIMEOUT_BASE + _TIMEOUT_PER_AUDIO_SECOND * audio_seconds ) except asyncio.TimeoutError: await self.stop_processing_metrics() yield ErrorFrame(error="SpeechAnalyzer helper timed out") return except Exception as e: await self.stop_processing_metrics() yield ErrorFrame(error=f"SpeechAnalyzer helper failed: {e}") return finally: os.unlink(path) await self.stop_processing_metrics() if returncode != 0: detail = stderr.strip() or stdout.strip() or f"exited {returncode}" yield ErrorFrame(error=f"SpeechAnalyzer helper failed: {detail}") return try: payload = json.loads(stdout) except json.JSONDecodeError: yield ErrorFrame(error=f"unreadable helper output: {stdout[:120]!r}") return if "error" in payload: yield ErrorFrame(error=f"SpeechAnalyzer: {payload['error']}") return text = (payload.get("text") or "").strip() if not text: return logger.debug(f"Transcription: [{text}]") yield TranscriptionFrame(text, self._user_id, time_now_iso8601(), self._language)