Initial commit of current state
This commit is contained in:
+119
@@ -0,0 +1,119 @@
|
||||
"""Text-to-speech using the macOS system voices.
|
||||
|
||||
Drives the `say` binary rather than AVSpeechSynthesizer. `say` can render
|
||||
straight to signed 16-bit little-endian PCM at a chosen sample rate, which is
|
||||
exactly what the pipeline wants, and it avoids reading raw AVAudioPCMBuffer
|
||||
channel pointers through pyobjc.
|
||||
|
||||
Note that Siri's voices are not reachable this way. Apple does not expose them
|
||||
to third-party apps; `say` and AVSpeechSynthesizer only see the voices listed
|
||||
under System Settings > Accessibility > Spoken Content.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import tempfile
|
||||
import wave
|
||||
from collections.abc import AsyncGenerator
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from pipecat.audio.utils import create_stream_resampler
|
||||
from pipecat.frames.frames import ErrorFrame, Frame, TTSAudioRawFrame
|
||||
from pipecat.services.settings import TTSSettings
|
||||
from pipecat.services.tts_service import TTSService
|
||||
|
||||
SAY = "/usr/bin/say"
|
||||
|
||||
# Chunked so playback can start before the whole file is read.
|
||||
_CHUNK_FRAMES = 2400 # 100ms at 24kHz
|
||||
|
||||
|
||||
def available_voices() -> list[tuple[str, str]]:
|
||||
"""Return (name, language) for every installed system voice."""
|
||||
import AVFoundation as AV
|
||||
|
||||
return [(v.name(), v.language()) for v in AV.AVSpeechSynthesisVoice.speechVoices()]
|
||||
|
||||
|
||||
def find_voice(name: str) -> tuple[str, str] | None:
|
||||
"""Look up an installed voice by name, case-insensitively."""
|
||||
for voice_name, language in available_voices():
|
||||
if voice_name.lower() == name.lower():
|
||||
return voice_name, language
|
||||
return None
|
||||
|
||||
|
||||
class AppleTTSService(TTSService):
|
||||
"""Speak text with a macOS system voice.
|
||||
|
||||
Args:
|
||||
voice: An installed system voice name, e.g. "Moira" for Irish English.
|
||||
rate_wpm: Speaking rate in words per minute. None uses the voice default.
|
||||
"""
|
||||
|
||||
def __init__(self, *, voice: str = "Moira", rate_wpm: int | None = None, **kwargs):
|
||||
super().__init__(
|
||||
push_start_frame=True,
|
||||
push_stop_frames=True,
|
||||
settings=TTSSettings(model=None, voice=voice, language=None),
|
||||
**kwargs,
|
||||
)
|
||||
self._voice = voice
|
||||
self._rate_wpm = rate_wpm
|
||||
self._resampler = create_stream_resampler()
|
||||
|
||||
if find_voice(voice) is None:
|
||||
names = ", ".join(sorted(n for n, _ in available_voices())[:8])
|
||||
raise ValueError(
|
||||
f"macOS has no voice named {voice!r}. Installed voices include: {names}… "
|
||||
"Add more under System Settings > Accessibility > Spoken Content > System Voice."
|
||||
)
|
||||
|
||||
def can_generate_metrics(self) -> bool:
|
||||
return True
|
||||
|
||||
async def run_tts(self, text: str, context_id: str) -> AsyncGenerator[Frame, None]:
|
||||
"""Synthesize one chunk of text."""
|
||||
await self.start_tts_usage_metrics(text)
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
|
||||
path = f.name
|
||||
try:
|
||||
command = [SAY, "-v", self._voice, "-o", path]
|
||||
if self._rate_wpm:
|
||||
command += ["-r", str(self._rate_wpm)]
|
||||
command += [f"--data-format=LEI16@{self.sample_rate}", "--", text]
|
||||
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
*command,
|
||||
stdout=asyncio.subprocess.DEVNULL,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
_, stderr = await process.communicate()
|
||||
if process.returncode != 0:
|
||||
detail = stderr.decode(errors="replace").strip()
|
||||
yield ErrorFrame(error=f"say failed ({process.returncode}): {detail}")
|
||||
return
|
||||
|
||||
with wave.open(path, "rb") as wav:
|
||||
source_rate = wav.getframerate()
|
||||
while chunk := wav.readframes(_CHUNK_FRAMES):
|
||||
await self.stop_ttfb_metrics()
|
||||
if source_rate != self.sample_rate:
|
||||
chunk = await self._resampler.resample(
|
||||
chunk, source_rate, self.sample_rate
|
||||
)
|
||||
yield TTSAudioRawFrame(
|
||||
audio=chunk,
|
||||
sample_rate=self.sample_rate,
|
||||
num_channels=1,
|
||||
context_id=context_id,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.exception(f"Apple TTS failed: {e}")
|
||||
yield ErrorFrame(error=f"Apple TTS failed: {e}")
|
||||
finally:
|
||||
await self.stop_ttfb_metrics()
|
||||
if os.path.exists(path):
|
||||
os.unlink(path)
|
||||
Reference in New Issue
Block a user