#!/usr/bin/env python3 """ engine_apple_transcribe.py - v3 Apple Speech backend Drop-in replacement for engine_transcribe.py using Apple's new SpeechAnalyzer + SpeechTranscriber (macOS 26+). Architecture: - Uses same Silero VAD chunking as whisper engine (audio_buffer + silence threshold) - Each finalized chunk -> temp WAV 16k mono -> apple-speech-transcribe subprocess - Parses bench JSON output -> emits same dict shape as whisper engine This keeps the rest of the pipeline (LLM polish, translation, distribution) unchanged. Binary location: apple_speech/.build/release/apple-speech-transcribe Fallback: ../apple-speech-transcribe (project root copy) """ from __future__ import annotations import sys from unittest.mock import MagicMock try: import lzma except ImportError: mock_lzma = MagicMock() mock_lzma.FORMAT_XZ, mock_lzma.FORMAT_ALONE, mock_lzma.FORMAT_RAW = 1, 2, 3 mock_lzma.CHECK_NONE, mock_lzma.CHECK_CRC32, mock_lzma.CHECK_CRC64, mock_lzma.CHECK_SHA256 = 0, 1, 4, 10 sys.modules["_lzma"] = MagicMock() sys.modules["lzma"] = mock_lzma import time import os import argparse import json import subprocess import tempfile import shutil from pathlib import Path from typing import Optional APPLE_BIN_CANDIDATES = [ Path(__file__).parent / "apple_speech" / ".build" / "release" / "apple-speech-transcribe", Path(__file__).parent / "apple_speech" / "apple-speech-transcribe", Path(__file__).parent / "apple-speech-transcribe", Path("/tmp/apple-speech-transcribe"), ] LANG_TO_LOCALE = { "en": "en-US", "en-us": "en-US", "en-gb": "en-GB", "en-au": "en-AU", "es": "es-ES", "es-es": "es-ES", "es-mx": "es-MX", "es-us": "es-US", "es-cl": "es-CL", "fr": "fr-FR", "fr-fr": "fr-FR", "fr-ca": "fr-CA", "de": "de-DE", "de-de": "de-DE", "it": "it-IT", "ja": "ja-JP", "ko": "ko-KR", "pt": "pt-PT", "pt-br": "pt-BR", "zh": "zh-CN", "zh-cn": "zh-CN", "zh-tw": "zh-TW", "yue": "yue-CN", } APPLE_SUPPORTED_LANGS = {"en", "es", "fr", "de", "it", "ja", "ko", "pt", "zh", "yue"} def resolve_binary() -> Optional[Path]: for p in APPLE_BIN_CANDIDATES: if p.exists() and os.access(p, os.X_OK): return p return None def resolve_locale(lang: Optional[str], apple_locale_arg: Optional[str] = None) -> str: if apple_locale_arg: return apple_locale_arg if not lang: return "en-US" lang_lower = lang.lower().strip() if lang_lower in LANG_TO_LOCALE: return LANG_TO_LOCALE[lang_lower] # try prefix prefix = lang_lower.split("-")[0] if prefix in LANG_TO_LOCALE: return LANG_TO_LOCALE[prefix] # if lang itself looks like a locale (contains -) try as-is if "-" in lang and len(lang) >= 4: return lang return "en-US" def list_audio_devices(): try: import sounddevice as sd print("\nAvailable Audio Devices:") print(sd.query_devices()) except ImportError: print("[Error] sounddevice not installed. Cannot list devices.") def _ensure_soundfile(): try: import soundfile as sf return sf except ImportError: return None def transcribe_chunk_apple(audio_np, sample_rate, binary_path: Path, locale: str, verbose: bool = False) -> Optional[dict]: """ Transcribe a numpy audio chunk via apple-speech-transcribe --bench Returns parsed JSON dict or None. """ return _transcribe_chunk_wav(audio_np, sample_rate, binary_path, locale, verbose, mode="bench") def _wav_bytes_from_np(audio_np, sample_rate: int): import numpy as np, wave, io if audio_np.dtype != np.float32: if audio_np.dtype == np.int16: audio_float = audio_np.astype(np.float32) / 32768.0 else: audio_float = audio_np.astype(np.float32) else: audio_float = audio_np if audio_float.ndim > 1: audio_float = audio_float.mean(axis=1) audio_float = np.clip(audio_float, -1.0, 1.0) if sample_rate != 16000: duration = len(audio_float) / sample_rate new_len = int(duration * 16000) if new_len > 0: x_old = np.linspace(0, 1, len(audio_float)) x_new = np.linspace(0, 1, new_len) audio_float = np.interp(x_new, x_old, audio_float).astype(np.float32) sample_rate = 16000 int16_data = (audio_float * 32767).astype(np.int16) bio = io.BytesIO() with wave.open(bio, 'w') as wf: wf.setnchannels(1); wf.setsampwidth(2); wf.setframerate(sample_rate) wf.writeframes(int16_data.tobytes()) return bio.getvalue(), sample_rate def _transcribe_chunk_wav(audio_np, sample_rate, binary_path: Path, locale: str, verbose: bool, mode: str = "bench") -> Optional[dict]: import numpy as np tmp_path = None try: import wave, struct, io wav_bytes, sr = _wav_bytes_from_np(audio_np, sample_rate) fd, tmp_path = tempfile.mkstemp(suffix=".wav") os.close(fd) with open(tmp_path, "wb") as f: f.write(wav_bytes) if mode == "bench": cmd = [str(binary_path), "--bench", tmp_path, "--locale", locale] else: # include volatile to get drafts cmd = [str(binary_path), tmp_path, "--locale", locale, "--include-volatile"] result = subprocess.run(cmd, capture_output=True, text=True, timeout=8.0) if result.returncode != 0: if verbose: print(f"[Apple] Binary failed {result.returncode}: {result.stderr[:500]}") return None if mode == "bench": try: payload = json.loads(result.stdout) except json.JSONDecodeError as e: if verbose: print(f"[Apple] JSON parse failed: {e} stdout={result.stdout[:500]}") return None return payload else: # parse JSONL with volatile+final finals = [] volatiles = [] for line in result.stdout.splitlines(): line=line.strip() if not line: continue try: seg=json.loads(line) if seg.get("isFinal"): finals.append(seg) else: volatiles.append(seg) except: pass # build combined text from finals full_text = " ".join(s.get("text","") for s in finals).strip() full_text = " ".join(full_text.split()) return {"text": full_text, "segments": finals, "volatiles": volatiles, "audio_duration_sec": len(audio_np)/16000} except subprocess.TimeoutExpired: if verbose: print("[Apple] Transcription timed out") return None except Exception as e: if verbose: print(f"[Apple] Error: {e}") return None finally: if tmp_path and os.path.exists(tmp_path): try: os.unlink(tmp_path) except: pass def transcribe_chunk_apple_with_draft(audio_np, sample_rate, binary_path: Path, locale: str, verbose: bool = False) -> Optional[dict]: return _transcribe_chunk_wav(audio_np, sample_rate, binary_path, locale, verbose, mode="draft") class ApplePipeTranscriber: """ Keeps one Swift binary process alive in --pipe mode. Write wav chunk -> reads draft/final JSON lines with ultra-low overhead (no subprocess spawn per chunk). Falls back to bench if pipe not available. """ def __init__(self, binary_path: Path, locale: str, verbose=False): self.binary_path = binary_path self.locale = locale self.verbose = verbose self.proc = None self.lock = None self._start() def _start(self): import threading self.lock = threading.Lock() cmd = [str(self.binary_path), "--pipe", "--locale", self.locale] if self.verbose: cmd.append("-v") try: self.proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=False, bufsize=0) # check alive after 1s import time time.sleep(0.2) if self.proc.poll() is not None: err = self.proc.stderr.read().decode(errors="ignore") if self.proc.stderr else "" if self.verbose: print(f"[ApplePipe] Failed to start, exit={self.proc.returncode} {err[:500]}") self.proc=None else: if self.verbose: print(f"[ApplePipe] Started pid={self.proc.pid} locale={self.locale}") except Exception as e: if self.verbose: print(f"[ApplePipe] Start error: {e}") self.proc=None def transcribe(self, audio_np, sample_rate) -> Optional[dict]: if not self.proc or self.proc.poll() is not None: self._start() if not self.proc: return None try: wav_bytes, _ = _wav_bytes_from_np(audio_np, sample_rate) import struct header = struct.pack(">I", len(wav_bytes)) with self.lock: self.proc.stdin.write(header + wav_bytes) self.proc.stdin.flush() import time, select finals=[]; volatiles=[] start=time.time() while True: remaining = 6.0 - (time.time()-start) if remaining<=0: break rready,_,_ = select.select([self.proc.stdout], [], [], min(remaining, 0.2)) if not rready: if finals: break continue line = self.proc.stdout.readline() if not line: break try: txt=line.decode('utf-8', errors='ignore').strip() if not txt: continue obj=json.loads(txt) t=obj.get("text","").strip() if not t: continue if obj.get("isFinal") or obj.get("event")=="final": finals.append(obj) stall=time.time()+0.35 while time.time() Optional[dict]: """ Streaming version: calls draft_callback(text) for each volatile update (word-by-word), then returns final dict. Use this for ultra-low latency draft captions. """ if not self.proc or self.proc.poll() is not None: self._start() if not self.proc: return None try: wav_bytes, _ = _wav_bytes_from_np(audio_np, sample_rate) import struct, time, select header = struct.pack(">I", len(wav_bytes)) with self.lock: self.proc.stdin.write(header + wav_bytes) self.proc.stdin.flush() finals=[]; volatiles=[] start=time.time() last_draft="" while True: remaining = 6.0 - (time.time()-start) if remaining<=0: break rr,_,_ = select.select([self.proc.stdout], [], [], min(remaining, 0.15)) if not rr: if finals: break continue line=self.proc.stdout.readline() if not line: break try: obj=json.loads(line.decode('utf-8',errors='ignore').strip()) t=obj.get("text","").strip() if not t: continue if obj.get("isFinal") or obj.get("event")=="final": finals.append(obj) stall=time.time()+0.35 while time.time()I", 0)) self.proc.stdin.flush() self.proc.stdin.close() except: pass self.proc.wait(timeout=3) except: try: self.proc.kill() except: pass self.proc=None def run_transcription(out_queue, args): """ Main entry compatible with multiprocessing.Process(target=run_transcription) Same signature as engine_transcribe.py """ import numpy as np import sounddevice as sd import torch from silero_vad import load_silero_vad, get_speech_timestamps import queue SAMPLERATE, BLOCK_SIZE, VAD_THRESHOLD = 16000, 512, 0.5 binary_path = resolve_binary() if binary_path is None: print(f"[AppleTranscribe] ERROR: binary not found. Searched: {APPLE_BIN_CANDIDATES}") print("[AppleTranscribe] Run: cd apple_speech && bash build.sh") return apple_locale = resolve_locale(getattr(args, "lang", None), getattr(args, "apple_locale", None)) verbose = getattr(args, "verbose", False) or getattr(args, "post_correct_debug", False) print(f"[AppleTranscribe] Using binary: {binary_path}") print(f"[AppleTranscribe] Locale: {apple_locale} (from lang={getattr(args,'lang',None)})") print(f"[AppleTranscribe] VAD model loading...") vad_model = load_silero_vad() # Quick health check try: chk = subprocess.run([str(binary_path), "--check"], capture_output=True, text=True, timeout=5) if chk.returncode != 0: print(f"[AppleTranscribe] --check failed: {chk.stderr}") else: # parse last line print(f"[AppleTranscribe] Binary check OK") if verbose: for line in chk.stdout.splitlines(): print(f" {line}") except Exception as e: print(f"[AppleTranscribe] Binary check error: {e}") # Optional: preload by transcribing 0.1s silence to warm model try: silence = np.zeros(int(SAMPLERATE * 0.1), dtype=np.float32) _ = transcribe_chunk_apple(silence, SAMPLERATE, binary_path, apple_locale, verbose=False) print(f"[AppleTranscribe] Warmup done") except: pass audio_queue = queue.Queue() def audio_callback(indata, frames, time_info, status): if status: print(status, file=sys.stderr) audio_queue.put(indata.copy()) audio_buffer, speech_started = [], False last_stream_time = time.time() last_draft_text = "" # Feature flags use_stream = getattr(args, "stream", False) or getattr(args, "apple_stream", True) # default ON for Apple (cheap) stream_interval = getattr(args, "apple_stream_interval", 1.0) # seconds between draft transcribes use_pipe = getattr(args, "apple_pipe", True) # keep one Swift process alive (faster than spawn per chunk) pipe_transcriber = None if use_pipe: try: pipe_transcriber = ApplePipeTranscriber(binary_path, apple_locale, verbose=verbose) if pipe_transcriber.proc is None: pipe_transcriber = None print("[AppleTranscribe] Pipe mode unavailable, falling back to subprocess per chunk") except Exception as e: if verbose: print(f"[AppleTranscribe] Pipe init failed: {e}") pipe_transcriber = None device = "mps" if torch.backends.mps.is_available() else "cpu" def do_transcribe_final(audio_np): if pipe_transcriber and pipe_transcriber.proc and pipe_transcriber.proc.poll() is None: return pipe_transcriber.transcribe(audio_np, SAMPLERATE) return transcribe_chunk_apple(audio_np, SAMPLERATE, binary_path, apple_locale, verbose=verbose) def do_transcribe_with_streaming_drafts(audio_np, on_draft): """Use pipe's volatile streaming for real-time drafts (60ms granularity).""" if pipe_transcriber and pipe_transcriber.proc and pipe_transcriber.proc.poll() is None: return pipe_transcriber.transcribe_with_draft_callback(audio_np, SAMPLERATE, draft_callback=on_draft) # fallback: no streaming, just final return do_transcribe_final(audio_np) try: with sd.InputStream(samplerate=SAMPLERATE, channels=getattr(args, "channels", 1), callback=audio_callback, blocksize=BLOCK_SIZE, device=getattr(args, "device", None)): print(f"[AppleTranscribe] Listening (silence={getattr(args,'silence',1000)}ms max_buffer={getattr(args,'max_buffer',20)}s stream={use_stream} pipe={pipe_transcriber is not None})...") while True: while not audio_queue.empty(): data = audio_queue.get() if data is not None and data.size > 0: if getattr(args, "channels", 1) > 1: data = np.mean(data, axis=1) audio_buffer.append(data.flatten()) if audio_buffer: current_audio = np.concatenate(audio_buffer) audio_tensor = torch.from_numpy(current_audio) speech_timestamps = get_speech_timestamps( audio_tensor, vad_model, sampling_rate=SAMPLERATE, threshold=VAD_THRESHOLD, min_silence_duration_ms=getattr(args, "silence", 1000) ) if speech_timestamps: speech_started = True # --- draft streaming (non-blocking poll every stream_interval) --- if use_stream and (time.time() - last_stream_time) > stream_interval: if len(current_audio) > SAMPLERATE * 0.6: def _emit_draft(txt): nonlocal last_draft_text t = " ".join(txt.split()).strip() if t and t != last_draft_text and len(t) > 1: out_queue.put({"draft": t, "ts": time.time(), "isApple": True}) if verbose: print(f"\r[DRAFT]: {t} ", end="", flush=True) last_draft_text = t # quick draft via pipe streaming - but we are inside input stream callback thread? # Use quick final call with draft callback if pipe else interim transcribe try: # For low-latency drafts while speaking, transcribe current buffer # with incremental volatile updates showing word-by-word snap = current_audio.copy() # avoid racing with audio_buffer appending # If pipe not available, fall back to fast volatile parse if pipe_transcriber and pipe_transcriber.proc: # Use a separate thread to not block VAD loop too long? But okay # We'll call blocking but it returns within 200ms _ = pipe_transcriber.transcribe_with_draft_callback( snap, SAMPLERATE, draft_callback=_emit_draft ) else: dr = transcribe_chunk_apple_with_draft(snap, SAMPLERATE, binary_path, apple_locale, verbose=False) if dr: txt = dr.get("text", "").strip() if txt: _emit_draft(txt) except Exception as e: if verbose: print(f"[Apple DRAFT err] {e}") last_stream_time = time.time() silence_frames = len(current_audio) - speech_timestamps[-1]['end'] should_flush = (silence_frames > (SAMPLERATE * getattr(args, "silence", 1000) / 1000)) or \ len(current_audio) > (SAMPLERATE * getattr(args, "max_buffer", 20)) if should_flush: # Cancel any pending draft display and transcribe final # Use streaming final for best latency + true partials t0 = time.time() # Use pipe final (fast) without draft callback for this flush result = do_transcribe_final(current_audio) dt = time.time() - t0 if result is None: print(f"[AppleTranscribe] Chunk failed (len={len(current_audio)/SAMPLERATE:.2f}s)") audio_buffer, speech_started = [], False last_draft_text="" last_stream_time = time.time() continue text = result.get("text", "").strip() text = " ".join(text.split()) if not text: audio_buffer, speech_started = [], False last_draft_text="" last_stream_time = time.time() continue detected_lang = getattr(args, "lang", None) or apple_locale.split("-")[0] audio_dur = result.get("audio_duration_sec", len(current_audio)/SAMPLERATE) rtf = result.get("rtf", result.get("processing_sec", 0) and (audio_dur / max(result.get("processing_sec",0.001),0.001)) or 0) if use_stream: print(flush=True) out_queue.put({ "original": text, "en_bridge": text, "detected_lang": detected_lang, "speaker": None, "ts": time.time(), "_apple_meta": { "locale": apple_locale, "rtf": rtf, "processing_sec": result.get("processing_sec"), "segments": result.get("segments", []), "pipe": result.get("pipe", False), } }) print(f"[AppleTranscribe] {detected_lang.upper()} [{audio_dur:.1f}s-> {dt*1000:.0f}ms RTF={rtf:.1f}x]: {text}") audio_buffer, speech_started = [], False last_draft_text="" last_stream_time = time.time() elif not speech_started and len(current_audio) > SAMPLERATE * 2: audio_buffer = [] last_draft_text="" except Exception as e: print(f"[AppleTranscribe] Error: {e}") import traceback traceback.print_exc() finally: if pipe_transcriber: pipe_transcriber.close() # For --file mode testing standalone (non-live) def transcribe_file(file_path: str, locale: str = "en-US", include_volatile: bool = False) -> Optional[dict]: binary_path = resolve_binary() if not binary_path: raise FileNotFoundError(f"Apple speech binary not found. Candidates: {APPLE_BIN_CANDIDATES}") # For file mode we can call binary directly with JSONL or bench cmd = [str(binary_path), file_path, "--locale", locale] if include_volatile: cmd.append("--include-volatile") result = subprocess.run(cmd, capture_output=True, text=True, timeout=20) if result.returncode != 0: print(f"Failed: {result.stderr}") return None # JSONL lines segments = [] for line in result.stdout.splitlines(): line=line.strip() if not line: continue try: seg = json.loads(line) segments.append(seg) except: pass full_text = " ".join(s.get("text","") for s in segments if s.get("isFinal")).strip() full_text = " ".join(full_text.split()) return {"text": full_text, "segments": segments} if __name__ == "__main__": import argparse parser = argparse.ArgumentParser() parser.add_argument("file", nargs="?", help="Audio file to transcribe") parser.add_argument("--locale", default="en-US") parser.add_argument("--bench", action="store_true") parser.add_argument("--include-volatile", action="store_true") parser.add_argument("--verbose", "-v", action="store_true") args = parser.parse_args() binary_path = resolve_binary() if not binary_path: print("Binary not found") sys.exit(1) if args.file: if args.bench: cmd = [str(binary_path), "--bench", args.file, "--locale", args.locale] if args.verbose: cmd.append("-v") subprocess.run(cmd) else: r = transcribe_file(args.file, args.locale, args.include_volatile) print(json.dumps(r, indent=2, ensure_ascii=False)) else: # live test (just list devices) list_audio_devices()