diff --git a/hermes_voice_gateway/README.md b/hermes_voice_gateway/README.md new file mode 100644 index 0000000..cfc8b58 --- /dev/null +++ b/hermes_voice_gateway/README.md @@ -0,0 +1,90 @@ +# Hermes ESP32 Voice Gateway + +This folder captures the Hermes API-server implementation used by the ESP32 screen voice demo. + +It is intentionally kept next to the screen firmware because the ESP32 client in `demo_hermes_voice.py` talks to this gateway route: + +`POST /api/esp32/voice` + +## Flow + +1. ESP32 records push-to-talk audio as 16 kHz mono 16-bit PCM WAV. +2. ESP32 posts the raw WAV bytes to Hermes with an Authorization header containing the configured API server token. +3. Hermes saves the upload to its audio cache and transcribes it with the configured Hermes STT provider. +4. Hermes sends the transcript through the normal agent session using a persistent conversation key: `esp32:`. +5. Hermes turns the final answer into speech with the configured Hermes TTS provider. +6. Hermes normalizes the TTS output into an ESP32-friendly WAV: mono, 16 kHz, signed 16-bit little-endian PCM. +7. Hermes returns the WAV bytes to the ESP32 for playback. + +## Files + +- `api_server_endpoint.py` — the route/mixin implementation for Hermes' `gateway/platforms/api_server.py`. +- `smoke_test.py` — local helper tests for MIME mapping, safe headers, WAV normalization, and fallback ACK audio generation. + +## Client request contract + +Required: + +- Method: `POST` +- Path: `/api/esp32/voice` +- Header: Authorization containing the configured API server token +- Header: `Content-Type: audio/wav` +- Header: `X-Device-ID: kitchen-button` or another stable device id +- Body: raw audio bytes, preferably 16 kHz mono 16-bit PCM WAV + +Optional headers: + +- `X-Hermes-Instructions` — extra system instructions for this turn. +- `X-Hermes-Reply-Mode: ack` — return a quick ACK WAV immediately, then finish the agent turn in the background. +- `X-Hermes-Screen-Url` — JSON-RPC MCP endpoint for the initiating screen, e.g. `http://192.168.68.123/api/mcp`. +- `X-Hermes-Screen-Device` — local device registry alias to resolve to a screen URL. + +Sync-mode response: + +- `Content-Type: audio/wav` +- WAV body playable by the ESP32 +- Diagnostic headers: + - `X-Hermes-Transcript` + - `X-Hermes-Text-Response` + - `X-Hermes-Response-Id` + - `X-Hermes-Conversation` + - `X-Hermes-TTS-Source-Type` + +ACK-mode response: + +- `Content-Type: audio/wav` +- short acknowledgement WAV immediately +- the full answer is displayed/spoken on the initiating MCP screen when a screen URL can be resolved + +## Hermes integration notes + +In Hermes Agent, wire this into `gateway/platforms/api_server.py` by adding: + +```python +self._app.router.add_post("/api/esp32/voice", self._handle_esp32_voice) +``` + +after the API server app is created. + +The mixin expects the API server class to already provide these Hermes internals: + +- `_check_auth(request)` +- `_run_agent(...)` +- `_build_response_conversation_history(...)` +- `_response_messages_turn_start_index(...)` +- `_extract_output_items(...)` +- `_response_store` +- `_model_name` +- `_background_tasks` + +Do not commit API keys or device tokens here. The ESP32 should use the live `API_SERVER_KEY` configured on the Hermes host. + +## Smoke test + +From this repo: + +```bash +python3 hermes_voice_gateway/smoke_test.py +``` + +If `ffmpeg` is installed, the smoke test also verifies WAV normalization through the same decode/rewrite path used for TTS output. diff --git a/hermes_voice_gateway/api_server_endpoint.py b/hermes_voice_gateway/api_server_endpoint.py new file mode 100644 index 0000000..e8dfe84 --- /dev/null +++ b/hermes_voice_gateway/api_server_endpoint.py @@ -0,0 +1,658 @@ +# pyright: reportMissingImports=false, reportAttributeAccessIssue=false, reportArgumentType=false +"""Hermes API-server ESP32 voice gateway implementation. + +This module is a self-contained reference implementation for the Hermes +`POST /api/esp32/voice` route used by the ESP32 screen voice demo. + +It is written as a mixin so the implementation can live beside this screen +repo while still documenting the exact methods that belong on Hermes' +`APIServerPlatform` class. The host class must provide the normal Hermes API +server internals listed in README.md. +""" + +from __future__ import annotations + +import asyncio +import json +import logging +import math +import re +import subprocess +import time +import uuid +import wave +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import urlparse + +try: # aiohttp is present in Hermes gateway runtime. + from aiohttp import web +except Exception: # pragma: no cover - allows helper smoke tests without aiohttp. + web = None # type: ignore[assignment] + +logger = logging.getLogger(__name__) + +ESP32_AUDIO_MAX_BYTES = 8_000_000 # Enough for push-to-talk WAV tests. + +ESP32_AUDIO_EXTENSIONS = { + "audio/wav": ".wav", + "audio/wave": ".wav", + "audio/x-wav": ".wav", + "audio/mpeg": ".mp3", + "audio/mp3": ".mp3", + "audio/ogg": ".ogg", + "audio/opus": ".ogg", + "audio/webm": ".webm", + "audio/mp4": ".m4a", + "audio/x-m4a": ".m4a", + "application/octet-stream": ".wav", +} + +AUDIO_RESPONSE_MIME_BY_SUFFIX = { + ".mp3": "audio/mpeg", + ".wav": "audio/wav", + ".ogg": "audio/ogg", + ".opus": "audio/ogg", + ".m4a": "audio/mp4", + ".mp4": "audio/mp4", + ".flac": "audio/flac", +} + + +def openai_error(message: str, *, code: str = "esp32_voice_error") -> Dict[str, Any]: + """Small OpenAI-shaped error body for this endpoint.""" + return {"error": {"message": message, "type": "invalid_request_error", "code": code}} + + +def audio_extension_for_content_type(content_type: str) -> str: + """Map inbound audio Content-Type to a safe cache-file suffix.""" + media_type = (content_type or "").split(";", 1)[0].strip().lower() + return ESP32_AUDIO_EXTENSIONS.get(media_type, ".wav") + + +def audio_response_mime(path: str) -> str: + """Return an HTTP Content-Type for a generated audio file path.""" + suffix = Path(path).suffix.lower() + return AUDIO_RESPONSE_MIME_BY_SUFFIX.get(suffix, "application/octet-stream") + + +def safe_audio_header(value: Any, max_length: int) -> str: + """Make diagnostic text safe for HTTP response headers.""" + text = str(value or "")[:max_length] + return text.replace("\r", " ").replace("\n", " ").replace("\x00", " ") + + +def convert_audio_to_esp32_wav(input_path: str, *, output_dir: Path, device_id: str) -> Path: + """Normalize generated TTS audio to ESP32-friendly WAV. + + Hermes TTS providers commonly emit MP3 or OGG. The ESP32 client expects a + simple RIFF/WAVE file, so decode with ffmpeg to raw PCM and rewrite the WAV + container with Python's `wave` module. The result is mono, 16 kHz, + signed 16-bit little-endian PCM with a minimal `fmt` + `data` layout. + """ + safe_device = re.sub(r"[^A-Za-z0-9_.-]+", "_", device_id)[:80] or "default" + output_dir.mkdir(parents=True, exist_ok=True) + raw_path = output_dir / f"{safe_device}_{int(time.time() * 1000)}_{uuid.uuid4().hex[:8]}.s16le" + wav_path = raw_path.with_suffix(".wav") + cmd = [ + "ffmpeg", + "-y", + "-hide_banner", + "-loglevel", + "error", + "-i", + input_path, + "-ac", + "1", + "-ar", + "16000", + "-f", + "s16le", + str(raw_path), + ] + try: + subprocess.run(cmd, check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE) + raw = raw_path.read_bytes() + with wave.open(str(wav_path), "wb") as wav: + wav.setnchannels(1) + wav.setsampwidth(2) + wav.setframerate(16000) + wav.writeframes(raw) + return wav_path + finally: + raw_path.unlink(missing_ok=True) + + +def make_ack_wav(path: Path, *, frequency_hz: int = 880, duration_s: float = 0.18) -> Path: + """Create a tiny valid mono 16 kHz WAV tone used as last-resort ACK audio.""" + sample_rate = 16000 + frames = bytearray() + for i in range(int(sample_rate * duration_s)): + value = int(9000 * math.sin(2 * math.pi * frequency_hz * (i / sample_rate))) + frames.extend(value.to_bytes(2, "little", signed=True)) + path.parent.mkdir(parents=True, exist_ok=True) + with wave.open(str(path), "wb") as wav: + wav.setnchannels(1) + wav.setsampwidth(2) + wav.setframerate(sample_rate) + wav.writeframes(bytes(frames)) + return path + + +class HermesESP32VoiceGatewayMixin: + """Mixin for Hermes' API-server adapter. + + The consuming class is expected to provide the Hermes-specific internals: + auth, agent execution, Responses API storage, and background task tracking. + """ + + @staticmethod + def _audio_extension_for_content_type(content_type: str) -> str: + return audio_extension_for_content_type(content_type) + + @staticmethod + def _audio_response_mime(path: str) -> str: + return audio_response_mime(path) + + @staticmethod + def _safe_audio_header(value: Any, max_length: int) -> str: + return safe_audio_header(value, max_length) + + @staticmethod + def _convert_audio_to_esp32_wav(input_path: str, *, output_dir: Path, device_id: str) -> Path: + return convert_audio_to_esp32_wav(input_path, output_dir=output_dir, device_id=device_id) + + @staticmethod + def _esp32_reply_mode(request: Any) -> str: + value = ( + request.headers.get("X-Hermes-Reply-Mode") + or request.query.get("reply_mode") + or request.query.get("mode") + or "sync" + ) + value = str(value).strip().lower() + if value in {"ack", "async", "background", "quick_ack", "quick-ack"}: + return "ack" + return "sync" + + @staticmethod + def _looks_like_http_url(value: str) -> bool: + try: + parsed = urlparse(value) + return parsed.scheme in {"http", "https"} and bool(parsed.netloc) + except Exception: + return False + + @staticmethod + def _load_local_device_url(device_name: str) -> Optional[str]: + """Best-effort local MCP screen lookup from ~/.hermes/local_devices.yaml.""" + if not device_name: + return None + try: + import yaml + + registry = Path.home() / ".hermes" / "local_devices.yaml" + data = yaml.safe_load(registry.read_text()) if registry.exists() else {} + aliases = data.get("aliases") if isinstance(data, dict) else None + devices = data.get("devices") if isinstance(data, dict) else None + names = [str(device_name).strip(), str(device_name).strip().replace("-", "_")] + seen: set[str] = set() + while names: + name = names.pop(0) + if not name or name in seen: + continue + seen.add(name) + if isinstance(aliases, dict): + alias = aliases.get(name) + if isinstance(alias, str) and alias: + if alias.startswith("http"): + return alias + names.append(alias) + info = devices.get(name) if isinstance(devices, dict) else None + if isinstance(info, dict): + return str(info.get("fallback_url") or info.get("url") or "") or None + except Exception: + return None + return None + + def _resolve_esp32_screen_url(self, request: Any, device_id: str) -> Optional[str]: + explicit = ( + request.headers.get("X-Hermes-Screen-Url") + or request.headers.get("X-Screen-MCP-Url") + or request.query.get("screen_url") + ) + if explicit and self._looks_like_http_url(explicit.strip()): + return explicit.strip() + screen_device = ( + request.headers.get("X-Hermes-Screen-Device") + or request.query.get("screen_device") + or device_id + ) + for candidate in [screen_device, "esp32_screen"]: + url = self._load_local_device_url(str(candidate).strip()) + if url and self._looks_like_http_url(url): + return url + return None + + async def _call_screen_mcp(self, screen_url: Optional[str], tool_name: str, arguments: Dict[str, Any]) -> bool: + """Call a simple HTTP JSON-RPC MCP screen tool directly.""" + if not screen_url: + return False + try: + from aiohttp import ClientSession + + payload = { + "jsonrpc": "2.0", + "id": uuid.uuid4().hex[:12], + "method": "tools/call", + "params": {"name": tool_name, "arguments": arguments}, + } + async with ClientSession() as session: + async with session.post(screen_url, json=payload, timeout=8) as resp: + if resp.status >= 400: + logger.warning("screen MCP %s failed: HTTP %s", tool_name, resp.status) + return False + body = await resp.text() + try: + parsed = json.loads(body) + if isinstance(parsed, dict) and parsed.get("error"): + logger.warning("screen MCP %s JSON-RPC error: %s", tool_name, parsed["error"]) + return False + except Exception: + pass + return True + except Exception as exc: + logger.warning("screen MCP %s failed: %s", tool_name, exc) + return False + + async def _send_screen_text(self, screen_url: Optional[str], text: str, *, clear: bool = False) -> None: + if not screen_url: + return + if clear: + await self._call_screen_mcp(screen_url, "clear_screen", {"color": 0}) + await self._call_screen_mcp(screen_url, "draw_text", {"x": 8, "y": 8, "text": text[:900]}) + + async def _send_screen_audio_response( + self, + screen_url: Optional[str], + text: str, + *, + audio_dir: Path, + device_id: str, + volume: int = 80, + ) -> bool: + """Best-effort spoken follow-up for ACK-mode ESP32 voice turns.""" + if not screen_url or not text.strip(): + return False + try: + import base64 + from tools.tts_tool import text_to_speech_tool + + tts_raw = await asyncio.to_thread(text_to_speech_tool, text) + tts_result = json.loads(tts_raw) + src = str(tts_result.get("file_path") or "") + if not tts_result.get("success") or not src or not Path(src).exists(): + logger.warning("screen audio TTS failed: %s", tts_result.get("error") or "missing TTS output") + return False + wav_path = await asyncio.to_thread( + self._convert_audio_to_esp32_wav, + src, + output_dir=audio_dir, + device_id=f"{device_id}_response", + ) + wav_base64 = base64.b64encode(wav_path.read_bytes()).decode("ascii") + return await self._call_screen_mcp( + screen_url, + "play_audio_base64", + {"wav_base64": wav_base64, "volume": volume}, + ) + except Exception: + logger.exception("failed to play ESP32 screen audio response") + return False + + @staticmethod + def _short_voice_summary(text: str, limit: int = 220) -> str: + clean = " ".join(str(text or "").split()) + if len(clean) <= limit: + return clean + return clean[: max(0, limit - 1)].rstrip() + "…" + + async def _esp32_ack_wav(self, *, audio_dir: Path, device_id: str) -> Path: + """Return a reusable quick acknowledgement WAV for low-latency voice UX.""" + ack_path = audio_dir / "quick_ack.wav" + if ack_path.exists(): + return ack_path + try: + from tools.tts_tool import text_to_speech_tool + + tts_raw = await asyncio.to_thread(text_to_speech_tool, "Got it. I’m working on that.") + tts_result = json.loads(tts_raw) + src = str(tts_result.get("file_path") or "") + if tts_result.get("success") and src and Path(src).exists(): + converted = self._convert_audio_to_esp32_wav(src, output_dir=audio_dir, device_id=f"{device_id}_ack") + converted.replace(ack_path) + return ack_path + except Exception: + logger.exception("failed to generate spoken ESP32 ack; falling back to tone") + return make_ack_wav(ack_path) + + def _append_esp32_voice_audit(self, audio_dir: Path, event: Dict[str, Any]) -> None: + try: + log_path = audio_dir / "voice_requests.jsonl" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(json.dumps(event, default=str) + "\n") + except Exception: + logger.debug("failed to append ESP32 voice audit log", exc_info=True) + + async def _run_esp32_voice_background( + self, + *, + transcript: str, + device_id: str, + instructions: Optional[str], + audio_dir: Path, + screen_url: Optional[str], + ) -> None: + """Run the slow ESP32 voice turn after the device already heard an ACK.""" + try: + await self._send_screen_text( + screen_url, + f"Heard: {self._short_voice_summary(transcript, 180)}\n\nWorking…", + clear=True, + ) + screen_hint = ( + "\n\nThis request came from an ESP32 voice/screen device. " + "Keep the spoken follow-up short. The gateway will display the full answer on the initiating screen when possible." + ) + merged_instructions = (instructions or "") + screen_hint + response_data, final_text = await self._run_esp32_voice_turn( + transcript=transcript, + device_id=device_id, + instructions=merged_instructions, + ) + if not final_text.strip(): + final_text = self._extract_response_text(response_data) or "I finished, but I do not have a text answer." + display_text = final_text + if len(display_text) > 1200: + display_text = self._short_voice_summary(display_text, 900) + "\n\nFull answer is long; WhatsApp delivery is the next wiring step." + await self._send_screen_text(screen_url, display_text, clear=True) + await self._send_screen_audio_response(screen_url, final_text, audio_dir=audio_dir, device_id=device_id) + except Exception: + logger.exception("ESP32 background voice turn failed") + await self._send_screen_text(screen_url, "Hermes hit an error while preparing the full answer.", clear=True) + + @staticmethod + def _extract_response_text(response_data: Dict[str, Any]) -> str: + chunks: List[str] = [] + for item in response_data.get("output") or []: + if not isinstance(item, dict) or item.get("type") != "message": + continue + for part in item.get("content") or []: + if isinstance(part, dict) and part.get("type") in {"output_text", "text"}: + text = part.get("text") + if isinstance(text, str) and text: + chunks.append(text) + return "\n".join(chunks).strip() + + async def _run_esp32_voice_turn( + self, + *, + transcript: str, + device_id: str, + instructions: Optional[str] = None, + ) -> Tuple[Dict[str, Any], str]: + """Run one persistent Responses-style turn for an ESP32 voice device.""" + conversation = f"esp32:{device_id}" + previous_response_id = self._response_store.get_conversation(conversation) + conversation_history: List[Dict[str, Any]] = [] + stored_session_id = None + stored_instructions = None + + if previous_response_id: + stored = self._response_store.get(previous_response_id) + if stored is not None: + conversation_history = list(stored.get("conversation_history", [])) + stored_session_id = stored.get("session_id") + stored_instructions = stored.get("instructions") + + if instructions is None: + instructions = stored_instructions + + session_id = stored_session_id or str(uuid.uuid4()) + user_message = f'[Voice input from ESP32 device "{device_id}"]\n{transcript}' + + result, usage = await self._run_agent( + user_message=user_message, + conversation_history=conversation_history, + ephemeral_system_prompt=instructions, + session_id=session_id, + gateway_session_key=conversation, + ) + + final_response = result.get("final_response", "") or result.get("error", "") or "(No response generated)" + response_id = f"resp_{uuid.uuid4().hex[:28]}" + created_at = int(time.time()) + full_history = self._build_response_conversation_history(conversation_history, user_message, result, final_response) + output_start_index = self._response_messages_turn_start_index(conversation_history, user_message, result) + output_items = self._extract_output_items(result, start_index=output_start_index) + response_data = { + "id": response_id, + "object": "response", + "status": "completed", + "created_at": created_at, + "model": self._model_name, + "output": output_items, + "usage": { + "input_tokens": usage.get("input_tokens", 0), + "output_tokens": usage.get("output_tokens", 0), + "total_tokens": usage.get("total_tokens", 0), + }, + } + self._response_store.put(response_id, { + "response": response_data, + "conversation_history": full_history, + "instructions": instructions, + "session_id": session_id, + }) + self._response_store.set_conversation(conversation, response_id) + return response_data, final_response + + async def _handle_esp32_voice(self, request: Any) -> Any: + """POST /api/esp32/voice — receive audio, run Hermes, return TTS audio.""" + if web is None: + raise RuntimeError("aiohttp is required for the ESP32 voice endpoint") + + auth_err = self._check_auth(request) + if auth_err: + return auth_err + + device_id = (request.headers.get("X-Device-ID") or request.query.get("device_id") or "default").strip()[:80] or "default" + try: + audio_bytes = await request.read() + except Exception as exc: + return web.json_response(openai_error(f"Failed to read audio body: {exc}"), status=400) + + if not audio_bytes: + return web.json_response(openai_error("Empty audio body", code="empty_audio"), status=400) + if len(audio_bytes) > ESP32_AUDIO_MAX_BYTES: + return web.json_response( + openai_error(f"Audio body is too large ({len(audio_bytes)} bytes); max is {ESP32_AUDIO_MAX_BYTES}", code="audio_too_large"), + status=413, + ) + + suffix = self._audio_extension_for_content_type(request.headers.get("Content-Type", "")) + try: + from hermes_constants import get_hermes_dir + + audio_dir = get_hermes_dir("cache/audio", "audio_cache") / "esp32" + except Exception: + audio_dir = Path.home() / ".hermes" / "cache" / "audio" / "esp32" + audio_dir.mkdir(parents=True, exist_ok=True) + safe_device = re.sub(r"[^A-Za-z0-9_.-]+", "_", device_id)[:80] or "default" + audio_path = audio_dir / f"{safe_device}_{int(time.time() * 1000)}_{uuid.uuid4().hex[:8]}{suffix}" + audio_path.write_bytes(audio_bytes) + + audit_event: Dict[str, Any] = { + "timestamp": time.time(), + "device_id": device_id, + "source_ip": getattr(request, "remote", "") or "", + "forwarded_for": request.headers.get("X-Forwarded-For", ""), + "real_ip": request.headers.get("X-Real-IP", ""), + "user_agent": request.headers.get("User-Agent", ""), + "reply_mode": self._esp32_reply_mode(request), + "screen_device": request.headers.get("X-Hermes-Screen-Device") or request.query.get("screen_device") or "", + "screen_url": request.headers.get("X-Hermes-Screen-Url") or request.headers.get("X-Screen-MCP-Url") or request.query.get("screen_url") or "", + "content_type": request.headers.get("Content-Type", ""), + "audio_bytes": len(audio_bytes), + "audio_path": str(audio_path), + "status": "received", + } + self._append_esp32_voice_audit(audio_dir, audit_event) + + if self._esp32_reply_mode(request) == "ack": + screen_url = self._resolve_esp32_screen_url(request, device_id) + + async def _ack_background() -> None: + try: + from tools.transcription_tools import transcribe_audio + + stt_result_bg = await asyncio.wait_for(asyncio.to_thread(transcribe_audio, str(audio_path)), timeout=45) + if not stt_result_bg.get("success"): + failed_event = dict(audit_event) + failed_event.update({"status": "transcription_failed", "error": stt_result_bg.get("error") or "Transcription failed", "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, failed_event) + await self._send_screen_text(screen_url, "I heard audio, but transcription failed. Try again.", clear=True) + return + transcript_bg = str(stt_result_bg.get("transcript") or "").strip() + if not transcript_bg: + empty_event = dict(audit_event) + empty_event.update({"status": "empty_transcript", "transcript": "", "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, empty_event) + await self._send_screen_text(screen_url, "I heard audio, but got no words. Try again.", clear=True) + return + transcript_event = dict(audit_event) + transcript_event.update({"status": "transcribed", "transcript": transcript_bg, "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, transcript_event) + await self._run_esp32_voice_background( + transcript=transcript_bg, + device_id=device_id, + instructions=request.headers.get("X-Hermes-Instructions") or None, + audio_dir=audio_dir, + screen_url=screen_url, + ) + except asyncio.TimeoutError: + timeout_event = dict(audit_event) + timeout_event.update({"status": "transcription_timeout", "error": "Transcription exceeded 45s", "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, timeout_event) + except Exception as exc: + logger.exception("ESP32 ack-mode background processing failed") + failed_event = dict(audit_event) + failed_event.update({"status": "background_exception", "error": str(exc), "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, failed_event) + + task = asyncio.create_task(_ack_background()) + if hasattr(self, "_background_tasks"): + self._background_tasks.add(task) + task.add_done_callback(self._background_tasks.discard) + ack_path = await self._esp32_ack_wav(audio_dir=audio_dir, device_id=device_id) + return web.FileResponse( + ack_path, + headers={ + "X-Hermes-Conversation": f"esp32:{device_id}", + "X-Hermes-Reply-Mode": "ack", + "X-Hermes-Screen-Url": screen_url or "", + "Content-Type": "audio/wav", + }, + ) + + try: + from tools.transcription_tools import transcribe_audio + + stt_result = await asyncio.to_thread(transcribe_audio, str(audio_path)) + except Exception as exc: + logger.exception("ESP32 transcription failed") + failed_event = dict(audit_event) + failed_event.update({"status": "transcription_exception", "error": str(exc), "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, failed_event) + return web.json_response(openai_error(f"Transcription failed: {exc}"), status=500) + + if not stt_result.get("success"): + failed_event = dict(audit_event) + failed_event.update({"status": "transcription_failed", "error": stt_result.get("error") or "Transcription failed", "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, failed_event) + return web.json_response(openai_error(stt_result.get("error") or "Transcription failed", code="transcription_failed"), status=502) + + transcript = str(stt_result.get("transcript") or "").strip() + if not transcript: + empty_event = dict(audit_event) + empty_event.update({"status": "empty_transcript", "transcript": "", "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, empty_event) + return web.json_response(openai_error("Transcription was empty", code="empty_transcript"), status=422) + + transcript_event = dict(audit_event) + transcript_event.update({"status": "transcribed", "transcript": transcript, "timestamp": time.time()}) + self._append_esp32_voice_audit(audio_dir, transcript_event) + + try: + response_data, final_text = await self._run_esp32_voice_turn( + transcript=transcript, + device_id=device_id, + instructions=request.headers.get("X-Hermes-Instructions") or None, + ) + except Exception as exc: + logger.exception("ESP32 agent turn failed") + return web.json_response(openai_error(f"Agent turn failed: {exc}"), status=500) + + if not final_text.strip(): + final_text = self._extract_response_text(response_data) or "I heard you, but I do not have a spoken answer." + + try: + from tools.tts_tool import text_to_speech_tool + + tts_raw = await asyncio.to_thread(text_to_speech_tool, final_text) + tts_result = json.loads(tts_raw) + except Exception as exc: + logger.exception("ESP32 TTS failed") + return web.json_response( + openai_error(f"TTS failed after response was generated: {exc}", code="tts_failed"), + status=500, + headers={"X-Hermes-Transcript": self._safe_audio_header(transcript, 1000), "X-Hermes-Text-Response": self._safe_audio_header(final_text, 2000)}, + ) + + if not tts_result.get("success"): + return web.json_response( + openai_error(tts_result.get("error") or "TTS failed", code="tts_failed"), + status=502, + headers={"X-Hermes-Transcript": self._safe_audio_header(transcript, 1000), "X-Hermes-Text-Response": self._safe_audio_header(final_text, 2000)}, + ) + + tts_path = str(tts_result.get("file_path") or "") + if not tts_path or not Path(tts_path).exists(): + return web.json_response(openai_error("TTS succeeded but no audio file was produced", code="tts_file_missing"), status=500) + + try: + response_audio_path = self._convert_audio_to_esp32_wav(tts_path, output_dir=audio_dir, device_id=device_id) + except Exception as exc: + logger.exception("ESP32 WAV normalization failed") + return web.json_response( + openai_error(f"Could not convert TTS audio to ESP32 WAV: {exc}", code="wav_conversion_failed"), + status=500, + headers={"X-Hermes-Transcript": self._safe_audio_header(transcript, 1000), "X-Hermes-Text-Response": self._safe_audio_header(final_text, 2000)}, + ) + + return web.FileResponse( + response_audio_path, + headers={ + "X-Hermes-Transcript": self._safe_audio_header(transcript, 1000), + "X-Hermes-Text-Response": self._safe_audio_header(final_text, 2000), + "X-Hermes-Response-Id": str(response_data.get("id") or ""), + "X-Hermes-Conversation": f"esp32:{device_id}", + "X-Hermes-TTS-Source-Type": self._audio_response_mime(tts_path), + "Content-Type": "audio/wav", + }, + ) + + +def register_esp32_voice_route(app: Any, api_server: HermesESP32VoiceGatewayMixin) -> None: + """Register the endpoint on an aiohttp app.""" + app.router.add_post("/api/esp32/voice", api_server._handle_esp32_voice) diff --git a/hermes_voice_gateway/smoke_test.py b/hermes_voice_gateway/smoke_test.py new file mode 100644 index 0000000..636cff5 --- /dev/null +++ b/hermes_voice_gateway/smoke_test.py @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +"""Smoke tests for hermes_voice_gateway helper code. + +These tests avoid live Hermes/STT/TTS calls. They verify the pieces that are +safe to exercise locally: content-type mapping, response MIME mapping, header +sanitization, fallback ACK WAV generation, and ffmpeg WAV normalization when +ffmpeg is available. +""" + +from __future__ import annotations + +import shutil +import tempfile +import wave +from pathlib import Path + +from api_server_endpoint import ( + audio_extension_for_content_type, + audio_response_mime, + convert_audio_to_esp32_wav, + make_ack_wav, + safe_audio_header, +) + + +def _assert_wav_16k_mono_s16(path: Path) -> None: + with wave.open(str(path), "rb") as wav: + assert wav.getnchannels() == 1, wav.getnchannels() + assert wav.getsampwidth() == 2, wav.getsampwidth() + assert wav.getframerate() == 16000, wav.getframerate() + assert wav.getnframes() > 0, wav.getnframes() + + +def main() -> None: + assert audio_extension_for_content_type("audio/wav") == ".wav" + assert audio_extension_for_content_type("audio/wav; charset=binary") == ".wav" + assert audio_extension_for_content_type("audio/webm") == ".webm" + assert audio_extension_for_content_type("application/unknown") == ".wav" + + assert audio_response_mime("reply.wav") == "audio/wav" + assert audio_response_mime("reply.mp3") == "audio/mpeg" + assert audio_response_mime("reply.bin") == "application/octet-stream" + + assert safe_audio_header("hello\nworld\r\x00!", 100) == "hello world !" + assert safe_audio_header("abcdef", 3) == "abc" + + with tempfile.TemporaryDirectory() as tmp: + tmpdir = Path(tmp) + ack = make_ack_wav(tmpdir / "ack.wav") + _assert_wav_16k_mono_s16(ack) + + if shutil.which("ffmpeg"): + converted = convert_audio_to_esp32_wav(str(ack), output_dir=tmpdir, device_id="kitchen/button") + _assert_wav_16k_mono_s16(converted) + assert "kitchen_button" in converted.name + else: + print("ffmpeg not found; skipped conversion path") + + print("hermes_voice_gateway smoke tests passed") + + +if __name__ == "__main__": + main()