# pyright: reportMissingImports=false, reportAttributeAccessIssue=false """Client firmware application for real-time WebSocket audio streaming to Hermes. Dynamically uses board_config to work across Hosyond and Waveshare RLCD boards. """ import time import struct import machine from machine import Pin, I2S import json import board_config from websocket_client import WebSocketClient # --- CONFIGURATION --- HERMES_WS_URL = "ws://192.168.68.126:8642/api/esp32/voice/ws" # Determine dynamic device ID based on board configuration DEVICE_ID = "esp32_screen" if board_config.BOARD_TYPE == 'WAVESHARE_RLCD' else "little32" # Placeholder for API Key. Since the key is stored on the Hermes server, # please copy-paste the token string here. HERMES_API_KEY = "mcT1YA1vOr9wXSiHpCYalweEGGZKX-PIfZv2drp8BSg" def main(): print("=== Starting Hermes WebSocket Voice Assistant Demo (Type: {}) ===".format(board_config.BOARD_TYPE)) # 1. Use pre-initialized display and touch from board_config display = board_config.display_instance touch = board_config.touch if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.show() # 2. Configure Audio Amp control pin based on board config amp_pin = None on_val = 0 off_val = 1 if board_config.audio_amp_pin is not None: on_val = 0 if board_config.audio_amp_active_level == 0 else 1 off_val = 1 if board_config.audio_amp_active_level == 0 else 0 amp_pin = Pin(board_config.audio_amp_pin, Pin.OUT, value=off_val) # Helper function to check if trigger is active (touch screen or button) def is_talk_trigger_active(): if touch: return touch.is_touched() elif hasattr(board_config, 'buttons') and board_config.buttons and board_config.buttons.key: return board_config.buttons.key.is_pressed() return False while True: if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text("Hold screen & ask", 50, 70, 1) display.text("a question...", 50, 90, 1) display.text("Status: Idle (WS)", 10, display.height - 20, 1) display.show() # Wait for touch or button trigger while not is_talk_trigger_active(): time.sleep_ms(30) print("Touch/Button detected! Connecting WebSocket...") if display: display.clear(0) display.text("Connecting...", 80, 80, 1) display.show() headers = { "Authorization": f"Bearer {HERMES_API_KEY}", "X-Device-ID": DEVICE_ID } ws = WebSocketClient(HERMES_WS_URL, headers=headers, timeout=30) try: ws.connect() # Send start event ws.send_text(json.dumps({ "event": "start", "device_id": DEVICE_ID, "sample_rate": 16000, "channels": 1, "sample_width": 2, "format": "pcm_s16le" })) # Read ready and listening events from server ws.recv_frame() # ready ws.recv_frame() # listening print("WebSocket connected and streaming started.") if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text("Listening...", 80, 80, 1) display.fill_rect(130, 110, 30, 30, 1) display.text("Status: Streaming", 10, display.height - 20, 1) display.show() # 3. Configure MCLK PWM mclk_pwm = None if board_config.audio_mclk_pin is not None: mclk_pin = Pin(board_config.audio_mclk_pin, Pin.OUT) mclk_pwm = machine.PWM(mclk_pin) mclk_pwm.freq(board_config.audio_mclk_freq) mclk_pwm.duty_u16(32768) # 4. Initialize microphone codec i2c = board_config.i2c_bus if board_config.audio_mic_codec == "ES7210": from audio_util import ES7210 codec = ES7210(i2c) codec.init(sample_rate=16000, bit_width=16) else: from audio_util import ES8311 codec = ES8311(i2c) if codec.init(sample_rate=16000): codec.set_volume(80) try: codec._write(0x14, 0x1A) codec._write(0x16, 0x01) codec._write(0x17, 0xC8) except: pass # 5. Open I2S RX for recording (Stereo 16kHz for ES7210, mono for ES8311) is_stereo = (board_config.audio_mic_codec == "ES7210") i2s_format = I2S.STEREO if is_stereo else I2S.MONO i2s_rx = I2S(1, sck=Pin(board_config.audio_i2s_sck), ws=Pin(board_config.audio_i2s_ws), sd=Pin(board_config.audio_i2s_rx_sd), mode=I2S.RX, ibuf=16000, rate=16000, bits=16, format=i2s_format) total_data_bytes = 0 buffer = bytearray(2048) mono_buf = bytearray(1024) rec_start_time = time.ticks_ms() max_rec_duration_ms = 10000 try: # Record loop - as long as touch/button is held while is_talk_trigger_active(): elapsed = time.ticks_diff(time.ticks_ms(), rec_start_time) if elapsed >= max_rec_duration_ms: print("Recording stopped: maximum duration reached") break bytes_read = i2s_rx.readinto(buffer) if bytes_read > 0: if is_stereo: # Stereo-to-mono: extract left channel (every other 16-bit sample) mono_len = bytes_read // 2 j = 0 for i in range(0, bytes_read, 4): mono_buf[j] = buffer[i] mono_buf[j + 1] = buffer[i + 1] j += 2 ws.send_binary(mono_buf[:mono_len]) total_data_bytes += mono_len else: ws.send_binary(buffer[:bytes_read]) total_data_bytes += bytes_read print(f"Touch/Button released! Sent {total_data_bytes} bytes.") except Exception as e: print("Error recording/streaming:", e) finally: i2s_rx.deinit() if board_config.audio_mic_codec == "ES8311": try: codec._write(0x16, 0x00) # Reset mic gain except: pass if total_data_bytes < 3200: print("Recording too short, cancelling session.") try: ws.send_text(json.dumps({"event": "cancel"})) ws.close() except: pass if mclk_pwm: mclk_pwm.deinit() continue # Send stop event ws.send_text(json.dumps({"event": "stop"})) if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text("Processing...", 50, 80, 1) display.text("Status: Thinking", 10, display.height - 20, 1) display.show() # Playback/Events loop i2s_tx = None received_audio_bytes = 0 speaker_write_failed = False while True: opcode, payload = ws.recv_frame() if opcode is None: break if opcode == 0x1: # Text frame (JSON event) try: event_data = json.loads(payload.decode('utf-8')) evt = event_data.get("event") if evt == "transcript": txt = event_data.get("text", "") print(f"Heard: {txt}") if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text("Heard:", 10, 40, 1) lines = [txt[i:i+30] for i in range(0, min(len(txt), 120), 30)] y_offset = 60 for line in lines: display.text(line, 10, y_offset, 1) y_offset += 20 display.show() elif evt == "thinking": pass elif evt == "response_text": txt = event_data.get("text", "") print(f"Response: {txt}") if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text("Response:", 10, 40, 1) lines = [txt[i:i+30] for i in range(0, min(len(txt), 120), 30)] y_offset = 60 for line in lines: display.text(line, 10, y_offset, 1) y_offset += 20 display.show() elif evt == "audio_start": print("Audio response started.") if amp_pin: amp_pin.value(on_val) # Enable amp # Initialize ES8311 Speaker DAC try: from audio_util import ES8311 dac = ES8311(i2c) dac.init(sample_rate=16000) dac.set_volume(85) except Exception as dace: print("Failed to initialize ES8311 DAC for playback:", dace) # Open I2S TX i2s_tx = I2S(1, sck=Pin(board_config.audio_i2s_sck), ws=Pin(board_config.audio_i2s_ws), sd=Pin(board_config.audio_i2s_tx_sd), mode=I2S.TX, ibuf=4096, rate=16000, bits=16, format=I2S.MONO) elif evt == "audio_end": print("Audio response ended.") if i2s_tx: time.sleep_ms(150) if amp_pin: amp_pin.value(off_val) # Disable amp i2s_tx.deinit() i2s_tx = None if display: display.clear(0) display.text("Hermes Assistant", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) if speaker_write_failed: display.text("Speaker write failed", 30, 80, 1) else: display.text(f"Recv: {received_audio_bytes} bytes", 30, 80, 1) display.text("Status: Idle (WS)", 10, display.height - 20, 1) display.show() elif evt == "done": break elif evt == "error": msg = event_data.get("message", "Unknown error") print(f"Error from server: {msg}") if display: display.clear(0) display.text("Error", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text(msg[:100], 10, 80, 1) display.show() time.sleep(3) break except Exception as e: print("Error parsing event text:", e) elif opcode == 0x2: # Binary frame (Audio WAV chunk) if i2s_tx: chunk = payload if chunk.startswith(b'RIFF') and len(chunk) > 44: chunk = chunk[44:] # Skip WAV header for direct play try: i2s_tx.write(chunk) received_audio_bytes += len(chunk) except Exception as e: speaker_write_failed = True print("Error writing to speaker:", e) ws.close() if mclk_pwm: mclk_pwm.deinit() except Exception as e: print("Failed to stream to Hermes server:", e) if display: display.clear(0) display.text("Server Error", 10, 10, 1) display.line(10, 22, display.width - 10, 22, 1) display.text("Could not connect", 50, 80, 1) display.text("to WebSocket server.", 50, 100, 1) display.show() time.sleep(3) try: ws.close() except Exception: pass if mclk_pwm: try: mclk_pwm.deinit() except: pass # Debounce touch release while is_talk_trigger_active(): time.sleep_ms(30) time.sleep_ms(300) if __name__ == "__main__": main()