"""Audio helpers for the Pi client. This module is split into two layers: * Deterministic helpers (no hardware) — `resample_24k_to_16k`. Tested. * `AudioStreams` — owns the persistent PortAudio InputStream / OutputStream. Smoke-tested by `main.py` running on the Pi. """ import queue import sys from dataclasses import dataclass, field from typing import Optional import numpy as np from scipy.signal import resample_poly from client.log import get_logger _log = get_logger("audio") SAMPLE_RATE = 24_000 FRAME_SAMPLES = 1920 # 80 ms at 24 kHz CHANNELS = 1 DTYPE = "int16" BYTES_PER_FRAME = FRAME_SAMPLES * 2 # int16 -> 2 bytes def resample_24k_to_16k(frame: np.ndarray) -> np.ndarray: """Downsample a 24 kHz int16 mono frame to 16 kHz int16 mono (2:3 poly). Input length is preserved as `len(frame) * 2 // 3`. For the 1920-sample frames produced by our 80 ms blocksize at 24 kHz, this yields 1280 samples (80 ms at 16 kHz), which is exactly what openwakeword expects. """ if frame.dtype != np.int16: raise TypeError(f"expected int16, got {frame.dtype}") if frame.size == 0: raise ValueError("empty input") floats = frame.astype(np.float32) resampled = resample_poly(floats, up=2, down=3) return np.clip(resampled, -32768, 32767).astype(np.int16) def find_usb_device(sd) -> int: """Return the PortAudio device index of the USB Speaker Phone. `sd` is the imported `sounddevice` module — passed in so the deferred import of sounddevice (which requires PA_ALSA_PLUGHW already set in env) stays in `main.py`. """ devices = sd.query_devices() for idx, dev in enumerate(devices): name = dev["name"].lower() if "usb" in name and dev["max_input_channels"] >= 1: return idx print("USB audio device not found. Devices:", file=sys.stderr) print(devices, file=sys.stderr) raise SystemExit("no USB audio device found") @dataclass class AudioStreams: """Wrapper around persistent 24 kHz/mono/int16 InputStream + OutputStream. The InputStream is callback-driven; each 1920-sample frame is pushed onto `mic_queue` (a `queue.Queue[bytes]`). The bridge thread drains it. The OutputStream is opened but writes are driven by `PlaybackWorker`. """ sd: object # sounddevice module device_index: int mic_queue: queue.Queue overflow_counter: int = 0 _in_stream: Optional[object] = field(default=None, repr=False) _out_stream: Optional[object] = field(default=None, repr=False) @classmethod def open(cls, sd, mic_queue: queue.Queue) -> "AudioStreams": device = find_usb_device(sd) _log.info("opening audio on device %d: %r", device, sd.query_devices(device)["name"]) streams = cls(sd=sd, device_index=device, mic_queue=mic_queue) def _on_input(indata, frames, time_info, status): # PortAudio thread if status: streams.overflow_counter += 1 _log.warning("input status: %s", status) mono = indata[:, 0].tobytes() try: mic_queue.put_nowait(mono) except queue.Full: try: mic_queue.get_nowait() except queue.Empty: pass try: mic_queue.put_nowait(mono) except queue.Full: pass _log.warning("mic_queue full; dropped oldest frame") streams._in_stream = sd.InputStream( samplerate=SAMPLE_RATE, channels=CHANNELS, dtype=DTYPE, device=device, blocksize=FRAME_SAMPLES, callback=_on_input, ) streams._out_stream = sd.OutputStream( samplerate=SAMPLE_RATE, channels=CHANNELS, dtype=DTYPE, device=device, blocksize=FRAME_SAMPLES, ) return streams def start(self) -> None: self._in_stream.start() self._out_stream.start() def write(self, pcm16_bytes: bytes) -> None: """Blocking write to the OutputStream (from the playback worker thread).""" if not pcm16_bytes: return buf = np.frombuffer(pcm16_bytes, dtype=np.int16) self._out_stream.write(buf) def close(self) -> None: for s in (self._in_stream, self._out_stream): if s is not None: try: s.stop() except Exception: pass try: s.close() except Exception: pass