"""Mic capture + simple energy-based VAD utterance recording. Ported from desk_client/bolt_desk.py's record_utterance() — same tuning knobs, same behavior. Kept independent of any UI/threading model so it can be unit tested by feeding it a fake "stream" object. """ from __future__ import annotations import io import wave from typing import Optional, Protocol import numpy as np from .. import config class AudioStream(Protocol): """Minimal shape of the object record_utterance() needs — matches sounddevice.InputStream's .read(frames) -> (data, overflowed).""" def read(self, frames: int): ... def pcm_to_wav_bytes(pcm: np.ndarray, sample_rate: int = config.SAMPLE_RATE) -> bytes: buf = io.BytesIO() with wave.open(buf, "wb") as wf: wf.setnchannels(1) wf.setsampwidth(2) wf.setframerate(sample_rate) wf.writeframes(pcm.tobytes()) return buf.getvalue() def rms(frame: np.ndarray) -> float: return float(np.sqrt(np.mean(frame.astype(np.float64) ** 2))) def record_utterance( stream: AudioStream, should_continue=lambda: True, rms_threshold: int = None, silence_end_sec: float = None, max_utterance_s: float = None, min_utterance_s: float = None, grace_s: float = None, frame_len: int = config.FRAME_LEN, sample_rate: int = config.SAMPLE_RATE, ) -> Optional[np.ndarray]: """Capture one utterance from *stream*: wait for speech to start, stop after trailing silence. Returns None if nothing usable was heard. *should_continue* is polled each frame so a caller can cancel recording (e.g. the pet window was closed) without needing threading primitives baked into this function. *grace_s* is how long to wait for speech to *begin* before giving up. The controller stretches it for follow-up questions, where you're being asked something and need a moment to think rather than having just said the wake word on purpose. """ rms_threshold = config.RMS_THRESHOLD if rms_threshold is None else rms_threshold silence_end_sec = config.SILENCE_END_SEC if silence_end_sec is None else silence_end_sec max_utterance_s = config.MAX_UTTERANCE_S if max_utterance_s is None else max_utterance_s min_utterance_s = config.MIN_UTTERANCE_S if min_utterance_s is None else min_utterance_s grace_s = config.GRACE_SECONDS if grace_s is None else grace_s frames: list[np.ndarray] = [] started = False silence_frames = 0 silence_limit = int(silence_end_sec * sample_rate / frame_len) max_frames = int(max_utterance_s * sample_rate / frame_len) grace_frames = int(grace_s * sample_rate / frame_len) # how long to wait for speech to begin waited = 0 while should_continue(): chunk, _ = stream.read(frame_len) frame = np.asarray(chunk)[:, 0].copy() frame_rms = rms(frame) if not started: waited += 1 if frame_rms >= rms_threshold: started = True frames.append(frame) elif waited > grace_frames: return None # woke it up but said nothing continue frames.append(frame) if frame_rms < rms_threshold: silence_frames += 1 if silence_frames >= silence_limit: break else: silence_frames = 0 if len(frames) >= max_frames: break if not frames: return None pcm = np.concatenate(frames) if len(pcm) < min_utterance_s * sample_rate: return None return pcm def open_input_stream(): """Real sounddevice input stream, imported lazily so pure-logic tests (record_utterance with a fake stream) don't need PortAudio installed.""" import sounddevice as sd return sd.InputStream( samplerate=config.SAMPLE_RATE, channels=1, dtype="int16", blocksize=config.FRAME_LEN, device=config.MIC_DEVICE, )