Upload
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,111 @@
|
||||
"""Mic capture + simple energy-based VAD utterance recording.
|
||||
|
||||
Ported from desk_client/bolt_desk.py's record_utterance() — same tuning
|
||||
knobs, same behavior. Kept independent of any UI/threading model so it can
|
||||
be unit tested by feeding it a fake "stream" object.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import wave
|
||||
from typing import Optional, Protocol
|
||||
|
||||
import numpy as np
|
||||
|
||||
from .. import config
|
||||
|
||||
|
||||
class AudioStream(Protocol):
|
||||
"""Minimal shape of the object record_utterance() needs — matches
|
||||
sounddevice.InputStream's .read(frames) -> (data, overflowed)."""
|
||||
|
||||
def read(self, frames: int): ...
|
||||
|
||||
|
||||
def pcm_to_wav_bytes(pcm: np.ndarray, sample_rate: int = config.SAMPLE_RATE) -> bytes:
|
||||
buf = io.BytesIO()
|
||||
with wave.open(buf, "wb") as wf:
|
||||
wf.setnchannels(1)
|
||||
wf.setsampwidth(2)
|
||||
wf.setframerate(sample_rate)
|
||||
wf.writeframes(pcm.tobytes())
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def rms(frame: np.ndarray) -> float:
|
||||
return float(np.sqrt(np.mean(frame.astype(np.float64) ** 2)))
|
||||
|
||||
|
||||
def record_utterance(
|
||||
stream: AudioStream,
|
||||
should_continue=lambda: True,
|
||||
rms_threshold: int = None,
|
||||
silence_end_sec: float = None,
|
||||
max_utterance_s: float = None,
|
||||
min_utterance_s: float = None,
|
||||
frame_len: int = config.FRAME_LEN,
|
||||
sample_rate: int = config.SAMPLE_RATE,
|
||||
) -> Optional[np.ndarray]:
|
||||
"""Capture one utterance from *stream*: wait for speech to start, stop
|
||||
after trailing silence. Returns None if nothing usable was heard.
|
||||
|
||||
*should_continue* is polled each frame so a caller can cancel recording
|
||||
(e.g. the pet window was closed) without needing threading primitives
|
||||
baked into this function.
|
||||
"""
|
||||
rms_threshold = config.RMS_THRESHOLD if rms_threshold is None else rms_threshold
|
||||
silence_end_sec = config.SILENCE_END_SEC if silence_end_sec is None else silence_end_sec
|
||||
max_utterance_s = config.MAX_UTTERANCE_S if max_utterance_s is None else max_utterance_s
|
||||
min_utterance_s = config.MIN_UTTERANCE_S if min_utterance_s is None else min_utterance_s
|
||||
|
||||
frames: list[np.ndarray] = []
|
||||
started = False
|
||||
silence_frames = 0
|
||||
silence_limit = int(silence_end_sec * sample_rate / frame_len)
|
||||
max_frames = int(max_utterance_s * sample_rate / frame_len)
|
||||
grace_frames = int(4.0 * sample_rate / frame_len) # wait up to 4s for speech to begin
|
||||
waited = 0
|
||||
|
||||
while should_continue():
|
||||
chunk, _ = stream.read(frame_len)
|
||||
frame = np.asarray(chunk)[:, 0].copy()
|
||||
frame_rms = rms(frame)
|
||||
if not started:
|
||||
waited += 1
|
||||
if frame_rms >= rms_threshold:
|
||||
started = True
|
||||
frames.append(frame)
|
||||
elif waited > grace_frames:
|
||||
return None # woke it up but said nothing
|
||||
continue
|
||||
frames.append(frame)
|
||||
if frame_rms < rms_threshold:
|
||||
silence_frames += 1
|
||||
if silence_frames >= silence_limit:
|
||||
break
|
||||
else:
|
||||
silence_frames = 0
|
||||
if len(frames) >= max_frames:
|
||||
break
|
||||
|
||||
if not frames:
|
||||
return None
|
||||
pcm = np.concatenate(frames)
|
||||
if len(pcm) < min_utterance_s * sample_rate:
|
||||
return None
|
||||
return pcm
|
||||
|
||||
|
||||
def open_input_stream():
|
||||
"""Real sounddevice input stream, imported lazily so pure-logic tests
|
||||
(record_utterance with a fake stream) don't need PortAudio installed."""
|
||||
import sounddevice as sd
|
||||
|
||||
return sd.InputStream(
|
||||
samplerate=config.SAMPLE_RATE,
|
||||
channels=1,
|
||||
dtype="int16",
|
||||
blocksize=config.FRAME_LEN,
|
||||
device=config.MIC_DEVICE,
|
||||
)
|
||||
Reference in New Issue
Block a user