Wake-word barge-in, Gitea auto-updater, hard_reset fix

This commit is contained in:
2026-07-23 07:30:08 -06:00
parent 80bef6f524
commit 843f52c507
13 changed files with 1127 additions and 47 deletions
+149 -21
View File
@@ -1,26 +1,52 @@
"""Barge-in: notice that the user started talking *while the pet is talking*
so playback can be cut short mid-sentence.
"""Barge-in: notice that the user wants to interrupt *while the pet is
talking* so playback can be cut short mid-sentence.
Deliberately dumber than the utterance VAD in mic.py. The mic hears the pet's
own voice coming back out of the speakers, so a single loud frame proves
nothing — this requires several consecutive frames well above the normal
speech threshold (BARGE_IN_RMS_THRESHOLD defaults to 4x VAD_RMS_THRESHOLD).
Takes the same injectable stream shape as mic.record_utterance, so tests feed
it fake frames instead of real audio hardware.
Two detectors, picked by BARGE_IN_MODE:
- **wake** (default) — the interruption has to be the wake word. Every mic
frame goes through the same openWakeWord model the idle listener uses, so
a sneeze, a door, or the TV can't cut Bolt off mid-sentence; only saying
"thunderbolt" does.
- **energy** — the original behaviour: N consecutive frames above
BARGE_IN_RMS_THRESHOLD. Faster to trigger and needs no model inference,
but it fires on any sustained noise. Deliberately dumber than the
utterance VAD in mic.py, since the mic hears the pet's own voice coming
back out of the speakers, so the threshold defaults to 4x the VAD one.
Both take the same injectable stream shape as mic.record_utterance and expose
the same reset()/check() pair, so tests feed them fake frames instead of real
audio hardware and controller.py doesn't care which one it holds.
"""
from __future__ import annotations
from typing import Callable, Optional, Union
import numpy as np
from .. import config
from .mic import AudioStream, rms
from .wake_word import WakeModel, _default_model, hard_reset
def _read_frame(stream: AudioStream, frame_len: int) -> Optional[np.ndarray]:
"""One mono frame, or None if the mic hiccuped or gave us nothing. Never
raises: a bad frame mid-playback should mean "no barge-in this frame",
not a dead reply."""
try:
chunk, _ = stream.read(frame_len)
except Exception:
return None
frame = np.asarray(chunk)
if frame.ndim > 1:
frame = frame[:, 0]
return frame if frame.size else None
class BargeInDetector:
"""Poll-driven: call check() repeatedly while audio plays. Each call
consumes exactly one mic frame (80ms at the default frame length), which
is also what paces the playback loop's polling."""
"""Energy mode. Poll-driven: call check() repeatedly while audio plays.
Each call consumes exactly one mic frame (80ms at the default frame
length), which is also what paces the playback loop's polling."""
def __init__(
self,
@@ -44,19 +70,121 @@ class BargeInDetector:
def check(self) -> bool:
"""True once the user has been loud for long enough to count as an
interruption. Never raises: a mic hiccup mid-playback should not kill
the reply, it should just mean "no barge-in this frame"."""
try:
chunk, _ = self._stream.read(self._frame_len)
except Exception:
return False
frame = np.asarray(chunk)
if frame.ndim > 1:
frame = frame[:, 0]
if frame.size == 0:
interruption."""
frame = _read_frame(self._stream, self._frame_len)
if frame is None:
return False
if rms(frame) >= self._threshold:
self._loud_frames += 1
else:
self._loud_frames = 0 # a single thump/cough shouldn't count
return self._loud_frames >= self._required
class WakeWordBargeIn:
"""Wake-word mode: only "thunderbolt" interrupts.
Same per-frame predict() loop as listen_for_wake_word, just driven by the
playback poll instead of its own read loop. The model instance is shared
with the idle listener by default — the two never run at the same time
(the pipeline is either speaking or listening), and reusing it avoids
loading a second copy of the ONNX graph.
Two wrinkles the energy detector doesn't have:
- The mic hears the pet's own voice, so the model is scoring Bolt's
speech too. That's harmless unless Bolt says its own wake word, which
is why the threshold can be raised independently
(BARGE_IN_WAKE_THRESHOLD) without desensitizing the idle listener.
- reset() has to be a *hard* reset. openwakeword keeps ~10s of audio
history in its preprocessor, so the "thunderbolt" that started this
turn is still in the model's window when playback begins — feed it one
new frame and it fires on the old phrase, cutting the reply off a word
in. Clearing that window is what makes wake-mode barge-in work at all.
"""
def __init__(
self,
stream: AudioStream,
model: Optional[WakeModel] = None,
threshold: Union[float, Callable[[], float], None] = None,
frame_len: int = config.FRAME_LEN,
on_score: Optional[Callable[[float, float], None]] = None,
):
self._stream = stream
self._model = model if model is not None else _default_model
if threshold is None:
threshold = config.BARGE_IN_WAKE_THRESHOLD or config.WAKE_WORD_THRESHOLD
self._resolve_threshold = threshold if callable(threshold) else (lambda: threshold)
self._frame_len = frame_len
self._on_score = on_score
self._frames = 0
self._peak = 0.0
self._last = 0.0
self._last_threshold = 0.0
# Scoring history for the current reply. Without this an interruption is
# indistinguishable from a crash in the logs — you can't tell a genuine
# "thunderbolt" from the model firing on Bolt's own voice, or on the first
# frame (a stale window) versus halfway through (something it heard).
@property
def frames_checked(self) -> int:
return self._frames
@property
def seconds_checked(self) -> float:
return self._frames * self._frame_len / config.SAMPLE_RATE
@property
def peak_score(self) -> float:
return self._peak
@property
def last_score(self) -> float:
return self._last
@property
def last_threshold(self) -> float:
return self._last_threshold
def reset(self) -> None:
hard_reset(self._model) # never raises
self._frames = 0
self._peak = 0.0
self._last = 0.0
def check(self) -> bool:
frame = _read_frame(self._stream, self._frame_len)
if frame is None:
return False
try:
scores = self._model.predict(frame)
except Exception:
return False # same contract as a mic hiccup: no barge-in, no crash
threshold = self._resolve_threshold()
best = max(scores.values()) if scores else 0.0
self._frames += 1
self._last = best
self._peak = max(self._peak, best)
self._last_threshold = threshold
if self._on_score is not None:
self._on_score(best, threshold)
if scores and best >= threshold:
self.reset()
return True
return False
def make_detector(
stream: AudioStream,
mode: str = None,
wake_threshold: Union[float, Callable[[], float], None] = None,
on_score: Optional[Callable[[float, float], None]] = None,
):
"""Build whichever detector BARGE_IN_MODE asks for. An unrecognized mode
falls back to energy rather than raising — a typo in .env shouldn't stop
the pet from starting."""
mode = (config.BARGE_IN_MODE if mode is None else mode).strip().lower()
if mode in ("wake", "wakeword", "wake_word"):
return WakeWordBargeIn(stream, threshold=wake_threshold, on_score=on_score)
return BargeInDetector(stream)