3a0959f55d
Latency: replies are spoken sentence-by-sentence off the desk API's NDJSON endpoint, so the wait is time-to-first-sentence rather than the whole model call, and Deepgram's live websocket transcribes while you're still talking instead of uploading the WAV afterwards. Both fall back invisibly — a stream that fails before anything was said drops to converse(), and a socket that never opens just means the old one-shot path. Speaking lived in four near-copies in the controller (a reply, a holding line, a streamed sentence, a dialogue scene) that had already drifted: one didn't arm barge-in, another skipped the follow-up rule. It's now speech.Speaker plus an Utterance describing the policy differences, with collaborators injected so the whole of it tests without Qt or audio. The mouth follows the audio rather than a timer: tts.level_of reduces each PCM frame to a 0..1 loudness on a sqrt curve (speech sits well below peak, and a linear map leaves the mouth barely open during normal talking) and that indexes the talking frames, which the sprite script now draws as an openness ramp. Offline pyttsx3 has no waveform, so stale levels hand control back to the timed loop instead of freezing the mouth mid-syllable. Also: the pet starts where you left it (ignoring positions on monitors that are no longer connected, since restoring those faithfully is how it ends up somewhere unreachable), and `python -m bolt_pet --doctor` is a preflight that says what to do about each problem rather than only what's wrong. tests/test_pipeline_smoke.py breaks the pure-logic rule on purpose. Every unit test passed all week while notifications sat unspoken for minutes, the pet said things twice and [laughing] got read aloud — each an interaction between two individually-correct units. It drives whole turns against a real HTTP server on a loopback port, faking only the mic and the speakers. It found a NameError in the paint path that would have fired on every repaint while talking. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
305 lines
12 KiB
Python
305 lines
12 KiB
Python
"""Text-to-speech: ElevenLabs, requested as raw PCM so playback is just
|
|
sounddevice — no external player binary (mpv/ffplay), unlike
|
|
desk_client/bolt_desk.py which shells out because it only targets Linux.
|
|
|
|
Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech
|
|
on macOS, espeak on Linux) if ElevenLabs isn't configured or the request
|
|
fails, so the pet can still talk with zero cloud config.
|
|
|
|
Every entry point takes an optional *voice_id* that overrides
|
|
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
|
|
reply marker reaches the speakers (see controller._apply_voice). The offline
|
|
fallback has no such concept and always sounds like itself.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import time
|
|
from typing import Iterable, Iterator, Optional
|
|
|
|
import numpy as np
|
|
import requests
|
|
|
|
from .. import config, speech_text
|
|
|
|
|
|
class TtsError(Exception):
|
|
pass
|
|
|
|
|
|
def voice_for(voice_id: Optional[str] = None) -> str:
|
|
"""The voice this call should use: an override (server `speak_as`) if
|
|
given, else the configured default."""
|
|
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
|
|
|
|
|
|
def model_for(text: str, voice_id: Optional[str] = None) -> str:
|
|
"""Which ElevenLabs model to synthesize with.
|
|
|
|
The default (`eleven_flash_v2`) is English-only, and both things that
|
|
reach this branch mean the reply probably isn't English: a voice the
|
|
server picked mid-conversation is nearly always about a language or an
|
|
accent, and non-ASCII text can't be English at all. Rendering either one
|
|
through the English model gets you a mangled phonetic reading rather
|
|
than a failure, which is worse — so those go through the multilingual
|
|
model instead."""
|
|
if (voice_id or "").strip() or not text.isascii():
|
|
return config.ELEVENLABS_MULTILINGUAL_MODEL_ID
|
|
return config.ELEVENLABS_MODEL_ID
|
|
|
|
|
|
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
|
|
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
|
|
callers should fall back to speak_offline() rather than treating this
|
|
as fatal."""
|
|
voice = voice_for(voice_id)
|
|
if not (config.ELEVENLABS_API_KEY and voice):
|
|
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
|
try:
|
|
response = requests.post(
|
|
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}",
|
|
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
|
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
|
json={"text": text, "model_id": model_for(text, voice_id)},
|
|
timeout=60,
|
|
)
|
|
response.raise_for_status()
|
|
except Exception as exc:
|
|
raise TtsError(f"ElevenLabs request failed: {exc}") from exc
|
|
pcm = np.frombuffer(response.content, dtype=np.int16)
|
|
if pcm.size == 0:
|
|
raise TtsError("ElevenLabs returned no audio")
|
|
return pcm, config.TTS_SAMPLE_RATE
|
|
|
|
|
|
def stream_pcm(
|
|
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
|
|
) -> Iterator[np.ndarray]:
|
|
"""Same audio as synthesize_pcm(), but yielded as it arrives from
|
|
ElevenLabs' /stream endpoint so playback can start on the first chunk
|
|
(~300ms) instead of after the whole clip is synthesized. Raises TtsError
|
|
before yielding anything if the request itself fails, so callers can fall
|
|
back cleanly; a mid-stream failure just ends the generator."""
|
|
voice = voice_for(voice_id)
|
|
if not (config.ELEVENLABS_API_KEY and voice):
|
|
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
|
try:
|
|
response = requests.post(
|
|
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}/stream",
|
|
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
|
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
|
json={"text": text, "model_id": model_for(text, voice_id)},
|
|
timeout=60,
|
|
stream=True,
|
|
)
|
|
response.raise_for_status()
|
|
except Exception as exc:
|
|
raise TtsError(f"ElevenLabs stream request failed: {exc}") from exc
|
|
return chunks_to_int16(response.iter_content(chunk_size=chunk_bytes))
|
|
|
|
|
|
def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
|
|
"""Reassemble a byte stream into int16 frames. HTTP chunk boundaries fall
|
|
wherever they like, including *inside* a 16-bit sample, so a trailing odd
|
|
byte has to be carried into the next chunk — otherwise every chunk after
|
|
the first is shifted by one byte and plays as static."""
|
|
carry = b""
|
|
for chunk in byte_chunks:
|
|
if not chunk:
|
|
continue
|
|
data = carry + chunk
|
|
usable = len(data) - (len(data) % 2)
|
|
carry = data[usable:]
|
|
if usable:
|
|
yield np.frombuffer(data[:usable], dtype=np.int16)
|
|
|
|
|
|
def synthesize_dialogue(
|
|
inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None
|
|
) -> tuple[np.ndarray, int]:
|
|
"""Multi-voice scene via ElevenLabs Text to Dialogue.
|
|
|
|
One request, one take: the whole exchange is synthesized together, which
|
|
is the point — the model hears the previous line, so reactions and timing
|
|
land instead of sounding like separately-rendered clips.
|
|
|
|
Same PCM-over-`requests` posture as the rest of this module (no SDK, no
|
|
`play()` shelling out to ffplay), so playback is the same sounddevice path
|
|
everything else uses and barge-in works on it unchanged. There is no
|
|
documented streaming variant, and a scene is a short set piece anyway, so
|
|
this is whole-clip only.
|
|
"""
|
|
if not (config.ELEVENLABS_API_KEY and inputs):
|
|
raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)")
|
|
body: dict = {
|
|
"inputs": [
|
|
{"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")}
|
|
for entry in inputs
|
|
],
|
|
"model_id": model_id or config.DIALOGUE_MODEL_ID,
|
|
}
|
|
if stability is not None:
|
|
body["settings"] = {"stability": float(stability)}
|
|
try:
|
|
response = requests.post(
|
|
"https://api.elevenlabs.io/v1/text-to-dialogue",
|
|
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
|
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
|
json=body,
|
|
timeout=120, # a multi-voice take is slower to render than one line
|
|
)
|
|
response.raise_for_status()
|
|
except Exception as exc:
|
|
detail = ""
|
|
# The API explains refusals (character limit, unknown voice) in the
|
|
# body; surfacing it is what lets Bolt fix the call and retry.
|
|
body_text = getattr(getattr(exc, "response", None), "text", "")
|
|
if body_text:
|
|
detail = f" — {body_text[:300]}"
|
|
raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc
|
|
pcm = np.frombuffer(response.content, dtype=np.int16)
|
|
if pcm.size == 0:
|
|
raise TtsError("ElevenLabs returned no dialogue audio")
|
|
return pcm, config.TTS_SAMPLE_RATE
|
|
|
|
|
|
# ── how loud is it right now ────────────────────────────────────────────────
|
|
# The PCM is already decoded here on its way to the speakers, so the amplitude
|
|
# envelope is free — and it is exactly what a mouth needs to move in time with
|
|
# speech. Throwing it away and animating the mouth on a timer instead is why
|
|
# most talking sprites look dubbed.
|
|
|
|
# int16 RMS that counts as "mouth fully open". Speech peaks around 8-12k;
|
|
# 6000 keeps normal talking in the upper half of the range without clipping
|
|
# every syllable to wide-open.
|
|
_LOUD_RMS = 6000.0
|
|
|
|
|
|
def level_of(frame: np.ndarray) -> float:
|
|
"""0..1 loudness for one chunk of PCM.
|
|
|
|
Square-rooted because perceived loudness is not linear in amplitude — a
|
|
linear mapping leaves the mouth barely moving through ordinary speech."""
|
|
if frame is None or len(frame) == 0:
|
|
return 0.0
|
|
rms = float(np.sqrt(np.mean(np.square(frame.astype(np.float32)))))
|
|
return float(min(1.0, (rms / _LOUD_RMS) ** 0.5))
|
|
|
|
|
|
def envelope(pcm: np.ndarray, sample_rate: int, fps: int = 30) -> list:
|
|
"""Per-frame loudness for a whole clip, for playback that isn't streamed."""
|
|
if pcm is None or len(pcm) == 0:
|
|
return []
|
|
window = max(1, int(sample_rate / max(1, fps)))
|
|
return [level_of(pcm[start:start + window]) for start in range(0, len(pcm), window)]
|
|
|
|
|
|
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None,
|
|
on_level=None) -> bool:
|
|
"""Play a whole clip. Returns True if it finished, False if *should_stop*
|
|
(barge-in) cut it short. *should_stop* is polled while audio plays — each
|
|
poll consumes one mic frame, which is what paces this loop."""
|
|
import sounddevice as sd
|
|
|
|
levels = envelope(pcm, sample_rate) if on_level is not None else []
|
|
started = time.monotonic()
|
|
sd.play(pcm, samplerate=sample_rate, device=config.SPEAKER_DEVICE)
|
|
if not blocking:
|
|
return True
|
|
if should_stop is None and on_level is None:
|
|
sd.wait()
|
|
return True
|
|
while True:
|
|
if levels:
|
|
# Indexed by elapsed time rather than by chunk, because this path
|
|
# hands the whole clip to the device at once and never sees it
|
|
# again — wall clock is the only position we have.
|
|
index = int((time.monotonic() - started) * 30)
|
|
if index < len(levels):
|
|
try:
|
|
on_level(levels[index])
|
|
except Exception:
|
|
levels = []
|
|
try:
|
|
if not sd.get_stream().active:
|
|
break
|
|
except Exception:
|
|
break # stream already torn down — playback is over
|
|
if should_stop is not None and should_stop():
|
|
sd.stop()
|
|
return False
|
|
return True
|
|
|
|
|
|
def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None,
|
|
on_level=None) -> bool:
|
|
"""Play int16 chunks as they arrive. Returns False if interrupted.
|
|
|
|
*on_level* receives each chunk's loudness (0..1) just before it is written,
|
|
which is what drives the mouth: the sprite is animated by the same audio
|
|
the speakers are getting, not by a guess about how long a word takes."""
|
|
import sounddevice as sd
|
|
|
|
with sd.OutputStream(
|
|
samplerate=sample_rate, channels=1, dtype="int16", device=config.SPEAKER_DEVICE
|
|
) as out:
|
|
for chunk in chunks:
|
|
if should_stop is not None and should_stop():
|
|
# abort() rather than draining: barge-in should stop the voice
|
|
# now, not at the end of the buffered chunk.
|
|
out.abort()
|
|
return False
|
|
if on_level is not None:
|
|
try:
|
|
on_level(level_of(chunk))
|
|
except Exception:
|
|
on_level = None # a broken listener must not stop playback
|
|
out.write(chunk)
|
|
return True
|
|
|
|
|
|
def speak_offline(text: str) -> None:
|
|
try:
|
|
import pyttsx3
|
|
except ImportError:
|
|
return # no TTS available at all — caller already logs the text
|
|
engine = pyttsx3.init()
|
|
engine.say(text)
|
|
engine.runAndWait()
|
|
|
|
|
|
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None,
|
|
on_level=None) -> bool:
|
|
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
|
|
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
|
|
exception when ElevenLabs fails (useful for logging) — a fallback still
|
|
runs either way. Returns False if barge-in interrupted playback.
|
|
*voice_id* overrides the configured voice for this line only.
|
|
|
|
The text is sanitized first (speech_text.for_speech): server replies are
|
|
written for a chat window, and a voice reads markdown/emoji literally
|
|
("asterisk asterisk"). Sanitizing here rather than at the call sites means
|
|
every path to the speakers — reply, heartbeat announcement — is covered."""
|
|
text = speech_text.for_speech(text)
|
|
if not text:
|
|
return True
|
|
if config.TTS_STREAMING:
|
|
try:
|
|
return play_stream(
|
|
stream_pcm(text, voice_id=voice_id),
|
|
config.TTS_SAMPLE_RATE,
|
|
should_stop=should_stop,
|
|
on_level=on_level,
|
|
)
|
|
except TtsError as exc:
|
|
if on_error is not None:
|
|
on_error(exc)
|
|
try:
|
|
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
|
|
return play_pcm(pcm, sample_rate, should_stop=should_stop, on_level=on_level)
|
|
except TtsError as exc:
|
|
if on_error is not None:
|
|
on_error(exc)
|
|
speak_offline(text)
|
|
return True
|