Files
Bolt-Pet/bolt_pet/audio/tts.py
T

301 lines
12 KiB
Python

"""Text-to-speech: the Bolt server's own `/desk/tts` — the same endpoint the
Android app streams from — requested as raw PCM so playback is just
sounddevice — no external player binary (mpv/ffplay), unlike
desk_client/bolt_desk.py which shells out because it only targets Linux.
No local ElevenLabs account needed for this: the server picks the voice
(ELEVENLABS_VOICE_ID, or a `speak_as` override) and the synthesis model
itself, authenticated with this pet's own DESK_API_KEY. Falls back to
pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech on macOS,
espeak on Linux) if the server call fails, so the pet can still talk even
with the server unreachable.
Every entry point takes an optional *voice_id* that overrides
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
reply marker reaches the speakers (see controller._apply_voice). The offline
fallback has no such concept and always sounds like itself.
`synthesize_dialogue()` below is the one exception: multi-voice
`dialoguectl` scenes have no server endpoint, so that one call still goes
to ElevenLabs' Text to Dialogue API directly and still needs
ELEVENLABS_API_KEY — see dialogue.py.
"""
from __future__ import annotations
import time
from typing import Iterable, Iterator, Optional
import numpy as np
import requests
from .. import config, speech_text
class TtsError(Exception):
pass
def voice_for(voice_id: Optional[str] = None) -> str:
"""The voice this call should use: an override (server `speak_as`) if
given, else the configured default."""
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
def _headers() -> dict:
return {"X-Desk-Api-Key": config.API_KEY}
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
callers should fall back to speak_offline() rather than treating this
as fatal."""
voice = voice_for(voice_id)
if not (config.is_configured() and voice):
raise TtsError("BOLT_SERVER_URL / DESK_API_KEY / ELEVENLABS_VOICE_ID not set")
try:
response = requests.post(
f"{config.SERVER_URL}/desk/tts",
headers=_headers(),
json={"session_id": config.SESSION_ID, "text": text, "voice_id": voice},
timeout=60,
)
response.raise_for_status()
except Exception as exc:
raise TtsError(f"server tts request failed: {exc}") from exc
pcm = np.frombuffer(response.content, dtype=np.int16)
if pcm.size == 0:
raise TtsError("server returned no audio")
return pcm, config.TTS_SAMPLE_RATE
def stream_pcm(
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
) -> Iterator[np.ndarray]:
"""Same audio as synthesize_pcm(), but yielded as it arrives from the
server so playback can start on the first chunk instead of after the
whole clip is synthesized. Raises TtsError before yielding anything if
the request itself fails, so callers can fall back cleanly; a mid-stream
failure just ends the generator."""
voice = voice_for(voice_id)
if not (config.is_configured() and voice):
raise TtsError("BOLT_SERVER_URL / DESK_API_KEY / ELEVENLABS_VOICE_ID not set")
try:
response = requests.post(
f"{config.SERVER_URL}/desk/tts",
headers=_headers(),
json={"session_id": config.SESSION_ID, "text": text, "voice_id": voice},
timeout=60,
stream=True,
)
response.raise_for_status()
except Exception as exc:
raise TtsError(f"server tts stream request failed: {exc}") from exc
return chunks_to_int16(response.iter_content(chunk_size=chunk_bytes))
def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
"""Reassemble a byte stream into int16 frames. HTTP chunk boundaries fall
wherever they like, including *inside* a 16-bit sample, so a trailing odd
byte has to be carried into the next chunk — otherwise every chunk after
the first is shifted by one byte and plays as static."""
carry = b""
for chunk in byte_chunks:
if not chunk:
continue
data = carry + chunk
usable = len(data) - (len(data) % 2)
carry = data[usable:]
if usable:
yield np.frombuffer(data[:usable], dtype=np.int16)
def synthesize_dialogue(
inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None
) -> tuple[np.ndarray, int]:
"""Multi-voice scene via ElevenLabs Text to Dialogue.
One request, one take: the whole exchange is synthesized together, which
is the point — the model hears the previous line, so reactions and timing
land instead of sounding like separately-rendered clips.
Same PCM-over-`requests` posture as the rest of this module (no SDK, no
`play()` shelling out to ffplay), so playback is the same sounddevice path
everything else uses and barge-in works on it unchanged. There is no
documented streaming variant, and a scene is a short set piece anyway, so
this is whole-clip only.
"""
if not (config.ELEVENLABS_API_KEY and inputs):
raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)")
body: dict = {
"inputs": [
{"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")}
for entry in inputs
],
"model_id": model_id or config.DIALOGUE_MODEL_ID,
}
if stability is not None:
body["settings"] = {"stability": float(stability)}
try:
response = requests.post(
"https://api.elevenlabs.io/v1/text-to-dialogue",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json=body,
timeout=120, # a multi-voice take is slower to render than one line
)
response.raise_for_status()
except Exception as exc:
detail = ""
# The API explains refusals (character limit, unknown voice) in the
# body; surfacing it is what lets Bolt fix the call and retry.
body_text = getattr(getattr(exc, "response", None), "text", "")
if body_text:
detail = f" — {body_text[:300]}"
raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc
pcm = np.frombuffer(response.content, dtype=np.int16)
if pcm.size == 0:
raise TtsError("ElevenLabs returned no dialogue audio")
return pcm, config.TTS_SAMPLE_RATE
# ── how loud is it right now ────────────────────────────────────────────────
# The PCM is already decoded here on its way to the speakers, so the amplitude
# envelope is free — and it is exactly what a mouth needs to move in time with
# speech. Throwing it away and animating the mouth on a timer instead is why
# most talking sprites look dubbed.
# int16 RMS that counts as "mouth fully open". Speech peaks around 8-12k;
# 6000 keeps normal talking in the upper half of the range without clipping
# every syllable to wide-open.
_LOUD_RMS = 6000.0
def level_of(frame: np.ndarray) -> float:
"""0..1 loudness for one chunk of PCM.
Square-rooted because perceived loudness is not linear in amplitude — a
linear mapping leaves the mouth barely moving through ordinary speech."""
if frame is None or len(frame) == 0:
return 0.0
rms = float(np.sqrt(np.mean(np.square(frame.astype(np.float32)))))
return float(min(1.0, (rms / _LOUD_RMS) ** 0.5))
def envelope(pcm: np.ndarray, sample_rate: int, fps: int = 30) -> list:
"""Per-frame loudness for a whole clip, for playback that isn't streamed."""
if pcm is None or len(pcm) == 0:
return []
window = max(1, int(sample_rate / max(1, fps)))
return [level_of(pcm[start:start + window]) for start in range(0, len(pcm), window)]
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None,
on_level=None) -> bool:
"""Play a whole clip. Returns True if it finished, False if *should_stop*
(barge-in) cut it short. *should_stop* is polled while audio plays — each
poll consumes one mic frame, which is what paces this loop."""
import sounddevice as sd
levels = envelope(pcm, sample_rate) if on_level is not None else []
started = time.monotonic()
sd.play(pcm, samplerate=sample_rate, device=config.SPEAKER_DEVICE)
if not blocking:
return True
if should_stop is None and on_level is None:
sd.wait()
return True
while True:
if levels:
# Indexed by elapsed time rather than by chunk, because this path
# hands the whole clip to the device at once and never sees it
# again — wall clock is the only position we have.
index = int((time.monotonic() - started) * 30)
if index < len(levels):
try:
on_level(levels[index])
except Exception:
levels = []
try:
if not sd.get_stream().active:
break
except Exception:
break # stream already torn down — playback is over
if should_stop is not None and should_stop():
sd.stop()
return False
return True
def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None,
on_level=None) -> bool:
"""Play int16 chunks as they arrive. Returns False if interrupted.
*on_level* receives each chunk's loudness (0..1) just before it is written,
which is what drives the mouth: the sprite is animated by the same audio
the speakers are getting, not by a guess about how long a word takes."""
import sounddevice as sd
with sd.OutputStream(
samplerate=sample_rate, channels=1, dtype="int16", device=config.SPEAKER_DEVICE
) as out:
for chunk in chunks:
if should_stop is not None and should_stop():
# abort() rather than draining: barge-in should stop the voice
# now, not at the end of the buffered chunk.
out.abort()
return False
if on_level is not None:
try:
on_level(level_of(chunk))
except Exception:
on_level = None # a broken listener must not stop playback
out.write(chunk)
return True
def speak_offline(text: str) -> None:
try:
import pyttsx3
except ImportError:
return # no TTS available at all — caller already logs the text
engine = pyttsx3.init()
engine.say(text)
engine.runAndWait()
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None,
on_level=None) -> bool:
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
exception when ElevenLabs fails (useful for logging) — a fallback still
runs either way. Returns False if barge-in interrupted playback.
*voice_id* overrides the configured voice for this line only.
The text is sanitized first (speech_text.for_speech): server replies are
written for a chat window, and a voice reads markdown/emoji literally
("asterisk asterisk"). Sanitizing here rather than at the call sites means
every path to the speakers — reply, heartbeat announcement — is covered."""
text = speech_text.for_speech(text)
if not text:
return True
if config.TTS_STREAMING:
try:
return play_stream(
stream_pcm(text, voice_id=voice_id),
config.TTS_SAMPLE_RATE,
should_stop=should_stop,
on_level=on_level,
)
except TtsError as exc:
if on_error is not None:
on_error(exc)
try:
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
return play_pcm(pcm, sample_rate, should_stop=should_stop, on_level=on_level)
except TtsError as exc:
if on_error is not None:
on_error(exc)
speak_offline(text)
return True