301 lines
12 KiB
Python
301 lines
12 KiB
Python
"""Text-to-speech: the Bolt server's own `/desk/tts` — the same endpoint the
|
|
Android app streams from — requested as raw PCM so playback is just
|
|
sounddevice — no external player binary (mpv/ffplay), unlike
|
|
desk_client/bolt_desk.py which shells out because it only targets Linux.
|
|
|
|
No local ElevenLabs account needed for this: the server picks the voice
|
|
(ELEVENLABS_VOICE_ID, or a `speak_as` override) and the synthesis model
|
|
itself, authenticated with this pet's own DESK_API_KEY. Falls back to
|
|
pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech on macOS,
|
|
espeak on Linux) if the server call fails, so the pet can still talk even
|
|
with the server unreachable.
|
|
|
|
Every entry point takes an optional *voice_id* that overrides
|
|
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
|
|
reply marker reaches the speakers (see controller._apply_voice). The offline
|
|
fallback has no such concept and always sounds like itself.
|
|
|
|
`synthesize_dialogue()` below is the one exception: multi-voice
|
|
`dialoguectl` scenes have no server endpoint, so that one call still goes
|
|
to ElevenLabs' Text to Dialogue API directly and still needs
|
|
ELEVENLABS_API_KEY — see dialogue.py.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import time
|
|
from typing import Iterable, Iterator, Optional
|
|
|
|
import numpy as np
|
|
import requests
|
|
|
|
from .. import config, speech_text
|
|
|
|
|
|
class TtsError(Exception):
|
|
pass
|
|
|
|
|
|
def voice_for(voice_id: Optional[str] = None) -> str:
|
|
"""The voice this call should use: an override (server `speak_as`) if
|
|
given, else the configured default."""
|
|
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
|
|
|
|
|
|
def _headers() -> dict:
|
|
return {"X-Desk-Api-Key": config.API_KEY}
|
|
|
|
|
|
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
|
|
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
|
|
callers should fall back to speak_offline() rather than treating this
|
|
as fatal."""
|
|
voice = voice_for(voice_id)
|
|
if not (config.is_configured() and voice):
|
|
raise TtsError("BOLT_SERVER_URL / DESK_API_KEY / ELEVENLABS_VOICE_ID not set")
|
|
try:
|
|
response = requests.post(
|
|
f"{config.SERVER_URL}/desk/tts",
|
|
headers=_headers(),
|
|
json={"session_id": config.SESSION_ID, "text": text, "voice_id": voice},
|
|
timeout=60,
|
|
)
|
|
response.raise_for_status()
|
|
except Exception as exc:
|
|
raise TtsError(f"server tts request failed: {exc}") from exc
|
|
pcm = np.frombuffer(response.content, dtype=np.int16)
|
|
if pcm.size == 0:
|
|
raise TtsError("server returned no audio")
|
|
return pcm, config.TTS_SAMPLE_RATE
|
|
|
|
|
|
def stream_pcm(
|
|
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
|
|
) -> Iterator[np.ndarray]:
|
|
"""Same audio as synthesize_pcm(), but yielded as it arrives from the
|
|
server so playback can start on the first chunk instead of after the
|
|
whole clip is synthesized. Raises TtsError before yielding anything if
|
|
the request itself fails, so callers can fall back cleanly; a mid-stream
|
|
failure just ends the generator."""
|
|
voice = voice_for(voice_id)
|
|
if not (config.is_configured() and voice):
|
|
raise TtsError("BOLT_SERVER_URL / DESK_API_KEY / ELEVENLABS_VOICE_ID not set")
|
|
try:
|
|
response = requests.post(
|
|
f"{config.SERVER_URL}/desk/tts",
|
|
headers=_headers(),
|
|
json={"session_id": config.SESSION_ID, "text": text, "voice_id": voice},
|
|
timeout=60,
|
|
stream=True,
|
|
)
|
|
response.raise_for_status()
|
|
except Exception as exc:
|
|
raise TtsError(f"server tts stream request failed: {exc}") from exc
|
|
return chunks_to_int16(response.iter_content(chunk_size=chunk_bytes))
|
|
|
|
|
|
def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
|
|
"""Reassemble a byte stream into int16 frames. HTTP chunk boundaries fall
|
|
wherever they like, including *inside* a 16-bit sample, so a trailing odd
|
|
byte has to be carried into the next chunk — otherwise every chunk after
|
|
the first is shifted by one byte and plays as static."""
|
|
carry = b""
|
|
for chunk in byte_chunks:
|
|
if not chunk:
|
|
continue
|
|
data = carry + chunk
|
|
usable = len(data) - (len(data) % 2)
|
|
carry = data[usable:]
|
|
if usable:
|
|
yield np.frombuffer(data[:usable], dtype=np.int16)
|
|
|
|
|
|
def synthesize_dialogue(
|
|
inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None
|
|
) -> tuple[np.ndarray, int]:
|
|
"""Multi-voice scene via ElevenLabs Text to Dialogue.
|
|
|
|
One request, one take: the whole exchange is synthesized together, which
|
|
is the point — the model hears the previous line, so reactions and timing
|
|
land instead of sounding like separately-rendered clips.
|
|
|
|
Same PCM-over-`requests` posture as the rest of this module (no SDK, no
|
|
`play()` shelling out to ffplay), so playback is the same sounddevice path
|
|
everything else uses and barge-in works on it unchanged. There is no
|
|
documented streaming variant, and a scene is a short set piece anyway, so
|
|
this is whole-clip only.
|
|
"""
|
|
if not (config.ELEVENLABS_API_KEY and inputs):
|
|
raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)")
|
|
body: dict = {
|
|
"inputs": [
|
|
{"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")}
|
|
for entry in inputs
|
|
],
|
|
"model_id": model_id or config.DIALOGUE_MODEL_ID,
|
|
}
|
|
if stability is not None:
|
|
body["settings"] = {"stability": float(stability)}
|
|
try:
|
|
response = requests.post(
|
|
"https://api.elevenlabs.io/v1/text-to-dialogue",
|
|
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
|
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
|
json=body,
|
|
timeout=120, # a multi-voice take is slower to render than one line
|
|
)
|
|
response.raise_for_status()
|
|
except Exception as exc:
|
|
detail = ""
|
|
# The API explains refusals (character limit, unknown voice) in the
|
|
# body; surfacing it is what lets Bolt fix the call and retry.
|
|
body_text = getattr(getattr(exc, "response", None), "text", "")
|
|
if body_text:
|
|
detail = f" — {body_text[:300]}"
|
|
raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc
|
|
pcm = np.frombuffer(response.content, dtype=np.int16)
|
|
if pcm.size == 0:
|
|
raise TtsError("ElevenLabs returned no dialogue audio")
|
|
return pcm, config.TTS_SAMPLE_RATE
|
|
|
|
|
|
# ── how loud is it right now ────────────────────────────────────────────────
|
|
# The PCM is already decoded here on its way to the speakers, so the amplitude
|
|
# envelope is free — and it is exactly what a mouth needs to move in time with
|
|
# speech. Throwing it away and animating the mouth on a timer instead is why
|
|
# most talking sprites look dubbed.
|
|
|
|
# int16 RMS that counts as "mouth fully open". Speech peaks around 8-12k;
|
|
# 6000 keeps normal talking in the upper half of the range without clipping
|
|
# every syllable to wide-open.
|
|
_LOUD_RMS = 6000.0
|
|
|
|
|
|
def level_of(frame: np.ndarray) -> float:
|
|
"""0..1 loudness for one chunk of PCM.
|
|
|
|
Square-rooted because perceived loudness is not linear in amplitude — a
|
|
linear mapping leaves the mouth barely moving through ordinary speech."""
|
|
if frame is None or len(frame) == 0:
|
|
return 0.0
|
|
rms = float(np.sqrt(np.mean(np.square(frame.astype(np.float32)))))
|
|
return float(min(1.0, (rms / _LOUD_RMS) ** 0.5))
|
|
|
|
|
|
def envelope(pcm: np.ndarray, sample_rate: int, fps: int = 30) -> list:
|
|
"""Per-frame loudness for a whole clip, for playback that isn't streamed."""
|
|
if pcm is None or len(pcm) == 0:
|
|
return []
|
|
window = max(1, int(sample_rate / max(1, fps)))
|
|
return [level_of(pcm[start:start + window]) for start in range(0, len(pcm), window)]
|
|
|
|
|
|
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None,
|
|
on_level=None) -> bool:
|
|
"""Play a whole clip. Returns True if it finished, False if *should_stop*
|
|
(barge-in) cut it short. *should_stop* is polled while audio plays — each
|
|
poll consumes one mic frame, which is what paces this loop."""
|
|
import sounddevice as sd
|
|
|
|
levels = envelope(pcm, sample_rate) if on_level is not None else []
|
|
started = time.monotonic()
|
|
sd.play(pcm, samplerate=sample_rate, device=config.SPEAKER_DEVICE)
|
|
if not blocking:
|
|
return True
|
|
if should_stop is None and on_level is None:
|
|
sd.wait()
|
|
return True
|
|
while True:
|
|
if levels:
|
|
# Indexed by elapsed time rather than by chunk, because this path
|
|
# hands the whole clip to the device at once and never sees it
|
|
# again — wall clock is the only position we have.
|
|
index = int((time.monotonic() - started) * 30)
|
|
if index < len(levels):
|
|
try:
|
|
on_level(levels[index])
|
|
except Exception:
|
|
levels = []
|
|
try:
|
|
if not sd.get_stream().active:
|
|
break
|
|
except Exception:
|
|
break # stream already torn down — playback is over
|
|
if should_stop is not None and should_stop():
|
|
sd.stop()
|
|
return False
|
|
return True
|
|
|
|
|
|
def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None,
|
|
on_level=None) -> bool:
|
|
"""Play int16 chunks as they arrive. Returns False if interrupted.
|
|
|
|
*on_level* receives each chunk's loudness (0..1) just before it is written,
|
|
which is what drives the mouth: the sprite is animated by the same audio
|
|
the speakers are getting, not by a guess about how long a word takes."""
|
|
import sounddevice as sd
|
|
|
|
with sd.OutputStream(
|
|
samplerate=sample_rate, channels=1, dtype="int16", device=config.SPEAKER_DEVICE
|
|
) as out:
|
|
for chunk in chunks:
|
|
if should_stop is not None and should_stop():
|
|
# abort() rather than draining: barge-in should stop the voice
|
|
# now, not at the end of the buffered chunk.
|
|
out.abort()
|
|
return False
|
|
if on_level is not None:
|
|
try:
|
|
on_level(level_of(chunk))
|
|
except Exception:
|
|
on_level = None # a broken listener must not stop playback
|
|
out.write(chunk)
|
|
return True
|
|
|
|
|
|
def speak_offline(text: str) -> None:
|
|
try:
|
|
import pyttsx3
|
|
except ImportError:
|
|
return # no TTS available at all — caller already logs the text
|
|
engine = pyttsx3.init()
|
|
engine.say(text)
|
|
engine.runAndWait()
|
|
|
|
|
|
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None,
|
|
on_level=None) -> bool:
|
|
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
|
|
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
|
|
exception when ElevenLabs fails (useful for logging) — a fallback still
|
|
runs either way. Returns False if barge-in interrupted playback.
|
|
*voice_id* overrides the configured voice for this line only.
|
|
|
|
The text is sanitized first (speech_text.for_speech): server replies are
|
|
written for a chat window, and a voice reads markdown/emoji literally
|
|
("asterisk asterisk"). Sanitizing here rather than at the call sites means
|
|
every path to the speakers — reply, heartbeat announcement — is covered."""
|
|
text = speech_text.for_speech(text)
|
|
if not text:
|
|
return True
|
|
if config.TTS_STREAMING:
|
|
try:
|
|
return play_stream(
|
|
stream_pcm(text, voice_id=voice_id),
|
|
config.TTS_SAMPLE_RATE,
|
|
should_stop=should_stop,
|
|
on_level=on_level,
|
|
)
|
|
except TtsError as exc:
|
|
if on_error is not None:
|
|
on_error(exc)
|
|
try:
|
|
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
|
|
return play_pcm(pcm, sample_rate, should_stop=should_stop, on_level=on_level)
|
|
except TtsError as exc:
|
|
if on_error is not None:
|
|
on_error(exc)
|
|
speak_offline(text)
|
|
return True
|