Update desktop app to android app capabilities.
This commit is contained in:
+14
-13
@@ -21,15 +21,14 @@ DESK_API_KEY=
|
|||||||
#WAKE_WORD_THRESHOLD=0.5
|
#WAKE_WORD_THRESHOLD=0.5
|
||||||
#WAKE_CHECK_INTERVAL_SECONDS=1.2
|
#WAKE_CHECK_INTERVAL_SECONDS=1.2
|
||||||
|
|
||||||
# ── STT (Deepgram) ──────────────────────────────────────────────────────────
|
# ── STT / TTS (server-hosted — same /desk/stt and /desk/tts the Android app
|
||||||
DEEPGRAM_API_KEY=
|
# uses) ──────────────────────────────────────────────────────────────────────
|
||||||
#DEEPGRAM_MODEL=nova-3
|
# No Deepgram or ElevenLabs account needed here: both go through
|
||||||
|
# BOLT_SERVER_URL/DESK_API_KEY above, the same as the rest of this file.
|
||||||
# ── TTS (ElevenLabs) — omit to use offline TTS only ─────────────────────────
|
# ELEVENLABS_VOICE_ID just tells the server which voice to request — omit it
|
||||||
ELEVENLABS_API_KEY=
|
# to fall back to offline pyttsx3 TTS instead (STT still works either way).
|
||||||
ELEVENLABS_VOICE_ID=
|
ELEVENLABS_VOICE_ID=
|
||||||
#ELEVENLABS_MODEL_ID=eleven_flash_v2
|
#TTS_SAMPLE_RATE=16000
|
||||||
#TTS_SAMPLE_RATE=24000
|
|
||||||
|
|
||||||
# Ask Bolt to use a different voice (or another language) and the server
|
# Ask Bolt to use a different voice (or another language) and the server
|
||||||
# picks one from the ElevenLabs voice library and tags the reply with it.
|
# picks one from the ElevenLabs voice library and tags the reply with it.
|
||||||
@@ -40,10 +39,14 @@ ELEVENLABS_VOICE_ID=
|
|||||||
#VOICE_STICKY=true
|
#VOICE_STICKY=true
|
||||||
|
|
||||||
# ── Multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
|
# ── Multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
|
||||||
|
# The one feature the server has no endpoint for, so this is the only place
|
||||||
|
# in the whole app that still needs a local ElevenLabs API key — everything
|
||||||
|
# else (the normal reply voice, transcription) goes through the server above.
|
||||||
# Lets Bolt play a short scene in several voices with delivery tags the v3
|
# Lets Bolt play a short scene in several voices with delivery tags the v3
|
||||||
# model acts on ("[cheerfully] Hello", "[whispering] He is lying"), driven by
|
# model acts on ("[cheerfully] Hello", "[whispering] He is lying"), driven by
|
||||||
# the server through a relayed `dialoguectl` command. Name the cast here —
|
# the server through a relayed `dialoguectl` command. Name the cast here —
|
||||||
# "self" always means whatever voice the pet is currently using.
|
# "self" always means whatever voice the pet is currently using.
|
||||||
|
ELEVENLABS_API_KEY=
|
||||||
#DIALOGUE=true
|
#DIALOGUE=true
|
||||||
#DIALOGUE_MODEL_ID=eleven_v3
|
#DIALOGUE_MODEL_ID=eleven_v3
|
||||||
#DIALOGUE_VOICES=narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD
|
#DIALOGUE_VOICES=narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD
|
||||||
@@ -57,9 +60,6 @@ ELEVENLABS_VOICE_ID=
|
|||||||
#SELF_RESTART=true
|
#SELF_RESTART=true
|
||||||
#SELF_RESTART_MAX=5
|
#SELF_RESTART_MAX=5
|
||||||
#SELF_RESTART_WINDOW_SECONDS=900
|
#SELF_RESTART_WINDOW_SECONDS=900
|
||||||
# Used instead of ELEVENLABS_MODEL_ID whenever the server picked the voice
|
|
||||||
# or the reply has non-ASCII in it — the flash_v2 default is English-only.
|
|
||||||
#ELEVENLABS_MULTILINGUAL_MODEL_ID=eleven_flash_v2_5
|
|
||||||
|
|
||||||
# ── Audio devices (optional — leave blank for the system default) ──────────
|
# ── Audio devices (optional — leave blank for the system default) ──────────
|
||||||
#MIC_DEVICE=
|
#MIC_DEVICE=
|
||||||
@@ -144,8 +144,9 @@ ELEVENLABS_VOICE_ID=
|
|||||||
# ── Latency: streaming the reply and the transcript ─────────────────────────
|
# ── Latency: streaming the reply and the transcript ─────────────────────────
|
||||||
# STREAMING_REPLIES speaks each sentence as the server generates it, instead of
|
# STREAMING_REPLIES speaks each sentence as the server generates it, instead of
|
||||||
# waiting out the whole model call before the first word. STT_STREAMING sends
|
# waiting out the whole model call before the first word. STT_STREAMING sends
|
||||||
# mic frames to Deepgram as you talk, so the transcript is ready the moment you
|
# mic frames to the server's /desk/stt relay as you talk, so the transcript is
|
||||||
# stop. Both fall back to the old path automatically if anything goes wrong.
|
# ready the moment you stop, instead of uploading the whole clip afterward.
|
||||||
|
# Both fall back to the old path automatically if anything goes wrong.
|
||||||
# VAD_SILENCE_END_SEC is the other half: it is dead air on every single turn,
|
# VAD_SILENCE_END_SEC is the other half: it is dead air on every single turn,
|
||||||
# so 0.8-1.0 feels markedly snappier than the 1.2 default.
|
# so 0.8-1.0 feels markedly snappier than the 1.2 default.
|
||||||
#STREAMING_REPLIES=true
|
#STREAMING_REPLIES=true
|
||||||
|
|||||||
@@ -11,14 +11,15 @@ dependency** on the server repo; it's a standalone HTTP client configured via
|
|||||||
its own `.env`.
|
its own `.env`.
|
||||||
|
|
||||||
Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk /
|
Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk /
|
||||||
click → record utterance → Deepgram STT → + active-window + screen-layout
|
click → record utterance → the server's own /desk/stt (same relay the
|
||||||
context → POST /desk/converse → [server may relay a shell command to run on
|
Android app uses — no local Deepgram account) → + active-window + screen-
|
||||||
this machine, or a `petctl` pseudo-command that moves/emotes the pet, jumps it
|
layout context → POST /desk/converse → [server may relay a shell command to
|
||||||
to another monitor, reads a screen's text back, or plays a multi-voice scene
|
run on this machine, or a `petctl` pseudo-command that moves/emotes the pet,
|
||||||
instead] → reply (optionally tagged with a voice the server picked for it) →
|
jumps it to another monitor, reads a screen's text back, or plays a
|
||||||
ElevenLabs streaming TTS (or offline pyttsx3 fallback) → speakers`, with the
|
multi-voice scene instead] → reply (optionally tagged with a voice the
|
||||||
pet sprite/speech bubble reflecting state throughout, and playback
|
server picked for it) → the server's own /desk/tts, streaming (or offline
|
||||||
interruptible by talking over it (barge-in).
|
pyttsx3 fallback) → speakers`, with the pet sprite/speech bubble reflecting
|
||||||
|
state throughout, and playback interruptible by talking over it (barge-in).
|
||||||
|
|
||||||
Because a relayed command's output goes back up the tool-result relay before
|
Because a relayed command's output goes back up the tool-result relay before
|
||||||
the final reply, a `petctl read` mid-turn means Bolt can look at a monitor and
|
the final reply, a `petctl read` mid-turn means Bolt can look at a monitor and
|
||||||
@@ -52,9 +53,12 @@ python scripts/slice_spritesheet.py path/to/sheet.png assets/sprites/idle --cols
|
|||||||
```
|
```
|
||||||
|
|
||||||
There is no lint/build step configured beyond pytest. `cp .env.example .env`
|
There is no lint/build step configured beyond pytest. `cp .env.example .env`
|
||||||
and fill in `BOLT_SERVER_URL` / `DESK_API_KEY` (+ `DEEPGRAM_API_KEY`,
|
and fill in `BOLT_SERVER_URL` / `DESK_API_KEY` before running — without them
|
||||||
`ELEVENLABS_API_KEY`) before running — without server config the controller
|
the controller logs a missing-config message and exits its thread instead of
|
||||||
logs a missing-config message and exits its thread instead of starting.
|
starting. That's also all STT and normal-reply TTS need now (both go through
|
||||||
|
the server); `ELEVENLABS_VOICE_ID` picks the voice, and `ELEVENLABS_API_KEY`
|
||||||
|
is only for the one feature with no server endpoint — multi-voice
|
||||||
|
`dialoguectl` scenes, see dialogue.py below.
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
@@ -107,16 +111,17 @@ logs a missing-config message and exits its thread instead of starting.
|
|||||||
off entirely with `RECEIVE_FILES=false`.
|
off entirely with `RECEIVE_FILES=false`.
|
||||||
- **`audio/`** — `mic.py` (energy-based VAD utterance capture, ported from the
|
- **`audio/`** — `mic.py` (energy-based VAD utterance capture, ported from the
|
||||||
server repo's `bolt_desk.py`), `wake_word.py` (openWakeWord `thunderbolt.onnx`
|
server repo's `bolt_desk.py`), `wake_word.py` (openWakeWord `thunderbolt.onnx`
|
||||||
detection + `NearMissLog` for threshold tuning — see below), `stt.py`
|
detection + `NearMissLog` for threshold tuning — see below), `stt.py` +
|
||||||
(Deepgram), `tts.py` (ElevenLabs, streaming by default — `stream_pcm()` +
|
`stt_stream.py` (the server's own `/desk/stt` websocket relay — no local
|
||||||
`play_stream()` start playback on the first chunk; `chunks_to_int16()`
|
Deepgram account; see below), `tts.py` (the server's own `/desk/tts`,
|
||||||
carries odd bytes across HTTP chunk boundaries, without which everything
|
streaming by default — `stream_pcm()` + `play_stream()` start playback on
|
||||||
after the first split sample plays as static — falling back to whole-clip
|
the first chunk; `chunks_to_int16()` carries odd bytes across HTTP chunk
|
||||||
PCM then offline `pyttsx3`; every entry point takes an optional `voice_id`
|
boundaries, without which everything after the first split sample plays as
|
||||||
overriding `ELEVENLABS_VOICE_ID`, and `model_for()` picks the multilingual
|
static — falling back to whole-clip PCM then offline `pyttsx3`; every entry
|
||||||
model whenever there's an override or non-ASCII text, since the default
|
point takes an optional `voice_id` overriding `ELEVENLABS_VOICE_ID` — which
|
||||||
`eleven_flash_v2` is English-only and would read either as garbled
|
model to synthesize with is the server's call now, not this client's;
|
||||||
phonetic English rather than failing), `barge_in.py` (two detectors behind one
|
see `synthesize_dialogue()` further down for the one path that's still
|
||||||
|
ElevenLabs-direct), `barge_in.py` (two detectors behind one
|
||||||
`reset()`/`check()` shape, chosen by `BARGE_IN_MODE` via `make_detector`:
|
`reset()`/`check()` shape, chosen by `BARGE_IN_MODE` via `make_detector`:
|
||||||
**wake** (default) scores every frame with the same openWakeWord model the
|
**wake** (default) scores every frame with the same openWakeWord model the
|
||||||
idle listener uses, so only the wake phrase cuts playback; **energy** is the
|
idle listener uses, so only the wake phrase cuts playback; **energy** is the
|
||||||
@@ -290,18 +295,30 @@ logs a missing-config message and exits its thread instead of starting.
|
|||||||
- **`hotkey.py`** — global push-to-talk via `pynput`; soft-fails with a logged
|
- **`hotkey.py`** — global push-to-talk via `pynput`; soft-fails with a logged
|
||||||
reason (Wayland, missing package, macOS permissions) since the wake word is
|
reason (Wayland, missing package, macOS permissions) since the wake word is
|
||||||
the primary trigger.
|
the primary trigger.
|
||||||
- **`audio/stt_stream.py`** — streaming speech-to-text. The one-shot path waits for
|
- **`audio/stt_stream.py`** / **`audio/stt.py`** — speech-to-text via the
|
||||||
the utterance to end, uploads the whole WAV, then waits again; that second wait is
|
server's `/desk/stt` websocket relay (audio up, Deepgram's JSON messages
|
||||||
dead time that grows with how long you spoke. Deepgram's live websocket removes it:
|
down untouched — the same relay the Android app uses; no local Deepgram
|
||||||
`record_utterance(on_frame=...)` hands each captured frame to a
|
account or API key). `stt_stream.py` is the opportunistic optimisation:
|
||||||
`StreamingTranscriber`, so by the time the VAD decides you stopped the transcript is
|
the one-shot path waits for the utterance to end, then sends the whole
|
||||||
essentially already there. Three deliberate limits: `open()` returning **None is an
|
clip and waits again; that second wait is dead time that grows with how
|
||||||
ordinary outcome** (no websocket-client, no network, no key) because the full audio is
|
long you spoke, and streaming removes it — `record_utterance(on_frame=...)`
|
||||||
still buffered and `controller._transcribe` just falls back; the **local VAD still
|
hands each captured frame to a `StreamingTranscriber` as it's captured, so
|
||||||
decides when you stopped** rather than Deepgram's endpointing, since barge-in,
|
by the time the VAD decides you stopped the transcript is essentially
|
||||||
follow-up listening and the grace period are all built on it and coupling them to the
|
already there. `stt.py`'s `transcribe()` is the *guaranteed* fallback for
|
||||||
network is not a first-pass change; and the socket is **per-utterance**, because
|
when that didn't produce anything: since there's no separate REST endpoint
|
||||||
holding one open across an idle pet bills for silence and dies on the first blip.
|
server-side, it opens the exact same relay via `stt_stream.connect()` and
|
||||||
|
just feeds the whole buffered utterance in one go — deliberately
|
||||||
|
*unconditional*, not gated by `STT_STREAMING`/`available()` the way the
|
||||||
|
opportunistic path is, since there's nothing left to fall back to if that
|
||||||
|
connection fails. Three more deliberate limits on the streaming half:
|
||||||
|
`open()` returning **None is an ordinary outcome** (no websocket-client, no
|
||||||
|
network, streaming turned off) because the full audio is still buffered
|
||||||
|
and `controller._transcribe` just falls back to `stt.transcribe()`; the
|
||||||
|
**local VAD still decides when you stopped** rather than Deepgram's
|
||||||
|
endpointing, since barge-in, follow-up listening and the grace period are
|
||||||
|
all built on it and coupling them to the network is not a first-pass
|
||||||
|
change; and the socket is **per-utterance**, because holding one open
|
||||||
|
across an idle pet bills for silence and dies on the first blip.
|
||||||
Off: `STT_STREAMING=false`.
|
Off: `STT_STREAMING=false`.
|
||||||
- **Streamed replies** — `server_client.converse_stream()` reads NDJSON from the desk
|
- **Streamed replies** — `server_client.converse_stream()` reads NDJSON from the desk
|
||||||
API's `/desk/converse_stream` and speaks each sentence as it arrives
|
API's `/desk/converse_stream` and speaks each sentence as it arrives
|
||||||
|
|||||||
@@ -3,16 +3,19 @@
|
|||||||
A little animated pet that lives on your desktop and is just a face on top
|
A little animated pet that lives on your desktop and is just a face on top
|
||||||
of your Bolt server — same brain, memory, tools, and persona as Discord
|
of your Bolt server — same brain, memory, tools, and persona as Discord
|
||||||
chat and the Linux desk client. It talks to `ai/desk_api.py` on the server
|
chat and the Linux desk client. It talks to `ai/desk_api.py` on the server
|
||||||
exactly the way `desk_client/bolt_desk.py` does; this project only adds the
|
exactly the way `desk_client/bolt_desk.py` does — including speech: STT and
|
||||||
on-screen pet and swaps Deepgram/ElevenLabs playback to be cross-platform
|
TTS are the server's own `/desk/stt` and `/desk/tts`, the same endpoints the
|
||||||
(no `mpv`/`ffplay`/`espeak-ng` subprocess calls — pure `sounddevice`).
|
Android app uses, so there's no separate Deepgram or ElevenLabs account to
|
||||||
|
set up for the pet to talk. Playback is cross-platform (no `mpv`/`ffplay`/
|
||||||
|
`espeak-ng` subprocess calls — pure `sounddevice`).
|
||||||
|
|
||||||
```
|
```
|
||||||
mic → wake-phrase spotter ("thunderbolt") / hotkey / click → record utterance
|
mic → wake-phrase spotter ("thunderbolt") / hotkey / click → record utterance
|
||||||
→ Deepgram STT (+ the focused window's title, for "what's this error?")
|
→ server-hosted STT (+ the focused window's title, for "what's this
|
||||||
→ POST /desk/converse on your Bolt server → [server may relay a shell
|
error?") → POST /desk/converse on your Bolt server → [server may relay
|
||||||
command back to run on THIS machine, or a `petctl` command that moves
|
a shell command back to run on THIS machine, or a `petctl` command
|
||||||
or emotes the pet] → reply → ElevenLabs streaming TTS → speakers
|
that moves or emotes the pet] → reply → server-hosted streaming TTS
|
||||||
|
→ speakers
|
||||||
→ shown in a speech bubble + the pet's sprite state (idle/listening/
|
→ shown in a speech bubble + the pet's sprite state (idle/listening/
|
||||||
thinking/talking) updates the whole time
|
thinking/talking) updates the whole time
|
||||||
```
|
```
|
||||||
@@ -36,10 +39,13 @@ all, so it can be copied anywhere and configured with its own `.env`.
|
|||||||
- `BOLT_SERVER_URL` + `DESK_API_KEY` — same as `desk_client/.env` on the
|
- `BOLT_SERVER_URL` + `DESK_API_KEY` — same as `desk_client/.env` on the
|
||||||
server side. Use the server's master `DESK_API_KEY`, or mint yourself a
|
server side. Use the server's master `DESK_API_KEY`, or mint yourself a
|
||||||
personal one via the desk-only `api_key_generate` marker (see the main
|
personal one via the desk-only `api_key_generate` marker (see the main
|
||||||
repo's `CLAUDE.md` → "Per-user API keys").
|
repo's `CLAUDE.md` → "Per-user API keys"). That's it for STT — no
|
||||||
- `DEEPGRAM_API_KEY` for STT.
|
Deepgram account needed, it goes through the server's own `/desk/stt`.
|
||||||
- `ELEVENLABS_API_KEY` + `ELEVENLABS_VOICE_ID` for TTS (optional — falls
|
- `ELEVENLABS_VOICE_ID` for TTS (optional — falls back to offline TTS via
|
||||||
back to offline TTS via `pyttsx3` if omitted or if a request fails).
|
`pyttsx3` if omitted or if the server call fails). No local ElevenLabs
|
||||||
|
API key needed for this either; `ELEVENLABS_API_KEY` is only for the
|
||||||
|
multi-voice `dialoguectl` scenes further down, the one feature the
|
||||||
|
server has no endpoint for.
|
||||||
3. Run it:
|
3. Run it:
|
||||||
- macOS/Linux: `./run.sh`
|
- macOS/Linux: `./run.sh`
|
||||||
- Windows: `run.bat`
|
- Windows: `run.bat`
|
||||||
@@ -96,9 +102,10 @@ Ask for a different voice — "use a clearer voice", "talk like a pirate", "say
|
|||||||
that in Japanese" — and Bolt searches the ElevenLabs voice library on the
|
that in Japanese" — and Bolt searches the ElevenLabs voice library on the
|
||||||
server, picks one, and tags his reply with it (`speak_as`); the pet is what
|
server, picks one, and tags his reply with it (`speak_as`); the pet is what
|
||||||
actually speaks in it. A Voice Library pick is added to your ElevenLabs
|
actually speaks in it. A Voice Library pick is added to your ElevenLabs
|
||||||
account automatically the first time it's used, and non-English replies (or
|
account automatically the first time it's used. The server also synthesizes
|
||||||
any picked voice) go through `ELEVENLABS_MULTILINGUAL_MODEL_ID` rather than
|
with one fixed, multilingual-capable model for every request now (not a
|
||||||
the English-only `eleven_flash_v2` default.
|
flash/multilingual switch per reply) — nothing to configure on this side,
|
||||||
|
and non-English text or a picked voice no longer needs special-casing here.
|
||||||
|
|
||||||
The new voice **stays on** for the rest of the conversation, because the
|
The new voice **stays on** for the rest of the conversation, because the
|
||||||
server tags a single reply and doesn't remember which voice it chose — so
|
server tags a single reply and doesn't remember which voice it chose — so
|
||||||
@@ -188,9 +195,9 @@ bolt_pet/
|
|||||||
audio/
|
audio/
|
||||||
mic.py input stream + energy-based VAD utterance capture
|
mic.py input stream + energy-based VAD utterance capture
|
||||||
wake_word.py openWakeWord thunderbolt.onnx detection (see above)
|
wake_word.py openWakeWord thunderbolt.onnx detection (see above)
|
||||||
stt.py Deepgram (one-shot)
|
stt.py server /desk/stt relay (one-shot: whole utterance at once)
|
||||||
stt_stream.py Deepgram live websocket — transcribes while you speak
|
stt_stream.py server /desk/stt live websocket — transcribes while you speak
|
||||||
tts.py ElevenLabs streaming PCM, offline pyttsx3 fallback,
|
tts.py server /desk/tts streaming PCM, offline pyttsx3 fallback,
|
||||||
plus the loudness envelope that drives the mouth
|
plus the loudness envelope that drives the mouth
|
||||||
barge_in.py "you started talking" detector, to cut playback short
|
barge_in.py "you started talking" detector, to cut playback short
|
||||||
ui/
|
ui/
|
||||||
|
|||||||
+17
-26
@@ -1,14 +1,20 @@
|
|||||||
"""Speech-to-text for the actual query, after the wake word fires.
|
"""Speech-to-text for the actual query, after the wake word fires.
|
||||||
|
|
||||||
Deepgram, same as desk_client/bolt_desk.py.
|
The "guaranteed" fallback for when stt_stream's opportunistic live-feed
|
||||||
|
session didn't produce a transcript (streaming disabled, or the socket
|
||||||
|
never came up). There is no separate one-shot REST endpoint server-side any
|
||||||
|
more — /desk/stt is a websocket relay only — so this connects the exact
|
||||||
|
same way stt_stream.py does and just feeds the whole buffered utterance in
|
||||||
|
one go instead of frame-by-frame as it's captured. That connection is
|
||||||
|
unconditional (not gated by STT_STREAMING, which only controls the
|
||||||
|
opportunistic optimisation), since there is nothing left to fall back to
|
||||||
|
if it fails.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import requests
|
|
||||||
|
|
||||||
from .. import config
|
from .. import config
|
||||||
from .mic import pcm_to_wav_bytes
|
from . import stt_stream
|
||||||
|
|
||||||
|
|
||||||
class SttError(Exception):
|
class SttError(Exception):
|
||||||
@@ -16,27 +22,12 @@ class SttError(Exception):
|
|||||||
|
|
||||||
|
|
||||||
def transcribe(pcm) -> str:
|
def transcribe(pcm) -> str:
|
||||||
if not config.DEEPGRAM_API_KEY:
|
if not config.is_configured():
|
||||||
raise SttError("DEEPGRAM_API_KEY is not set")
|
raise SttError("BOLT_SERVER_URL / DESK_API_KEY not set")
|
||||||
try:
|
try:
|
||||||
response = requests.post(
|
socket = stt_stream.connect()
|
||||||
"https://api.deepgram.com/v1/listen",
|
|
||||||
params={"model": config.DEEPGRAM_MODEL, "language": "en", "smart_format": "true"},
|
|
||||||
headers={
|
|
||||||
"Authorization": f"Token {config.DEEPGRAM_API_KEY}",
|
|
||||||
"Content-Type": "audio/wav",
|
|
||||||
},
|
|
||||||
data=pcm_to_wav_bytes(pcm),
|
|
||||||
timeout=30,
|
|
||||||
)
|
|
||||||
response.raise_for_status()
|
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
raise SttError(f"transcription request failed: {exc}") from exc
|
raise SttError(f"couldn't reach the transcription server: {exc}") from exc
|
||||||
try:
|
session = stt_stream.StreamingTranscriber(socket)
|
||||||
return (
|
session.feed(pcm)
|
||||||
response.json()
|
return session.finish()
|
||||||
.get("results", {}).get("channels", [{}])[0]
|
|
||||||
.get("alternatives", [{}])[0].get("transcript", "")
|
|
||||||
).strip()
|
|
||||||
except Exception as exc:
|
|
||||||
raise SttError(f"couldn't parse transcription response: {exc}") from exc
|
|
||||||
|
|||||||
@@ -1,21 +1,26 @@
|
|||||||
"""Streaming speech-to-text — transcribing *while* you talk, not after.
|
"""Streaming speech-to-text — transcribing *while* you talk, not after.
|
||||||
|
|
||||||
The one-shot path (`stt.transcribe`) waits for the utterance to finish, then
|
The one-shot path (`stt.transcribe`) waits for the utterance to finish, then
|
||||||
uploads the whole WAV and waits again. That second wait is dead time between
|
sends the whole clip and waits again. That second wait is dead time between
|
||||||
you stopping and the pet reacting, and it grows with the length of what you
|
you stopping and the pet reacting, and it grows with the length of what you
|
||||||
said — a thirty-second question costs noticeably more than a five-second one.
|
said — a thirty-second question costs noticeably more than a five-second one.
|
||||||
|
|
||||||
Deepgram's live endpoint removes it: frames go up as they are captured, so by
|
This connects to the Bolt server's own `/desk/stt` — the same websocket relay
|
||||||
the time the VAD decides you have stopped, the transcript is essentially
|
the Android app uses — which forwards audio to Deepgram and Deepgram's JSON
|
||||||
already there. Same model, same account, same accuracy — the difference is
|
messages back untouched. Frames go up as they are captured, so by the time the
|
||||||
purely when the work happens.
|
VAD decides you have stopped, the transcript is essentially already there.
|
||||||
|
There is no local Deepgram account or API key any more; auth is this pet's own
|
||||||
|
`DESK_API_KEY`, same as every other call to the server.
|
||||||
|
|
||||||
Design constraints that shaped this:
|
Design constraints that shaped this:
|
||||||
|
|
||||||
- **Failure must be invisible.** No websocket, no network, a mid-utterance
|
- **Failure must be invisible.** No websocket, no network, a mid-utterance
|
||||||
disconnect — all of it falls back to the one-shot path, which still has the
|
disconnect — all of it falls back to the one-shot path, which still has the
|
||||||
full audio buffered. Streaming is an optimisation, never a dependency, so
|
full audio buffered. Streaming is an optimisation here, never a dependency —
|
||||||
`open()` returning None is an ordinary outcome rather than an error.
|
`available()`/`open()` returning None/False is an ordinary outcome, not an
|
||||||
|
error. `stt.transcribe()` (the *guaranteed* fallback) talks to the exact same
|
||||||
|
server relay via `connect()` directly, bypassing that opportunistic gate,
|
||||||
|
since there is no second, different backend left to fall back to.
|
||||||
- **The VAD still decides when you stopped.** Deepgram has its own endpointing
|
- **The VAD still decides when you stopped.** Deepgram has its own endpointing
|
||||||
and using it would save more, but it would also move a decision the rest of
|
and using it would save more, but it would also move a decision the rest of
|
||||||
the pipeline is built around (barge-in, follow-up listening, the grace
|
the pipeline is built around (barge-in, follow-up listening, the grace
|
||||||
@@ -42,16 +47,14 @@ from .. import config
|
|||||||
|
|
||||||
logger = logging.getLogger("bolt_pet.stt_stream")
|
logger = logging.getLogger("bolt_pet.stt_stream")
|
||||||
|
|
||||||
_ENDPOINT = (
|
|
||||||
"wss://api.deepgram.com/v1/listen"
|
|
||||||
"?encoding=linear16&channels=1&sample_rate={rate}&model={model}"
|
|
||||||
"&language=en&smart_format=true&interim_results=false"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def available() -> bool:
|
def available() -> bool:
|
||||||
"""Whether streaming STT can even be attempted in this install."""
|
"""Whether the streaming (transcribe-while-talking) optimization should
|
||||||
if not config.STT_STREAMING or not config.DEEPGRAM_API_KEY:
|
be attempted opportunistically. Not a gate on transcription itself — the
|
||||||
|
server relay is the only way to transcribe at all now, so
|
||||||
|
stt.transcribe() connects via connect() directly rather than through
|
||||||
|
this, and isn't affected by STT_STREAMING being off."""
|
||||||
|
if not config.STT_STREAMING or not config.is_configured():
|
||||||
return False
|
return False
|
||||||
try:
|
try:
|
||||||
import websocket # noqa: F401 (websocket-client)
|
import websocket # noqa: F401 (websocket-client)
|
||||||
@@ -60,6 +63,33 @@ def available() -> bool:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _connect(sample_rate: int = None):
|
||||||
|
"""Open a websocket to the server's `/desk/stt` relay. Raises on any
|
||||||
|
failure — this is the "no fallback left" connector `stt.transcribe()`
|
||||||
|
uses directly, as well as the default for `StreamingTranscriber.open()`.
|
||||||
|
|
||||||
|
The server chooses the STT model and the endpointing behaviour; this
|
||||||
|
only states the audio format about to be sent, which is fixed by the
|
||||||
|
wake model upstream of it."""
|
||||||
|
import websocket
|
||||||
|
|
||||||
|
rate = sample_rate or config.SAMPLE_RATE
|
||||||
|
url = (
|
||||||
|
config.SERVER_URL.replace("http", "ws", 1)
|
||||||
|
+ "/desk/stt"
|
||||||
|
+ f"?session_id={config.SESSION_ID}&encoding=linear16&sample_rate={rate}"
|
||||||
|
)
|
||||||
|
return websocket.create_connection(
|
||||||
|
url, header=[f"X-Desk-Api-Key: {config.API_KEY}"], timeout=10,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# Public name for external callers (stt.py, tests) — named separately from
|
||||||
|
# the module-private def so StreamingTranscriber.open()'s `connect` parameter
|
||||||
|
# can shadow the bare name locally without losing access to this.
|
||||||
|
connect = _connect
|
||||||
|
|
||||||
|
|
||||||
class StreamingTranscriber:
|
class StreamingTranscriber:
|
||||||
"""One utterance's worth of live transcription.
|
"""One utterance's worth of live transcription.
|
||||||
|
|
||||||
@@ -95,16 +125,7 @@ class StreamingTranscriber:
|
|||||||
return None
|
return None
|
||||||
rate = sample_rate or config.SAMPLE_RATE
|
rate = sample_rate or config.SAMPLE_RATE
|
||||||
try:
|
try:
|
||||||
if connect is not None:
|
socket = connect() if connect is not None else _connect(rate)
|
||||||
socket = connect()
|
|
||||||
else:
|
|
||||||
import websocket
|
|
||||||
|
|
||||||
socket = websocket.create_connection(
|
|
||||||
_ENDPOINT.format(rate=rate, model=config.DEEPGRAM_MODEL),
|
|
||||||
header={"Authorization": f"Token {config.DEEPGRAM_API_KEY}"},
|
|
||||||
timeout=10,
|
|
||||||
)
|
|
||||||
return cls(socket, sample_rate=rate)
|
return cls(socket, sample_rate=rate)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
logger.info("Streaming STT unavailable (%s) — using the one-shot path.", exc)
|
logger.info("Streaming STT unavailable (%s) — using the one-shot path.", exc)
|
||||||
|
|||||||
+33
-37
@@ -1,15 +1,24 @@
|
|||||||
"""Text-to-speech: ElevenLabs, requested as raw PCM so playback is just
|
"""Text-to-speech: the Bolt server's own `/desk/tts` — the same endpoint the
|
||||||
|
Android app streams from — requested as raw PCM so playback is just
|
||||||
sounddevice — no external player binary (mpv/ffplay), unlike
|
sounddevice — no external player binary (mpv/ffplay), unlike
|
||||||
desk_client/bolt_desk.py which shells out because it only targets Linux.
|
desk_client/bolt_desk.py which shells out because it only targets Linux.
|
||||||
|
|
||||||
Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech
|
No local ElevenLabs account needed for this: the server picks the voice
|
||||||
on macOS, espeak on Linux) if ElevenLabs isn't configured or the request
|
(ELEVENLABS_VOICE_ID, or a `speak_as` override) and the synthesis model
|
||||||
fails, so the pet can still talk with zero cloud config.
|
itself, authenticated with this pet's own DESK_API_KEY. Falls back to
|
||||||
|
pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech on macOS,
|
||||||
|
espeak on Linux) if the server call fails, so the pet can still talk even
|
||||||
|
with the server unreachable.
|
||||||
|
|
||||||
Every entry point takes an optional *voice_id* that overrides
|
Every entry point takes an optional *voice_id* that overrides
|
||||||
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
|
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
|
||||||
reply marker reaches the speakers (see controller._apply_voice). The offline
|
reply marker reaches the speakers (see controller._apply_voice). The offline
|
||||||
fallback has no such concept and always sounds like itself.
|
fallback has no such concept and always sounds like itself.
|
||||||
|
|
||||||
|
`synthesize_dialogue()` below is the one exception: multi-voice
|
||||||
|
`dialoguectl` scenes have no server endpoint, so that one call still goes
|
||||||
|
to ElevenLabs' Text to Dialogue API directly and still needs
|
||||||
|
ELEVENLABS_API_KEY — see dialogue.py.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -33,19 +42,8 @@ def voice_for(voice_id: Optional[str] = None) -> str:
|
|||||||
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
|
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
|
||||||
|
|
||||||
|
|
||||||
def model_for(text: str, voice_id: Optional[str] = None) -> str:
|
def _headers() -> dict:
|
||||||
"""Which ElevenLabs model to synthesize with.
|
return {"X-Desk-Api-Key": config.API_KEY}
|
||||||
|
|
||||||
The default (`eleven_flash_v2`) is English-only, and both things that
|
|
||||||
reach this branch mean the reply probably isn't English: a voice the
|
|
||||||
server picked mid-conversation is nearly always about a language or an
|
|
||||||
accent, and non-ASCII text can't be English at all. Rendering either one
|
|
||||||
through the English model gets you a mangled phonetic reading rather
|
|
||||||
than a failure, which is worse — so those go through the multilingual
|
|
||||||
model instead."""
|
|
||||||
if (voice_id or "").strip() or not text.isascii():
|
|
||||||
return config.ELEVENLABS_MULTILINGUAL_MODEL_ID
|
|
||||||
return config.ELEVENLABS_MODEL_ID
|
|
||||||
|
|
||||||
|
|
||||||
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
|
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
|
||||||
@@ -53,48 +51,46 @@ def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarra
|
|||||||
callers should fall back to speak_offline() rather than treating this
|
callers should fall back to speak_offline() rather than treating this
|
||||||
as fatal."""
|
as fatal."""
|
||||||
voice = voice_for(voice_id)
|
voice = voice_for(voice_id)
|
||||||
if not (config.ELEVENLABS_API_KEY and voice):
|
if not (config.is_configured() and voice):
|
||||||
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
raise TtsError("BOLT_SERVER_URL / DESK_API_KEY / ELEVENLABS_VOICE_ID not set")
|
||||||
try:
|
try:
|
||||||
response = requests.post(
|
response = requests.post(
|
||||||
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}",
|
f"{config.SERVER_URL}/desk/tts",
|
||||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
headers=_headers(),
|
||||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
json={"session_id": config.SESSION_ID, "text": text, "voice_id": voice},
|
||||||
json={"text": text, "model_id": model_for(text, voice_id)},
|
|
||||||
timeout=60,
|
timeout=60,
|
||||||
)
|
)
|
||||||
response.raise_for_status()
|
response.raise_for_status()
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
raise TtsError(f"ElevenLabs request failed: {exc}") from exc
|
raise TtsError(f"server tts request failed: {exc}") from exc
|
||||||
pcm = np.frombuffer(response.content, dtype=np.int16)
|
pcm = np.frombuffer(response.content, dtype=np.int16)
|
||||||
if pcm.size == 0:
|
if pcm.size == 0:
|
||||||
raise TtsError("ElevenLabs returned no audio")
|
raise TtsError("server returned no audio")
|
||||||
return pcm, config.TTS_SAMPLE_RATE
|
return pcm, config.TTS_SAMPLE_RATE
|
||||||
|
|
||||||
|
|
||||||
def stream_pcm(
|
def stream_pcm(
|
||||||
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
|
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
|
||||||
) -> Iterator[np.ndarray]:
|
) -> Iterator[np.ndarray]:
|
||||||
"""Same audio as synthesize_pcm(), but yielded as it arrives from
|
"""Same audio as synthesize_pcm(), but yielded as it arrives from the
|
||||||
ElevenLabs' /stream endpoint so playback can start on the first chunk
|
server so playback can start on the first chunk instead of after the
|
||||||
(~300ms) instead of after the whole clip is synthesized. Raises TtsError
|
whole clip is synthesized. Raises TtsError before yielding anything if
|
||||||
before yielding anything if the request itself fails, so callers can fall
|
the request itself fails, so callers can fall back cleanly; a mid-stream
|
||||||
back cleanly; a mid-stream failure just ends the generator."""
|
failure just ends the generator."""
|
||||||
voice = voice_for(voice_id)
|
voice = voice_for(voice_id)
|
||||||
if not (config.ELEVENLABS_API_KEY and voice):
|
if not (config.is_configured() and voice):
|
||||||
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
raise TtsError("BOLT_SERVER_URL / DESK_API_KEY / ELEVENLABS_VOICE_ID not set")
|
||||||
try:
|
try:
|
||||||
response = requests.post(
|
response = requests.post(
|
||||||
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}/stream",
|
f"{config.SERVER_URL}/desk/tts",
|
||||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
headers=_headers(),
|
||||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
json={"session_id": config.SESSION_ID, "text": text, "voice_id": voice},
|
||||||
json={"text": text, "model_id": model_for(text, voice_id)},
|
|
||||||
timeout=60,
|
timeout=60,
|
||||||
stream=True,
|
stream=True,
|
||||||
)
|
)
|
||||||
response.raise_for_status()
|
response.raise_for_status()
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
raise TtsError(f"ElevenLabs stream request failed: {exc}") from exc
|
raise TtsError(f"server tts stream request failed: {exc}") from exc
|
||||||
return chunks_to_int16(response.iter_content(chunk_size=chunk_bytes))
|
return chunks_to_int16(response.iter_content(chunk_size=chunk_bytes))
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+29
-19
@@ -54,31 +54,41 @@ WAKE_WORD_THRESHOLD = float(os.environ.get("WAKE_WORD_THRESHOLD", "0.5"))
|
|||||||
# heartbeat poll in controller.py during quiet stretches with no wake word.
|
# heartbeat poll in controller.py during quiet stretches with no wake word.
|
||||||
WAKE_CHECK_INTERVAL_SECONDS = float(os.environ.get("WAKE_CHECK_INTERVAL_SECONDS", "1.2"))
|
WAKE_CHECK_INTERVAL_SECONDS = float(os.environ.get("WAKE_CHECK_INTERVAL_SECONDS", "1.2"))
|
||||||
|
|
||||||
# ── STT (Deepgram, same as bolt_desk.py) ────────────────────────────────────
|
# ── STT (server-hosted, same /desk/stt relay the Android app uses) ─────────
|
||||||
|
# No local Deepgram account needed any more: audio/stt_stream.py opens a
|
||||||
|
# websocket to this pet's own BOLT_SERVER_URL/DESK_API_KEY, which the server
|
||||||
|
# relays to Deepgram and meters it against the same credit ledger as a chat turn.
|
||||||
|
# There is no separate one-shot REST path server-side, so audio/stt.py's
|
||||||
|
# "guaranteed" fallback uses this exact same connection too — just fed the
|
||||||
|
# whole utterance at once instead of frame-by-frame.
|
||||||
|
|
||||||
DEEPGRAM_API_KEY = os.environ.get("DEEPGRAM_API_KEY", "")
|
# Transcribe *while* you talk instead of waiting for the utterance to end:
|
||||||
DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3")
|
# frames go up as they are captured, so the transcript is ready the moment the
|
||||||
# Transcribe *while* you talk instead of uploading the finished clip: frames go
|
# VAD says you stopped. Needs `websocket-client` (in requirements.txt); when
|
||||||
# up as they are captured, so the transcript is ready the moment the VAD says
|
# it can't connect the one-shot path (audio/stt.py) still tries the same
|
||||||
# you stopped. Needs `websocket-client`; falls back to the one-shot upload
|
# server relay itself, so turning this off only costs latency, not the
|
||||||
# whenever it can't connect, so turning it on can only help.
|
# ability to transcribe at all.
|
||||||
STT_STREAMING = os.environ.get("STT_STREAMING", "true").lower() in ("1", "true", "yes", "on")
|
STT_STREAMING = os.environ.get("STT_STREAMING", "true").lower() in ("1", "true", "yes", "on")
|
||||||
|
|
||||||
# ── TTS (ElevenLabs, requested as raw PCM so playback needs no external
|
# ── TTS (server-hosted, same /desk/tts endpoint the Android app uses) ──────
|
||||||
# player binary — cross-platform via sounddevice instead of shelling out to
|
# No local ElevenLabs account needed for the normal reply voice any more —
|
||||||
# mpv/ffplay like bolt_desk.py does on Linux) ───────────────────────────────
|
# audio/tts.py posts to this pet's own BOLT_SERVER_URL/DESK_API_KEY and gets
|
||||||
|
# back raw 16 kHz mono PCM16, same as the phone. Falls back to offline
|
||||||
|
# pyttsx3 if the server call fails.
|
||||||
|
#
|
||||||
|
# ELEVENLABS_API_KEY is still read directly by this client for exactly one
|
||||||
|
# feature the server has no endpoint for: multi-voice `dialoguectl` scenes
|
||||||
|
# (audio/tts.synthesize_dialogue, see dialogue.py) — leave it blank and
|
||||||
|
# everything except that one feature works with zero local API keys.
|
||||||
|
|
||||||
ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "")
|
ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "")
|
||||||
|
# Which voice to ask the server for — an ElevenLabs voice id (or a "vb:"-
|
||||||
|
# prefixed cloned voice, if this desk key owns one). The server picks the
|
||||||
|
# synthesis model itself now; there is nothing left for this client to choose.
|
||||||
ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "")
|
ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "")
|
||||||
ELEVENLABS_MODEL_ID = os.environ.get("ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
# Fixed by the server (ai/desk_media.py's PCM_SAMPLE_RATE) — not a free
|
||||||
# eleven_flash_v2 is English-only, and the two cases that swap the voice
|
# tunable any more, but still an env override in case that ever changes.
|
||||||
# (server-picked `speak_as`, or a reply with non-ASCII in it) are usually
|
TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "16000"))
|
||||||
# exactly the cases where the reply isn't English — see tts.model_for().
|
|
||||||
ELEVENLABS_MULTILINGUAL_MODEL_ID = os.environ.get(
|
|
||||||
"ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5"
|
|
||||||
)
|
|
||||||
# ElevenLabs PCM output formats are named pcm_<sample_rate>.
|
|
||||||
TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "24000"))
|
|
||||||
|
|
||||||
# Does a voice the server picks (its speak_as marker — "talk like a pirate",
|
# Does a voice the server picks (its speak_as marker — "talk like a pirate",
|
||||||
# "say that in Japanese") stay on for later replies, or last one reply only?
|
# "say that in Japanese") stay on for later replies, or last one reply only?
|
||||||
|
|||||||
+28
-15
@@ -137,29 +137,42 @@ def check_wake_model() -> Check:
|
|||||||
|
|
||||||
|
|
||||||
def check_stt() -> Check:
|
def check_stt() -> Check:
|
||||||
if not config.DEEPGRAM_API_KEY:
|
"""STT is a websocket relay to the server (/desk/stt) — no local Deepgram
|
||||||
return Check("speech-to-text", FAIL, "no DEEPGRAM_API_KEY",
|
account, but websocket-client is now load-bearing for transcription to
|
||||||
"set it in .env — nothing you say can be transcribed without it")
|
work at all, not just the streaming optimisation (there's no separate
|
||||||
if config.STT_STREAMING and not _module("websocket"):
|
REST fallback any more)."""
|
||||||
return Check("speech-to-text", WARN, "streaming on, but websocket-client is missing",
|
if not config.is_configured():
|
||||||
"pip install websocket-client — it falls back to one-shot uploads")
|
return Check("speech-to-text", FAIL, "no BOLT_SERVER_URL/DESK_API_KEY",
|
||||||
mode = "streaming" if config.STT_STREAMING else "one-shot"
|
"set them in .env — nothing you say can be transcribed without the server")
|
||||||
return Check("speech-to-text", OK, f"Deepgram {config.DEEPGRAM_MODEL}, {mode}")
|
if not _module("websocket"):
|
||||||
|
return Check("speech-to-text", FAIL, "websocket-client is missing",
|
||||||
|
"pip install -r requirements.txt — /desk/stt is a websocket relay "
|
||||||
|
"with no REST fallback")
|
||||||
|
mode = "streaming" if config.STT_STREAMING else "one-shot (still via the server relay)"
|
||||||
|
return Check("speech-to-text", OK, f"server relay, {mode}")
|
||||||
|
|
||||||
|
|
||||||
def check_tts() -> Check:
|
def check_tts() -> Check:
|
||||||
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
|
"""TTS is the server's /desk/tts — no local ElevenLabs account needed for
|
||||||
if _module("pyttsx3"):
|
the normal reply voice, just a voice id for it to request."""
|
||||||
return Check("text-to-speech", WARN, "no ElevenLabs key/voice — offline voice only",
|
if config.is_configured() and config.ELEVENLABS_VOICE_ID:
|
||||||
"set ELEVENLABS_API_KEY and ELEVENLABS_VOICE_ID for the real voice")
|
return Check("text-to-speech", OK,
|
||||||
return Check("text-to-speech", FAIL, "no ElevenLabs config and no pyttsx3 fallback")
|
f"server relay, voice …{config.ELEVENLABS_VOICE_ID[-6:]}")
|
||||||
return Check("text-to-speech", OK,
|
if _module("pyttsx3"):
|
||||||
f"ElevenLabs {config.ELEVENLABS_MODEL_ID}, voice …{config.ELEVENLABS_VOICE_ID[-6:]}")
|
return Check("text-to-speech", WARN, "server/voice not configured — offline voice only",
|
||||||
|
"set BOLT_SERVER_URL/DESK_API_KEY and ELEVENLABS_VOICE_ID for the real voice")
|
||||||
|
return Check("text-to-speech", FAIL, "no server/voice config and no pyttsx3 fallback",
|
||||||
|
"set BOLT_SERVER_URL/DESK_API_KEY/ELEVENLABS_VOICE_ID, "
|
||||||
|
"or pip install pyttsx3 for an offline voice")
|
||||||
|
|
||||||
|
|
||||||
def check_dialogue() -> Check:
|
def check_dialogue() -> Check:
|
||||||
if not config.DIALOGUE:
|
if not config.DIALOGUE:
|
||||||
return Check("multi-voice scenes", WARN, "disabled (DIALOGUE=false)")
|
return Check("multi-voice scenes", WARN, "disabled (DIALOGUE=false)")
|
||||||
|
if not config.ELEVENLABS_API_KEY:
|
||||||
|
return Check("multi-voice scenes", WARN, "no ELEVENLABS_API_KEY",
|
||||||
|
"set it in .env — dialogue scenes are the one feature still calling "
|
||||||
|
"ElevenLabs directly, since the server has no equivalent endpoint")
|
||||||
from . import dialogue
|
from . import dialogue
|
||||||
|
|
||||||
cast = dialogue.parse_voice_map(config.DIALOGUE_VOICES)
|
cast = dialogue.parse_voice_map(config.DIALOGUE_VOICES)
|
||||||
|
|||||||
+5
-3
@@ -8,8 +8,10 @@ numpy>=1.24
|
|||||||
|
|
||||||
# HTTP client to the Bolt desk API
|
# HTTP client to the Bolt desk API
|
||||||
requests>=2.31
|
requests>=2.31
|
||||||
# Streaming speech-to-text (audio/stt_stream.py). Optional in practice: without
|
# Speech-to-text, via the server's /desk/stt websocket relay (audio/stt.py,
|
||||||
# it the pet falls back to uploading the finished clip, exactly as before.
|
# audio/stt_stream.py) — there is no separate REST fallback, so this is now
|
||||||
|
# load-bearing for transcription to work at all, not just for the
|
||||||
|
# transcribe-while-talking optimisation STT_STREAMING controls.
|
||||||
websocket-client>=1.7
|
websocket-client>=1.7
|
||||||
|
|
||||||
# Wake-word detection (local, offline after first run) — runs the
|
# Wake-word detection (local, offline after first run) — runs the
|
||||||
@@ -19,7 +21,7 @@ websocket-client>=1.7
|
|||||||
# under the package's own resources/ dir afterward) — needs internet once.
|
# under the package's own resources/ dir afterward) — needs internet once.
|
||||||
openwakeword
|
openwakeword
|
||||||
|
|
||||||
# Offline TTS fallback if ElevenLabs isn't configured or a request fails.
|
# Offline TTS fallback if the server call isn't configured or fails.
|
||||||
# Uses SAPI5 on Windows, NSSpeechSynthesizer on macOS, espeak on Linux
|
# Uses SAPI5 on Windows, NSSpeechSynthesizer on macOS, espeak on Linux
|
||||||
# (Linux also needs: sudo apt install espeak-ng).
|
# (Linux also needs: sudo apt install espeak-ng).
|
||||||
pyttsx3>=2.90
|
pyttsx3>=2.90
|
||||||
|
|||||||
@@ -20,8 +20,7 @@ from bolt_pet.doctor import FAIL, OK, WARN
|
|||||||
|
|
||||||
def test_every_check_returns_a_verdict_on_a_bare_machine(monkeypatch):
|
def test_every_check_returns_a_verdict_on_a_bare_machine(monkeypatch):
|
||||||
"""Nothing configured, nothing installed — still a full report."""
|
"""Nothing configured, nothing installed — still a full report."""
|
||||||
for name in ("SERVER_URL", "API_KEY", "DEEPGRAM_API_KEY",
|
for name in ("SERVER_URL", "API_KEY", "ELEVENLABS_API_KEY", "ELEVENLABS_VOICE_ID"):
|
||||||
"ELEVENLABS_API_KEY", "ELEVENLABS_VOICE_ID"):
|
|
||||||
monkeypatch.setattr(config, name, "")
|
monkeypatch.setattr(config, name, "")
|
||||||
monkeypatch.setattr(doctor, "_module", lambda _n: False)
|
monkeypatch.setattr(doctor, "_module", lambda _n: False)
|
||||||
|
|
||||||
@@ -71,7 +70,7 @@ def test_a_configured_server_passes_without_being_contacted(monkeypatch):
|
|||||||
def test_every_problem_comes_with_something_to_do_about_it(monkeypatch):
|
def test_every_problem_comes_with_something_to_do_about_it(monkeypatch):
|
||||||
""""screen reading: warn" is useless on its own; "apt install tesseract-ocr"
|
""""screen reading: warn" is useless on its own; "apt install tesseract-ocr"
|
||||||
is the entire point of the tool."""
|
is the entire point of the tool."""
|
||||||
for name in ("SERVER_URL", "API_KEY", "DEEPGRAM_API_KEY"):
|
for name in ("SERVER_URL", "API_KEY"):
|
||||||
monkeypatch.setattr(config, name, "")
|
monkeypatch.setattr(config, name, "")
|
||||||
monkeypatch.setattr(doctor, "_module", lambda _n: False)
|
monkeypatch.setattr(doctor, "_module", lambda _n: False)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,103 @@
|
|||||||
|
"""stt.transcribe() — the "guaranteed" one-shot fallback.
|
||||||
|
|
||||||
|
There is no separate REST endpoint server-side any more: /desk/stt is a
|
||||||
|
websocket relay only, so this connects the exact same way stt_stream.py's
|
||||||
|
opportunistic streaming path does (via stt_stream.connect(), unconditionally
|
||||||
|
— not gated by STT_STREAMING, since there's nothing left to fall back to).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
from bolt_pet.audio import stt, stt_stream
|
||||||
|
|
||||||
|
|
||||||
|
class FakeSocket:
|
||||||
|
def __init__(self, messages=(), fail_to_connect=False):
|
||||||
|
self.sent = []
|
||||||
|
self.closed = False
|
||||||
|
self._fail_to_connect = fail_to_connect
|
||||||
|
self._messages = list(messages)
|
||||||
|
|
||||||
|
def send_binary(self, data):
|
||||||
|
self.sent.append(data)
|
||||||
|
|
||||||
|
def send(self, text):
|
||||||
|
self.sent.append(text)
|
||||||
|
|
||||||
|
def recv(self):
|
||||||
|
if self._messages:
|
||||||
|
return self._messages.pop(0)
|
||||||
|
time.sleep(0.01)
|
||||||
|
raise ConnectionError("closed")
|
||||||
|
|
||||||
|
def close(self):
|
||||||
|
self.closed = True
|
||||||
|
|
||||||
|
|
||||||
|
def _results(transcript, is_final=True):
|
||||||
|
return json.dumps({
|
||||||
|
"type": "Results", "is_final": is_final,
|
||||||
|
"channel": {"alternatives": [{"transcript": transcript}]},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def _configured(monkeypatch):
|
||||||
|
monkeypatch.setattr(stt.config, "SERVER_URL", "http://test-server:5002")
|
||||||
|
monkeypatch.setattr(stt.config, "API_KEY", "test-key")
|
||||||
|
|
||||||
|
|
||||||
|
def test_transcribe_raises_when_not_configured(monkeypatch):
|
||||||
|
monkeypatch.setattr(stt.config, "API_KEY", "")
|
||||||
|
with pytest.raises(stt.SttError, match="BOLT_SERVER_URL"):
|
||||||
|
stt.transcribe(np.zeros(320, dtype=np.int16))
|
||||||
|
|
||||||
|
|
||||||
|
def test_transcribe_connects_and_feeds_the_whole_utterance(monkeypatch):
|
||||||
|
socket = FakeSocket([_results("turn on the lights")])
|
||||||
|
monkeypatch.setattr(stt_stream, "connect", lambda: socket)
|
||||||
|
|
||||||
|
pcm = np.full(3200, 500, dtype=np.int16)
|
||||||
|
text = stt.transcribe(pcm)
|
||||||
|
|
||||||
|
assert text == "turn on the lights"
|
||||||
|
# one binary frame (the whole utterance) plus the close message
|
||||||
|
assert len(socket.sent) == 2
|
||||||
|
assert socket.closed
|
||||||
|
|
||||||
|
|
||||||
|
def test_transcribe_raises_when_the_server_is_unreachable(monkeypatch):
|
||||||
|
def refuse():
|
||||||
|
raise OSError("no route to host")
|
||||||
|
|
||||||
|
monkeypatch.setattr(stt_stream, "connect", refuse)
|
||||||
|
|
||||||
|
with pytest.raises(stt.SttError, match="couldn't reach"):
|
||||||
|
stt.transcribe(np.zeros(320, dtype=np.int16))
|
||||||
|
|
||||||
|
|
||||||
|
def test_transcribe_returns_empty_string_for_silence_not_an_error(monkeypatch):
|
||||||
|
"""No speech recognized is a legitimate outcome, not a failure — callers
|
||||||
|
(controller.py) treat "" as "say nothing" rather than logging an error."""
|
||||||
|
socket = FakeSocket([]) # never says anything back
|
||||||
|
monkeypatch.setattr(stt_stream, "connect", lambda: socket)
|
||||||
|
|
||||||
|
assert stt.transcribe(np.zeros(320, dtype=np.int16)) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_transcribe_is_unaffected_by_stt_streaming_being_off(monkeypatch):
|
||||||
|
"""The opportunistic accelerator and the guaranteed fallback share a
|
||||||
|
connector, but STT_STREAMING must only gate the former."""
|
||||||
|
monkeypatch.setattr(stt.config, "STT_STREAMING", False)
|
||||||
|
socket = FakeSocket([_results("still works")])
|
||||||
|
monkeypatch.setattr(stt_stream, "connect", lambda: socket)
|
||||||
|
|
||||||
|
assert stt.transcribe(np.zeros(320, dtype=np.int16)) == "still works"
|
||||||
@@ -130,16 +130,67 @@ def test_open_returns_a_session_when_it_can():
|
|||||||
assert session.finish() == "hi"
|
assert session.finish() == "hi"
|
||||||
|
|
||||||
|
|
||||||
def test_streaming_is_off_without_the_switch_or_a_key(monkeypatch):
|
def test_streaming_is_off_without_the_switch_or_server_config(monkeypatch):
|
||||||
from bolt_pet.audio import stt_stream
|
from bolt_pet.audio import stt_stream
|
||||||
|
|
||||||
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", False)
|
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", False)
|
||||||
|
monkeypatch.setattr(stt_stream.config, "SERVER_URL", "http://test-server:5002")
|
||||||
|
monkeypatch.setattr(stt_stream.config, "API_KEY", "test-key")
|
||||||
assert stt_stream.available() is False
|
assert stt_stream.available() is False
|
||||||
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", True)
|
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", True)
|
||||||
monkeypatch.setattr(stt_stream.config, "DEEPGRAM_API_KEY", "")
|
assert stt_stream.available() is True
|
||||||
|
monkeypatch.setattr(stt_stream.config, "API_KEY", "")
|
||||||
assert stt_stream.available() is False
|
assert stt_stream.available() is False
|
||||||
|
|
||||||
|
|
||||||
|
# ── connecting to the server relay ──────────────────────────────────────────
|
||||||
|
|
||||||
|
def test_connect_builds_the_server_relay_url(monkeypatch):
|
||||||
|
"""No local Deepgram account any more — the pet connects to its own
|
||||||
|
server's /desk/stt, authenticated with its own desk key."""
|
||||||
|
from bolt_pet.audio import stt_stream
|
||||||
|
|
||||||
|
monkeypatch.setattr(stt_stream.config, "SERVER_URL", "http://my-server:5002")
|
||||||
|
monkeypatch.setattr(stt_stream.config, "API_KEY", "my-desk-key")
|
||||||
|
monkeypatch.setattr(stt_stream.config, "SESSION_ID", "pet-test")
|
||||||
|
|
||||||
|
captured = {}
|
||||||
|
|
||||||
|
class _FakeWebsocketModule:
|
||||||
|
@staticmethod
|
||||||
|
def create_connection(url, header=None, timeout=None):
|
||||||
|
captured["url"] = url
|
||||||
|
captured["header"] = header
|
||||||
|
return "a-socket"
|
||||||
|
|
||||||
|
monkeypatch.setitem(sys.modules, "websocket", _FakeWebsocketModule())
|
||||||
|
|
||||||
|
result = stt_stream.connect(sample_rate=16000)
|
||||||
|
|
||||||
|
assert result == "a-socket"
|
||||||
|
assert captured["url"] == (
|
||||||
|
"ws://my-server:5002/desk/stt?session_id=pet-test"
|
||||||
|
"&encoding=linear16&sample_rate=16000"
|
||||||
|
)
|
||||||
|
assert captured["header"] == ["X-Desk-Api-Key: my-desk-key"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_open_uses_connect_by_default(monkeypatch):
|
||||||
|
"""StreamingTranscriber.open() with no injected connect() goes through
|
||||||
|
the real server-relay connector."""
|
||||||
|
from bolt_pet.audio import stt_stream
|
||||||
|
|
||||||
|
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", True)
|
||||||
|
monkeypatch.setattr(stt_stream.config, "SERVER_URL", "http://my-server:5002")
|
||||||
|
monkeypatch.setattr(stt_stream.config, "API_KEY", "my-desk-key")
|
||||||
|
monkeypatch.setattr(stt_stream, "_connect", lambda rate: FakeSocket([_results("hi")]))
|
||||||
|
|
||||||
|
session = stt_stream.StreamingTranscriber.open()
|
||||||
|
|
||||||
|
assert session is not None
|
||||||
|
assert session.finish() == "hi"
|
||||||
|
|
||||||
|
|
||||||
# ── the capture hook ────────────────────────────────────────────────────────
|
# ── the capture hook ────────────────────────────────────────────────────────
|
||||||
|
|
||||||
class _Stream:
|
class _Stream:
|
||||||
|
|||||||
+81
-13
@@ -3,13 +3,16 @@ no network, no audio device, no ONNX model."""
|
|||||||
|
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
import pytest
|
||||||
|
|
||||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
from bolt_pet import config as tts_config
|
from bolt_pet import config as tts_config
|
||||||
from bolt_pet.audio.tts import chunks_to_int16, model_for, voice_for
|
from bolt_pet.audio import tts
|
||||||
|
from bolt_pet.audio.tts import TtsError, chunks_to_int16, voice_for
|
||||||
from bolt_pet.audio.wake_word import NearMissLog
|
from bolt_pet.audio.wake_word import NearMissLog
|
||||||
|
|
||||||
|
|
||||||
@@ -83,7 +86,7 @@ def test_clear_resets_peak_and_entries():
|
|||||||
assert log.entries() == [] and log.peak == 0.0
|
assert log.entries() == [] and log.peak == 0.0
|
||||||
|
|
||||||
|
|
||||||
# ── voice / model selection (server speak_as) ───────────────────────────────
|
# ── voice selection (server speak_as) ───────────────────────────────────────
|
||||||
|
|
||||||
def test_the_override_voice_wins_over_the_configured_one(monkeypatch):
|
def test_the_override_voice_wins_over_the_configured_one(monkeypatch):
|
||||||
monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "DEFAULT")
|
monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "DEFAULT")
|
||||||
@@ -92,16 +95,81 @@ def test_the_override_voice_wins_over_the_configured_one(monkeypatch):
|
|||||||
assert voice_for(None) == "DEFAULT"
|
assert voice_for(None) == "DEFAULT"
|
||||||
|
|
||||||
|
|
||||||
def test_english_replies_in_the_default_voice_use_the_default_model(monkeypatch):
|
# ── server-hosted synthesis (/desk/tts) ─────────────────────────────────────
|
||||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
# No local ElevenLabs account: both the whole-clip and streaming paths post
|
||||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
|
# to this pet's own server, same as server_client.py's other endpoints.
|
||||||
assert model_for("all good here", None) == "eleven_flash_v2"
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def _configured(monkeypatch):
|
||||||
|
monkeypatch.setattr(tts_config, "SERVER_URL", "http://test-server:5002")
|
||||||
|
monkeypatch.setattr(tts_config, "API_KEY", "test-key")
|
||||||
|
monkeypatch.setattr(tts_config, "SESSION_ID", "pet-test")
|
||||||
|
monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "default-voice")
|
||||||
|
|
||||||
|
|
||||||
def test_a_picked_voice_or_non_english_text_uses_the_multilingual_model(monkeypatch):
|
def _mock_response(content=b"", ok=True):
|
||||||
# eleven_flash_v2 is English-only: it would read either of these as
|
resp = MagicMock()
|
||||||
# mangled phonetic English rather than failing outright.
|
resp.content = content
|
||||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
resp.raise_for_status = MagicMock() if ok else MagicMock(side_effect=Exception("boom"))
|
||||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
|
return resp
|
||||||
assert model_for("all good here", "VOICE1") == "eleven_flash_v2_5"
|
|
||||||
assert model_for("こんにちは", None) == "eleven_flash_v2_5"
|
|
||||||
|
def test_synthesize_pcm_posts_to_the_servers_tts_endpoint():
|
||||||
|
pcm_bytes = np.array([1, 2, 3], dtype=np.int16).tobytes()
|
||||||
|
with patch.object(tts.requests, "post") as post:
|
||||||
|
post.return_value = _mock_response(pcm_bytes)
|
||||||
|
pcm, rate = tts.synthesize_pcm("hello there")
|
||||||
|
post.assert_called_once()
|
||||||
|
args, kwargs = post.call_args
|
||||||
|
assert args[0] == "http://test-server:5002/desk/tts"
|
||||||
|
assert kwargs["headers"] == {"X-Desk-Api-Key": "test-key"}
|
||||||
|
assert kwargs["json"] == {
|
||||||
|
"session_id": "pet-test", "text": "hello there", "voice_id": "default-voice",
|
||||||
|
}
|
||||||
|
assert pcm.tolist() == [1, 2, 3]
|
||||||
|
assert rate == tts_config.TTS_SAMPLE_RATE
|
||||||
|
|
||||||
|
|
||||||
|
def test_synthesize_pcm_uses_the_override_voice():
|
||||||
|
with patch.object(tts.requests, "post") as post:
|
||||||
|
post.return_value = _mock_response(b"\x01\x00")
|
||||||
|
tts.synthesize_pcm("hi", voice_id="picked-voice")
|
||||||
|
assert post.call_args.kwargs["json"]["voice_id"] == "picked-voice"
|
||||||
|
|
||||||
|
|
||||||
|
def test_synthesize_pcm_raises_when_not_configured(monkeypatch):
|
||||||
|
monkeypatch.setattr(tts_config, "API_KEY", "")
|
||||||
|
with pytest.raises(TtsError, match="BOLT_SERVER_URL"):
|
||||||
|
tts.synthesize_pcm("hi")
|
||||||
|
|
||||||
|
|
||||||
|
def test_synthesize_pcm_raises_on_empty_audio():
|
||||||
|
with patch.object(tts.requests, "post") as post:
|
||||||
|
post.return_value = _mock_response(b"")
|
||||||
|
with pytest.raises(TtsError, match="no audio"):
|
||||||
|
tts.synthesize_pcm("hi")
|
||||||
|
|
||||||
|
|
||||||
|
def test_synthesize_pcm_raises_when_the_request_fails():
|
||||||
|
with patch.object(tts.requests, "post") as post:
|
||||||
|
post.return_value = _mock_response(b"", ok=False)
|
||||||
|
with pytest.raises(TtsError, match="server tts request failed"):
|
||||||
|
tts.synthesize_pcm("hi")
|
||||||
|
|
||||||
|
|
||||||
|
def test_stream_pcm_posts_to_the_same_endpoint_with_stream_true():
|
||||||
|
with patch.object(tts.requests, "post") as post:
|
||||||
|
response = MagicMock()
|
||||||
|
response.raise_for_status = MagicMock()
|
||||||
|
response.iter_content.return_value = [np.array([4, 5], dtype=np.int16).tobytes()]
|
||||||
|
post.return_value = response
|
||||||
|
chunks = list(tts.stream_pcm("hi"))
|
||||||
|
assert post.call_args.kwargs["stream"] is True
|
||||||
|
assert post.call_args[0][0] == "http://test-server:5002/desk/tts"
|
||||||
|
assert np.concatenate(chunks).tolist() == [4, 5]
|
||||||
|
|
||||||
|
|
||||||
|
def test_stream_pcm_raises_when_not_configured(monkeypatch):
|
||||||
|
monkeypatch.setattr(tts_config, "SERVER_URL", "")
|
||||||
|
with pytest.raises(TtsError, match="BOLT_SERVER_URL"):
|
||||||
|
list(tts.stream_pcm("hi"))
|
||||||
|
|||||||
Reference in New Issue
Block a user