Add text-to-dialogue, self-restart capability, and misc updates

This commit is contained in:
2026-07-30 20:50:41 -06:00
parent 5b49670983
commit 96afc351ac
21 changed files with 1819 additions and 59 deletions
+32 -1
View File
@@ -52,7 +52,38 @@
"Bash(docker exec bolt *)",
"Bash(QT_QPA_PLATFORM=offscreen /root/Documents/bolt-pet/.venv/bin/pytest /home/themajesticmagician/Documents/Bolt-Pet/tests/test_file_ops.py -q)",
"Bash(grep -rn *)",
"Bash(git ls-tree *)"
"Bash(git ls-tree *)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_controller_features.py -q)",
"Read(//home/maji/Documents/tmn-api/**)",
"Bash(timeout 300 .venv/bin/python -m pytest tests/test_desk_voice.py -q)",
"Bash(echo \"exit=$?\")",
"Bash(timeout 300 /home/maji/Documents/tmn-api/.venv/bin/python -m pytest /home/maji/Documents/tmn-api/tests/test_desk_voice.py -q -p no:cacheprovider --rootdir=/home/maji/Documents/tmn-api)",
"Bash(echo \"EXIT=$?\")",
"Bash(/home/maji/Documents/tmn-api/.venv/bin/python -c \"import ast,pathlib; ast.parse\\(pathlib.Path\\('/home/maji/Documents/tmn-api/ai/desk_api.py'\\).read_text\\(\\)\\); print\\('desk_api.py parses OK'\\)\")",
"Bash(ps -eo pid,etime,cmd)",
"Bash(systemctl --user list-units --type=service)",
"Read(//run/user/1000/gvfs/sftp:host=192.168.2.231,user=root/Main/Docker-Compose/TMN-API/tmn-api/ai/**)",
"Bash(findmnt -T /home/maji/Documents/tmn-api -o TARGET,SOURCE,FSTYPE)",
"Bash(echo \"rc=$?\")",
"Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py -q -p no:cacheprovider)",
"Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py -q -p no:cacheprovider)",
"Bash(awk 'NR>=1150 && NR<=1310 && \\(/return / || /def /\\)' ai/agents/default.py)",
"Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python -c ' *)",
"Bash(timeout 900 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py -q -p no:cacheprovider)",
"Bash(timeout 300 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_proactive.py::test_send_trims_swallowed_tool_lines -q -p no:cacheprovider)",
"Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python *)",
"Bash(./tmnvenv/bin/pip install *)",
"Bash(./tmnvenv/bin/python -c \"import pytest,dotenv,yaml; print\\('scratch venv ready'\\)\")",
"Bash(/tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/pip install *)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_default_agent.py -q -p no:cacheprovider)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py -q -p no:cacheprovider)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py tests/test_desk_api.py tests/test_main_helpers.py -q -p no:cacheprovider)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py --ignore=tests/test_billing_web.py)",
"WebFetch(domain:elevenlabs.io)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_dialogue.py -q)"
]
}
}
+30
View File
@@ -31,6 +31,36 @@ ELEVENLABS_VOICE_ID=
#ELEVENLABS_MODEL_ID=eleven_flash_v2
#TTS_SAMPLE_RATE=24000
# Ask Bolt to use a different voice (or another language) and the server
# picks one from the ElevenLabs voice library and tags the reply with it.
# It only tags one reply, and it can't remember the id afterwards — so the
# pet keeps using that voice until a new one is picked or you choose "Use
# default voice" in the tray. VOICE_STICKY=false makes each pick last for
# exactly the one reply it came with instead.
#VOICE_STICKY=true
# ── Multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
# Lets Bolt play a short scene in several voices with delivery tags the v3
# model acts on ("[cheerfully] Hello", "[whispering] He is lying"), driven by
# the server through a relayed `dialoguectl` command. Name the cast here —
# "self" always means whatever voice the pet is currently using.
#DIALOGUE=true
#DIALOGUE_MODEL_ID=eleven_v3
#DIALOGUE_VOICES=narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD
# ── Self-restart (optional) ─────────────────────────────────────────────────
# `petctl self_restart <why>` lets Bolt reload the pet after editing its own
# code, so he can check the change live. The code is import-checked first, the
# restart waits for the current turn to finish, and the reason is carried
# across so the new process reports back. The guard refuses more than
# SELF_RESTART_MAX restarts within SELF_RESTART_WINDOW_SECONDS.
#SELF_RESTART=true
#SELF_RESTART_MAX=5
#SELF_RESTART_WINDOW_SECONDS=900
# Used instead of ELEVENLABS_MODEL_ID whenever the server picked the voice
# or the reply has non-ASCII in it — the flash_v2 default is English-only.
#ELEVENLABS_MULTILINGUAL_MODEL_ID=eleven_flash_v2_5
# ── Audio devices (optional — leave blank for the system default) ──────────
#MIC_DEVICE=
#SPEAKER_DEVICE=
+90 -9
View File
@@ -14,7 +14,8 @@ Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk /
click → record utterance → Deepgram STT → + active-window + screen-layout
context → POST /desk/converse → [server may relay a shell command to run on
this machine, or a `petctl` pseudo-command that moves/emotes the pet, jumps it
to another monitor, or reads a screen's text back instead] → reply →
to another monitor, reads a screen's text back, or plays a multi-voice scene
instead] → reply (optionally tagged with a voice the server picked for it) →
ElevenLabs streaming TTS (or offline pyttsx3 fallback) → speakers`, with the
pet sprite/speech bubble reflecting state throughout, and playback
interruptible by talking over it (barge-in).
@@ -79,7 +80,10 @@ logs a missing-config message and exits its thread instead of starting.
`/desk/tool_result` until the server sends a final `reply` (capped at
`_MAX_RELAY_HOPS`). This is the same "full desktop control" trust model as
the server repo's other desk clients — commands only ever originate from
the user's own voice/click requests in their own session. `list_outbox_files`
the user's own voice/click requests in their own session. A final reply is
returned as a `Reply(text, voice_id, voice_name)` rather than a bare string,
because the server can tag it with a voice — see "Voices" below.
`list_outbox_files`
/ `download_outbox_file` hit the same `/desk/files` and `/desk/files/<id>`
endpoints the server's `deliver_files` tool queues onto — see `file_delivery.py`.
- **`file_delivery.py`** — the filesystem half of receiving files the server
@@ -104,7 +108,11 @@ logs a missing-config message and exits its thread instead of starting.
`play_stream()` start playback on the first chunk; `chunks_to_int16()`
carries odd bytes across HTTP chunk boundaries, without which everything
after the first split sample plays as static — falling back to whole-clip
PCM then offline `pyttsx3`), `barge_in.py` (two detectors behind one
PCM then offline `pyttsx3`; every entry point takes an optional `voice_id`
overriding `ELEVENLABS_VOICE_ID`, and `model_for()` picks the multilingual
model whenever there's an override or non-ASCII text, since the default
`eleven_flash_v2` is English-only and would read either as garbled
phonetic English rather than failing), `barge_in.py` (two detectors behind one
`reset()`/`check()` shape, chosen by `BARGE_IN_MODE` via `make_detector`:
**wake** (default) scores every frame with the same openWakeWord model the
idle listener uses, so only the wake phrase cuts playback; **energy** is the
@@ -115,8 +123,8 @@ logs a missing-config message and exits its thread instead of starting.
accepts an injectable stream/model/protocol so tests don't need real audio
hardware or a display.
- **`pet_actions.py`** — `petctl` pseudo-commands (`petctl move top-left`,
`petctl emote wave`, `say`/`wander`/`nap`, plus the screen verbs
`jump`/`monitors`/`read`). The desk API has no "move the pet" payload type
`petctl emote wave`, `say`/`wander`/`nap`, the screen verbs
`jump`/`monitors`/`read`, and `voice reset`). The desk API has no "move the pet" payload type
and this repo can't change the server, so these ride the existing
shell-command relay: `controller._handle_command` parses them and they never
reach `subprocess`; anything else is a real shell command exactly as before.
@@ -125,7 +133,51 @@ logs a missing-config message and exits its thread instead of starting.
module doesn't have, so the spec passes through to `monitors.resolve()`.
Query verbs (`monitors`, `read`) are answered in `_handle_command` rather
than by `pet_actions.describe()`, because their output *is* the point: it
goes back up the tool-result relay for Bolt to use in his reply.
goes back up the tool-result relay for Bolt to use in his reply — as is
`voice reset`, which reports what it dropped since the server can't see
which voice is in use. `voice` only ever resets: picking one is the
server's job (`speak_as`, which it already knows how to use), so a
`petctl voice <name>` attempt is an error pointing back at that marker.
- **`self_restart.py`** — `petctl self_restart`, the pet restarting itself so
Bolt can *see* a code change he just made instead of waiting for a human to
restart it. Three problems shape it, and all three are the interesting part.
(1) The restart can't happen inline: killing the process mid-turn would drop
the HTTP tool relay before the result was posted, leaving the server to wait
out its timeout on a turn that can never finish — so the command only
*arms* it (`controller._arm_self_restart`) and
`controller._maybe_self_restart` fires it after the reply is spoken, the
same "only between turns" rule the updater follows. (2) A broken edit must
not be fatal, so `preflight()` imports the package in a **subprocess**
before arming — this process holds the old modules, so an in-process import
would pass on a file that no longer parses — and a SyntaxError comes back as
the command's output, in the same turn, with the pet still running. (3) The
reason has to outlive the process, so it's written to
`~/.cache/bolt-pet/restart_context.json` (never inside the repo Bolt is
editing) and read on the way back up by `controller._report_self_restart`,
which posts it to the server as an ordinary turn — that's what makes
"restart and check the sprites load" finish as a spoken sentence rather than
a silence. `check_loop_guard` refuses after `SELF_RESTART_MAX` restarts in
`SELF_RESTART_WINDOW_SECONDS`, so an edit-restart-crash cycle stops itself.
Off switch: `SELF_RESTART=false`.
- **`dialogue.py`** — `dialoguectl` pseudo-commands: a multi-voice *scene*
through ElevenLabs' Text to Dialogue endpoint (`audio/tts.
synthesize_dialogue`), checked in `_handle_command` between petctl and
filectl. Same single-line-JSON wire format as filectl and for the same
reason (the server's `command` marker captures only up to the next
newline), and it accepts the ElevenLabs field names (`inputs`/`voice_id`)
as well as its own (`lines`/`voice`) because the model has read that API
and copying its shape is the obvious thing to try. Voices are *named*
(`DIALOGUE_VOICES` maps names to ids) rather than pasted as raw ids, and
`self` resolves to whatever voice the pet is speaking with right now —
including a `speak_as` pick — so Bolt sounds like himself in his own
scenes. The API's limits (10 distinct voices, ~2000 characters) are
enforced *before* the request so a mistake comes back up the tool-result
relay as a sentence Bolt can act on rather than an HTTP 422 he can't see.
Unlike the normal reply path there is no streaming variant, so a scene is
whole-clip: `controller._play_dialogue` plays it with the same bubble,
transcript and barge-in handling a spoken reply gets, and returns to
THINKING afterwards (not IDLE) because the server is still waiting on the
tool result — that leg is why `state.py` allows TALKING -> THINKING.
- **`file_ops.py`** — `filectl` pseudo-commands, checked in `_handle_command`
right after petctl and before falling through to a real shell command.
Executing arbitrary commands already worked via the shell relay
@@ -274,9 +326,38 @@ logs a missing-config message and exits its thread instead of starting.
name rather than by `PetState`, with `has()` reporting whether a key is
backed by real art so callers can decline a placeholder instead of trotting
a blob across the desktop; `tray.py` is the system tray menu (talk now / mute / nap / wander /
click-through / history / wake-word tuning / quit) — the pet window has no
title bar or taskbar entry; `history_window.py` and `wake_tuner.py` are the
two dialogs it opens.
click-through / history / wake-word tuning / use-default-voice / quit) — the
pet window has no title bar or taskbar entry; `history_window.py` and
`wake_tuner.py` are the two dialogs it opens.
### Voices (the server's `speak_as`)
Ask Bolt to talk like someone else, or in another language, and the *server*
does the picking: its desk-only `voice_search` marker browses the ElevenLabs
voice library, and `speak_as: <voice_id>` on the final reply tags that reply
with the chosen voice (adding a Voice Library pick to the ElevenLabs account
first, so the id is usable by the time it reaches us). Nothing about that is
this repo's to decide — all the client owes it is actually speaking in the
voice it was handed: `converse()` returns it on `Reply`, `_apply_voice()`
records it, and `_speak()` passes it to `tts.speak(voice_id=...)`.
Two things are decided *here*, though, because the server can't:
- **The voice sticks** (`VOICE_STICKY`, default on). The server tags one
reply and strips the marker before storing the turn, so it never sees the
id again — "keep talking like that" would send it searching for a voice all
over again, and it'd likely land on a different one. Holding the id
client-side is what makes the rest of the conversation stay in that voice.
An untagged reply therefore never *changes* the voice; only a new
`speak_as`, `VOICE_STICKY=false`, or a reset does.
- **There's a way back.** Since the server was never told Bolt's own voice
id, it can't ask for it back with `speak_as` — so reverting is local: the
tray's **Use default voice** entry (enabled only while a picked voice is
in use, kept in sync by the `voice_changed` signal), a restart, or
`petctl voice reset`, which is what lets Bolt honour "go back to your
normal voice" out loud. That last one needs the server's pet prompt block
(`ai/desk_api.py`, `pet_tools`) to mention the verb, or the model never
emits it — the desk API's prompt is where petctl is advertised.
### Wake-word detection
+39 -4
View File
@@ -69,8 +69,8 @@ limitations).
speaks, and barging in starts your next turn immediately (`BARGE_IN`).
- Right-click the tray icon for **Talk now**, **Mute mic**, **Nap**,
**Wander around**, **Click through the pet**, **History…**, **Wake word
tuning…** and **Quit** — the pet window itself has no title bar or taskbar
entry.
tuning…**, **Use default voice** and **Quit** — the pet window itself has
no title bar or taskbar entry.
- **Click the speech bubble** to copy what it just said; the tray's
**History…** window keeps the last `HISTORY_LIMIT` turns.
@@ -80,8 +80,8 @@ limitations).
listening/thinking/talking or while a bubble is up.
- **Moves and emotes on command.** Bolt can relay `petctl move top-left`,
`petctl emote wave|hop|spin|nod|shake`, `petctl say ...`, `petctl wander
on|off`, `petctl nap on|off`. These are intercepted here and never reach a
shell.
on|off`, `petctl nap on|off`, `petctl voice reset`, and `dialoguectl` for a
multi-voice scene. These are intercepted here and never reach a shell.
- **Naps** during `QUIET_HOURS` (e.g. `23:00-08:00`) or while a fullscreen
app is focused (`DND_ON_FULLSCREEN`) — it dims, stops wandering, and makes
no proactive noise. It still answers when you speak to it.
@@ -90,6 +90,41 @@ limitations).
notifications get forwarded to the server, so it can tell you the deploy
went green. Off by default: each one costs a round trip.
## Speaking in another voice
Ask for a different voice — "use a clearer voice", "talk like a pirate", "say
that in Japanese" — and Bolt searches the ElevenLabs voice library on the
server, picks one, and tags his reply with it (`speak_as`); the pet is what
actually speaks in it. A Voice Library pick is added to your ElevenLabs
account automatically the first time it's used, and non-English replies (or
any picked voice) go through `ELEVENLABS_MULTILINGUAL_MODEL_ID` rather than
the English-only `eleven_flash_v2` default.
The new voice **stays on** for the rest of the conversation, because the
server tags a single reply and doesn't remember which voice it chose — so
"keep talking like that" would otherwise send it hunting for a voice again.
To get his own voice back: ask him ("use your normal voice" — he relays
`petctl voice reset`), use **Use default voice** in the tray menu (greyed
out unless a picked voice is active), or restart the pet. Set
`VOICE_STICKY=false` in `.env` if you'd rather each pick lasted exactly one
reply.
## Multi-voice dialogue
Ask for a scene — "do the argument between the two of them", "read that back
as a radio play" — and Bolt can relay a `dialoguectl` command that the pet
renders through ElevenLabs' Text to Dialogue endpoint: several voices in one
take, with delivery tags the v3 model acts on (`[cheerfully]`, `[whispering]`,
`[stuttering]`). One request per scene, so the voices actually react to each
other instead of sounding like clips glued together.
Name the cast in `.env` (`DIALOGUE_VOICES=narrator:9BWts…,villain:IKne3…`);
the name `self` always means whatever voice the pet is currently using, so
Bolt sounds like himself in his own scenes — including after a `speak_as`
switch. Scenes show up in the speech bubble with the tags stripped, count as
normal speech for the transcript, and can be talked over like any other reply.
`DIALOGUE=false` turns the whole thing off on this device.
## Wake-word detection
`bolt_pet/audio/wake_word.py` feeds every mic frame into `thunderbolt.onnx`
+96 -12
View File
@@ -5,11 +5,16 @@ desk_client/bolt_desk.py which shells out because it only targets Linux.
Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech
on macOS, espeak on Linux) if ElevenLabs isn't configured or the request
fails, so the pet can still talk with zero cloud config.
Every entry point takes an optional *voice_id* that overrides
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
reply marker reaches the speakers (see controller._apply_voice). The offline
fallback has no such concept and always sounds like itself.
"""
from __future__ import annotations
from typing import Iterable, Iterator
from typing import Iterable, Iterator, Optional
import numpy as np
import requests
@@ -21,18 +26,40 @@ class TtsError(Exception):
pass
def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
def voice_for(voice_id: Optional[str] = None) -> str:
"""The voice this call should use: an override (server `speak_as`) if
given, else the configured default."""
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
def model_for(text: str, voice_id: Optional[str] = None) -> str:
"""Which ElevenLabs model to synthesize with.
The default (`eleven_flash_v2`) is English-only, and both things that
reach this branch mean the reply probably isn't English: a voice the
server picked mid-conversation is nearly always about a language or an
accent, and non-ASCII text can't be English at all. Rendering either one
through the English model gets you a mangled phonetic reading rather
than a failure, which is worse — so those go through the multilingual
model instead."""
if (voice_id or "").strip() or not text.isascii():
return config.ELEVENLABS_MULTILINGUAL_MODEL_ID
return config.ELEVENLABS_MODEL_ID
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
callers should fall back to speak_offline() rather than treating this
as fatal."""
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
voice = voice_for(voice_id)
if not (config.ELEVENLABS_API_KEY and voice):
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
try:
response = requests.post(
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}",
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
json={"text": text, "model_id": model_for(text, voice_id)},
timeout=60,
)
response.raise_for_status()
@@ -44,20 +71,23 @@ def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
return pcm, config.TTS_SAMPLE_RATE
def stream_pcm(text: str, chunk_bytes: int = 4096) -> Iterator[np.ndarray]:
def stream_pcm(
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
) -> Iterator[np.ndarray]:
"""Same audio as synthesize_pcm(), but yielded as it arrives from
ElevenLabs' /stream endpoint so playback can start on the first chunk
(~300ms) instead of after the whole clip is synthesized. Raises TtsError
before yielding anything if the request itself fails, so callers can fall
back cleanly; a mid-stream failure just ends the generator."""
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
voice = voice_for(voice_id)
if not (config.ELEVENLABS_API_KEY and voice):
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
try:
response = requests.post(
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}/stream",
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}/stream",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
json={"text": text, "model_id": model_for(text, voice_id)},
timeout=60,
stream=True,
)
@@ -83,6 +113,55 @@ def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
yield np.frombuffer(data[:usable], dtype=np.int16)
def synthesize_dialogue(
inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None
) -> tuple[np.ndarray, int]:
"""Multi-voice scene via ElevenLabs Text to Dialogue.
One request, one take: the whole exchange is synthesized together, which
is the point — the model hears the previous line, so reactions and timing
land instead of sounding like separately-rendered clips.
Same PCM-over-`requests` posture as the rest of this module (no SDK, no
`play()` shelling out to ffplay), so playback is the same sounddevice path
everything else uses and barge-in works on it unchanged. There is no
documented streaming variant, and a scene is a short set piece anyway, so
this is whole-clip only.
"""
if not (config.ELEVENLABS_API_KEY and inputs):
raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)")
body: dict = {
"inputs": [
{"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")}
for entry in inputs
],
"model_id": model_id or config.DIALOGUE_MODEL_ID,
}
if stability is not None:
body["settings"] = {"stability": float(stability)}
try:
response = requests.post(
"https://api.elevenlabs.io/v1/text-to-dialogue",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json=body,
timeout=120, # a multi-voice take is slower to render than one line
)
response.raise_for_status()
except Exception as exc:
detail = ""
# The API explains refusals (character limit, unknown voice) in the
# body; surfacing it is what lets Bolt fix the call and retry.
body_text = getattr(getattr(exc, "response", None), "text", "")
if body_text:
detail = f"{body_text[:300]}"
raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc
pcm = np.frombuffer(response.content, dtype=np.int16)
if pcm.size == 0:
raise TtsError("ElevenLabs returned no dialogue audio")
return pcm, config.TTS_SAMPLE_RATE
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None) -> bool:
"""Play a whole clip. Returns True if it finished, False if *should_stop*
(barge-in) cut it short. *should_stop* is polled while audio plays — each
@@ -134,11 +213,12 @@ def speak_offline(text: str) -> None:
engine.runAndWait()
def speak(text: str, on_error=None, should_stop=None) -> bool:
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None) -> bool:
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
exception when ElevenLabs fails (useful for logging) — a fallback still
runs either way. Returns False if barge-in interrupted playback.
*voice_id* overrides the configured voice for this line only.
The text is sanitized first (speech_text.for_speech): server replies are
written for a chat window, and a voice reads markdown/emoji literally
@@ -149,12 +229,16 @@ def speak(text: str, on_error=None, should_stop=None) -> bool:
return True
if config.TTS_STREAMING:
try:
return play_stream(stream_pcm(text), config.TTS_SAMPLE_RATE, should_stop=should_stop)
return play_stream(
stream_pcm(text, voice_id=voice_id),
config.TTS_SAMPLE_RATE,
should_stop=should_stop,
)
except TtsError as exc:
if on_error is not None:
on_error(exc)
try:
pcm, sample_rate = synthesize_pcm(text)
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
return play_pcm(pcm, sample_rate, should_stop=should_stop)
except TtsError as exc:
if on_error is not None:
+38 -1
View File
@@ -66,9 +66,38 @@ DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3")
ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "")
ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "")
ELEVENLABS_MODEL_ID = os.environ.get("ELEVENLABS_MODEL_ID", "eleven_flash_v2")
# eleven_flash_v2 is English-only, and the two cases that swap the voice
# (server-picked `speak_as`, or a reply with non-ASCII in it) are usually
# exactly the cases where the reply isn't English — see tts.model_for().
ELEVENLABS_MULTILINGUAL_MODEL_ID = os.environ.get(
"ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5"
)
# ElevenLabs PCM output formats are named pcm_<sample_rate>.
TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "24000"))
# Does a voice the server picks (its speak_as marker — "talk like a pirate",
# "say that in Japanese") stay on for later replies, or last one reply only?
# Sticky by default: the server tags a single reply and does *not* keep the
# voice id in its history, so a one-reply-only voice can't be re-used when
# you say "keep talking like that" — it would have to search for a voice
# again. Reset it from the tray ("Use default voice") or by restarting.
VOICE_STICKY = os.environ.get("VOICE_STICKY", "true").lower() in ("1", "true", "yes", "on")
# ── multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
# Lets Bolt play a short scene in several voices with delivery tags the v3
# model acts on ("[cheerfully] Hello"), instead of one voice reading a line.
# Driven by the server through the `dialoguectl` relayed command — see
# dialogue.py. Costs a separate (slower, whole-clip) request per scene, so
# it's a set piece, not the normal reply path.
#
# DIALOGUE_VOICES names the cast: "narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD".
# The name "self" always resolves to the voice the pet is currently using,
# including one the server picked with speak_as.
DIALOGUE = os.environ.get("DIALOGUE", "true").lower() in ("1", "true", "yes", "on")
DIALOGUE_MODEL_ID = os.environ.get("DIALOGUE_MODEL_ID", "eleven_v3")
DIALOGUE_VOICES = os.environ.get("DIALOGUE_VOICES", "")
# ── mic / VAD (same tuning knobs as bolt_desk.py) ───────────────────────────
MIC_DEVICE = os.environ.get("MIC_DEVICE", "") or None # sounddevice name/index
@@ -91,7 +120,7 @@ GRACE_SECONDS = float(os.environ.get("VAD_GRACE_SECONDS", "4"))
# keeps feeding it noise. 0 means no cap.
FOLLOW_UP_LISTEN = os.environ.get("FOLLOW_UP_LISTEN", "true").lower() in ("1", "true", "yes", "on")
FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "3"))
FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "10"))
FOLLOW_UP_GRACE_SECONDS = float(os.environ.get("FOLLOW_UP_GRACE_SECONDS", "7"))
COMMAND_TIMEOUT_SECONDS = int(os.environ.get("COMMAND_TIMEOUT_SECONDS", "30"))
@@ -110,6 +139,14 @@ SUDO_ASKPASS_HELPER = os.environ.get("SUDO_ASKPASS_HELPER", "") # blank = auto-
SUDO_COMMAND_TIMEOUT_SECONDS = int(os.environ.get("SUDO_COMMAND_TIMEOUT_SECONDS", "180"))
HEARTBEAT_INTERVAL_SECONDS = float(os.environ.get("HEARTBEAT_INTERVAL_SECONDS", "60"))
# ── self-restart ────────────────────────────────────────────────────────────
# `petctl self_restart` lets Bolt restart the pet after editing its code, so
# he can see his own change running instead of waiting for someone to restart
# it by hand. The code is import-checked in a subprocess first, and the reason
# is carried across the restart so the new process can report back — see
# self_restart.py. SELF_RESTART_MAX/_WINDOW_SECONDS bound the crash-loop case.
SELF_RESTART = os.environ.get("SELF_RESTART", "true").lower() in ("1", "true", "yes", "on")
# ── barge-in (interrupt playback while the pet is talking) ──────────────────
# The mic stays live while the pet talks. BARGE_IN_MODE decides what counts
# as an interruption:
+211 -6
View File
@@ -20,10 +20,12 @@ from typing import Optional
from PySide6.QtCore import QObject, Signal
from . import (
config, file_delivery, file_ops, history as history_mod,
monitors as monitors_mod, notifications, pet_actions, quiet,
screen_context, screen_text, server_client, speech_text, updater,
config, dialogue as dialogue_mod, file_delivery, file_ops,
history as history_mod, monitors as monitors_mod, notifications,
pet_actions, quiet, screen_context, screen_text, self_restart,
server_client, speech_text, updater,
)
from . import __version__
from .audio import barge_in, mic, stt, tts, wake_word
from .state import PetState, PetStateMachine
@@ -38,6 +40,7 @@ class PetController(QObject):
log = Signal(str)
action = Signal(dict) # parsed petctl action for the UI to perform
napping = Signal(bool) # quiet hours / fullscreen do-not-disturb
voice_changed = Signal(str) # name of the server-picked voice ("" = default)
restart_requested = Signal(str) # version we just updated to
finished = Signal()
@@ -66,6 +69,13 @@ class PetController(QObject):
self._wake_threshold = config.WAKE_WORD_THRESHOLD
self._near_misses = wake_word.NearMissLog()
# The voice the server last picked for us with `speak_as` ("" = the
# configured default). Held here rather than passed straight through
# to one tts.speak() call because it's sticky by default — see
# _apply_voice for why.
self._voice_id = ""
self._voice_name = ""
self._barge_in: Optional[barge_in.BargeInDetector] = None
self._napping = False
self._nap_forced: Optional[bool] = None # petctl nap on/off overrides the schedule
@@ -80,6 +90,9 @@ class PetController(QObject):
self._last_update_check = 0.0
self._update_pending = False # applied on disk, waiting for the restart
# Armed by `petctl self_restart`, fired after the turn it was asked in
# (see _arm_self_restart for why it can't happen inline).
self._restart_context = None
self._notification_watcher: Optional[notifications.NotificationWatcher] = None
self._notification_gate = notifications.NotificationGate(
@@ -117,6 +130,20 @@ class PetController(QObject):
def reset_wake_stats(self) -> None:
self._near_misses.clear()
def current_voice(self) -> str:
"""Name (or id) of the server-picked voice in use, "" for the default."""
return self._voice_name or self._voice_id
def reset_voice(self) -> None:
"""Drop a server-picked voice and go back to Bolt's own. The tray's
way out of a voice you didn't want to keep — the server has no way to
ask for the default back, since it never learns what it is."""
if not self._voice_id:
return
self._voice_id = self._voice_name = ""
self.log.emit("Voice: back to the default.")
self.voice_changed.emit("")
def stop(self) -> None:
self._running = False
self._talk_now.set() # wake up anything blocked waiting on it
@@ -167,6 +194,7 @@ class PetController(QObject):
except Exception as exc:
self.log.emit(f"Server not reachable yet ({exc}) — will keep trying per-request.")
self._start_notification_bridge()
self._report_self_restart()
self._loop()
if self._notification_watcher is not None:
self._notification_watcher.stop()
@@ -260,8 +288,10 @@ class PetController(QObject):
return
self._check_deliveries()
self._speak(reply)
self._apply_voice(reply)
self._speak(reply.text)
self._state.transition(PetState.IDLE)
self._maybe_self_restart()
def _with_context(self, text: str) -> str:
"""Everything the server gets alongside what you actually said: the
@@ -307,6 +337,18 @@ class PetController(QObject):
kind = action["action"]
if kind == "monitors":
return monitors_mod.describe(self._monitors, self._pet_monitor)
if kind == "self_restart":
return self._arm_self_restart(action.get("reason") or "")
if kind == "voice":
# Answered here, not by describe(): the UI has no part in it,
# and the server needs to hear whether there was anything to
# drop — it can't see which voice we're using.
previous = self.current_voice()
self.reset_voice()
return (
f"[pet] back to your own voice (was {previous})" if previous
else "[pet] already using your own voice"
)
if kind == "read":
return self._read_screen(action["target"])
if kind == "jump":
@@ -327,6 +369,14 @@ class PetController(QObject):
self.action.emit(action)
return pet_actions.describe(action)
try:
scene = dialogue_mod.parse(command)
except dialogue_mod.DialogueError as exc:
self.log.emit(f"dialoguectl: {exc}")
return f"[dialogue] {exc}"
if scene is not None:
return self._play_dialogue(scene)
try:
file_action = file_ops.parse(command)
except file_ops.FileOpError as exc:
@@ -365,6 +415,159 @@ class PetController(QObject):
self.log.emit(f"Reading monitor {monitor.number} ({monitor.name})…")
return screen_text.read_monitor(monitor, limit)
def _arm_self_restart(self, reason: str) -> str:
"""`petctl self_restart` — check the code, then arm a restart.
Nothing restarts here. The tool result has to get back up the relay
before this process can die (otherwise the server waits out its
timeout on a turn that will never finish), so the restart is armed and
`_maybe_self_restart` fires it once the turn has been spoken. The
preflight import runs *now*, in this turn, so a syntax error Bolt just
introduced comes back as something he can read and fix rather than as
a pet that never comes back."""
if not config.SELF_RESTART:
return "[pet] self-restart is disabled on this device (SELF_RESTART=false)"
if self._restart_context is not None:
return "[pet] a restart is already armed for the end of this turn"
try:
self_restart.check_loop_guard(self_restart.load())
self.log.emit("Self-restart requested — checking the code imports first…")
self_restart.preflight()
except self_restart.RestartError as exc:
self.log.emit(f"Self-restart refused: {exc}")
return f"[pet] restart refused — {exc}"
recent = [entry.text[:120] for entry in self.history.entries()[-4:]]
self._restart_context = self_restart.arm(
reason or "no reason given",
verify=reason,
version=__version__,
session=config.SESSION_ID,
recent=recent,
)
self.log.emit("Self-restart armed; it happens after this turn.")
return (
"[pet] code imports cleanly; restarting as soon as this turn finishes. "
"I'll come back and tell you what version I'm on and what I found — "
"wrap up your reply now, the next thing you hear from me is the report."
)
def _maybe_self_restart(self) -> bool:
"""Fire an armed restart, once the turn is over and the reply spoken.
Returns True if a restart was requested, so the caller can stop
driving the pipeline — the process is on its way out."""
if self._restart_context is None:
return False
self._update_pending = True # same latch the updater uses: no double restart
self.log.emit("Restarting now.")
self._state.force(PetState.IDLE)
self.restart_requested.emit(f"self-restart: {self._restart_context.reason[:60]}")
return True
def _report_self_restart(self) -> None:
"""On the way up: tell the server we're back, and why we left.
Runs once, before the listen loop starts, and only when a context file
was left behind. The report goes through the ordinary conversation
path, so Bolt's answer is spoken out loud like any other turn — which
is what makes "restart and check the sprites load" finish as a
sentence instead of a silence."""
context = self_restart.load()
if context is None:
return
self_restart.clear()
message = self_restart.report(context, version=__version__)
self.log.emit(f"Back from a self-restart ({context.reason[:80]}).")
self.history.add(history_mod.SYSTEM, message, time.time())
try:
reply = server_client.converse(message, on_command=self._handle_command)
except server_client.ServerError as exc:
# The restart still worked; only the report failed. Say so locally
# rather than pretending nothing happened.
self.log.emit(f"Couldn't report the restart to the server: {exc}")
return
self._apply_voice(reply)
if reply.text.strip() and not self._napping:
self._speak(reply.text)
self._state.force(PetState.IDLE)
def _play_dialogue(self, scene: dict) -> str:
"""Play a `dialoguectl` scene and report back up the relay.
This runs *mid-turn* (the server is still waiting on the tool result),
so the pet has to look like it's talking and then go back to waiting —
hence the TALKING → THINKING leg rather than the usual return to IDLE.
Everything a normal reply gets, a scene gets too: the bubble, the
transcript, and barge-in, so a long scene can be talked over exactly
like a long answer."""
if not config.DIALOGUE:
return "[dialogue] disabled on this device (DIALOGUE=false)"
try:
inputs = dialogue_mod.resolve(
scene,
voices=dialogue_mod.parse_voice_map(config.DIALOGUE_VOICES),
self_voice=self._voice_id or config.ELEVENLABS_VOICE_ID,
)
except dialogue_mod.DialogueError as exc:
self.log.emit(f"dialoguectl: {exc}")
return f"[dialogue] {exc}"
text = dialogue_mod.spoken_text(scene)
self.log.emit(f"Dialogue ({len(inputs)} lines): {text[:120]}")
try:
pcm, sample_rate = tts.synthesize_dialogue(
inputs, model_id=scene.get("model"), stability=scene.get("stability")
)
except tts.TtsError as exc:
self.log.emit(f"Dialogue failed: {exc}")
# Reported, not raised: the server can read this, shorten the
# scene or fix the voice, and try again inside the same turn.
return f"[dialogue] couldn't synthesize it: {exc}"
resume = self._state.state
self._state.transition(PetState.TALKING)
self.said.emit(speech_text.for_display(text))
self.history.add(history_mod.PET, text, time.time())
should_stop = None
if self._barge_in is not None:
self._barge_in.reset()
should_stop = self._barge_in.check
completed = tts.play_pcm(pcm, sample_rate, should_stop=should_stop)
if self._barge_in is not None:
self._barge_in.reset() # the pet's own voices are in the wake window
if resume in (PetState.THINKING, PetState.IDLE):
self._state.transition(resume)
if not completed:
return dialogue_mod.describe(scene) + " (interrupted — they talked over it)"
return dialogue_mod.describe(scene)
def _apply_voice(self, reply) -> None:
"""Adopt (or drop) the voice the server tagged this reply with.
The server's `speak_as` marker names an ElevenLabs voice it just
picked — and it adds a Voice Library pick to the account first, so by
the time the id gets here it's usable for TTS. It tags *one* reply,
but the voice sticks by default: the server strips the marker before
storing the turn, so it can't recall the id later, and "keep talking
like that" would otherwise send it searching for a voice all over
again. `VOICE_STICKY=false` makes each pick last exactly one reply.
Untagged replies never *change* the voice — with stickiness on they
just keep whatever's in use, which is what makes the rest of the
conversation stay in the requested voice."""
voice_id = getattr(reply, "voice_id", "")
if voice_id:
if voice_id != self._voice_id:
self._voice_id = voice_id
self._voice_name = getattr(reply, "voice_name", "") or ""
self.log.emit(f"Voice: {self.current_voice()}")
self.voice_changed.emit(self.current_voice())
elif not config.VOICE_STICKY:
self.reset_voice()
def _speak(self, text: str) -> None:
self._state.transition(PetState.TALKING)
# Bubble gets the markdown stripped but emoji kept (it can't render
@@ -382,6 +585,7 @@ class PetController(QObject):
text,
on_error=lambda exc: self.log.emit(f"TTS failed: {exc}"),
should_stop=should_stop,
voice_id=self._voice_id or None,
)
# Read the scoring history *before* resetting, or the log reports the
# blank counters instead of what actually fired.
@@ -504,8 +708,9 @@ class PetController(QObject):
self.log.emit(f"Couldn't forward notification: {exc}")
return
self._check_deliveries()
if reply.strip():
self._speak(reply)
self._apply_voice(reply)
if reply.text.strip():
self._speak(reply.text)
self._state.transition(PetState.IDLE)
# ── file delivery ────────────────────────────────────────────────────
+219
View File
@@ -0,0 +1,219 @@
"""`dialoguectl` — multi-voice dialogue playback (ElevenLabs Text to Dialogue).
Normal replies are one voice saying one thing (audio/tts.py). This is the
other mode: a short *scene* two or more voices, with delivery tags the v3
model acts on (`[cheerfully]`, `[stuttering]`, `[whispering]`) synthesized
as a single take so the timing and reactions between lines actually sound
like a conversation rather than clips glued together.
Wire format, the same discipline as file_ops.py and for the same reason: it
rides the server's ordinary `command` tool marker, whose extractor only
captures up to the next newline, so the payload is a **single-line compact
JSON object**.
dialoguectl {"lines": [{"voice": "self", "text": "[cheerfully] Morning!"},
{"voice": "narrator", "text": "[whispering] He lies."}]}
The ElevenLabs field names are accepted too (`inputs` / `voice_id`), because
the model has read that API and copying its shape is the obvious thing to
try:
dialoguectl {"inputs": [{"voice_id": "9BWtsMINqrJLrRacOk9x", "text": "hi"}]}
Voices are *named*, not pasted as ids. `DIALOGUE_VOICES` in .env maps names
to ids (`narrator:9BWts,villain:IKne3`), and `self` always means the voice
the pet is speaking with right now including a voice the server picked
mid-conversation with `speak_as`, so a scene featuring Bolt sounds like
whoever Bolt currently is.
Pure parsing and validation here; the HTTP call is
`audio/tts.synthesize_dialogue` and the playback/state handling is
`controller._play_dialogue`, matching the parse/execute split used by
pet_actions.py and file_ops.py.
The API's own limits are enforced *here*, before the request goes out, so a
mistake comes back through the tool-result relay as a sentence Bolt can act
on ("too many characters, split it") rather than as an HTTP 422 he can't see.
"""
from __future__ import annotations
import json
import re
from typing import Iterable, Optional
_PREFIXES = ("dialoguectl", "dialogue", "scene")
# ElevenLabs Text to Dialogue limits (docs, 2026-07): at most 10 distinct
# voice ids per request and ~2000 characters across all inputs.
MAX_VOICES = 10
MAX_CHARS = 2000
# Names that always mean "the voice the pet is using right now".
SELF_NAMES = ("self", "bolt", "me", "pet")
# A raw ElevenLabs voice id: 20 URL-safe characters, no separators. Used to
# tell "the model pasted an id" from "the model used a name".
_VOICE_ID_RE = re.compile(r"^[A-Za-z0-9]{20}$")
class DialogueError(Exception):
"""Bad dialoguectl syntax or an unusable request — reported back to the
server as this command's output."""
def is_dialogue_command(command: str) -> bool:
parts = (command or "").strip().split(None, 1)
return bool(parts) and parts[0].lower() in _PREFIXES
def parse(command: str) -> Optional[dict]:
"""Parse `dialoguectl <json>` into {"lines": [{"voice", "text"}], ...}.
Returns None if this isn't a dialogue command at all (the caller then
tries filectl, then a real shell command). Raises DialogueError on a
dialogue command that doesn't make sense."""
if not is_dialogue_command(command):
return None
_, _, payload = (command or "").strip().partition(" ")
payload = payload.strip()
if not payload:
raise DialogueError(
'dialoguectl needs a JSON argument, e.g. dialoguectl {"lines": '
'[{"voice": "self", "text": "[cheerfully] hello"}]}'
)
try:
data = json.loads(payload)
except json.JSONDecodeError as exc:
raise DialogueError(
f"couldn't parse the JSON ({exc}). It must be one line of compact "
"JSON — put line breaks inside text as \\n, never as real newlines."
) from exc
if not isinstance(data, dict):
raise DialogueError("the argument must be a JSON object, not a list or a bare value")
raw_lines = data.get("lines")
if raw_lines is None:
raw_lines = data.get("inputs") # the ElevenLabs field name
if not isinstance(raw_lines, list) or not raw_lines:
raise DialogueError('needs a non-empty "lines" array of {"voice", "text"} objects')
lines: list[dict] = []
for index, entry in enumerate(raw_lines, start=1):
if not isinstance(entry, dict):
raise DialogueError(f"line {index} must be an object with 'voice' and 'text'")
text = str(entry.get("text") or "").strip()
if not text:
raise DialogueError(f"line {index} has no text")
voice = str(entry.get("voice") or entry.get("voice_id") or "self").strip()
lines.append({"voice": voice, "text": text})
action = {"action": "dialogue", "lines": lines}
model = str(data.get("model") or data.get("model_id") or "").strip()
if model:
action["model"] = model
stability = data.get("stability")
if stability is not None:
try:
action["stability"] = min(1.0, max(0.0, float(stability)))
except (TypeError, ValueError):
raise DialogueError("stability must be a number between 0 and 1") from None
return action
def parse_voice_map(spec: str) -> dict[str, str]:
"""Parse DIALOGUE_VOICES ("narrator:9BWts…, villain:IKne3…") into a map.
Malformed entries are skipped rather than raising: a typo in .env should
cost that one voice, not the whole feature."""
voices: dict[str, str] = {}
for chunk in str(spec or "").split(","):
name, separator, voice_id = chunk.partition(":")
name, voice_id = name.strip().lower(), voice_id.strip()
if separator and name and voice_id:
voices[name] = voice_id
return voices
def resolve(
action: dict,
*,
voices: Optional[dict] = None,
self_voice: str = "",
) -> list[dict]:
"""Turn parsed lines into the API's `inputs`, resolving names to ids.
*self_voice* is the pet's current voice (which may be a `speak_as` pick,
not the configured default), so "self" tracks whoever Bolt sounds like
right now."""
known = dict(voices or {})
resolved: list[dict] = []
for index, line in enumerate(action.get("lines") or [], start=1):
name = str(line.get("voice") or "self")
key = name.lower()
if key in SELF_NAMES:
voice_id = self_voice
if not voice_id:
raise DialogueError(
"no voice is configured for the pet itself — set "
"ELEVENLABS_VOICE_ID, or name a voice from DIALOGUE_VOICES"
)
elif key in known:
voice_id = known[key]
elif _VOICE_ID_RE.match(name):
voice_id = name # a raw id pasted straight from the voice library
else:
available = ", ".join(sorted(known) + list(SELF_NAMES[:1])) or "self"
raise DialogueError(
f"line {index}: unknown voice {name!r}. Known names: {available}. "
"Use one of those, 'self' for your own voice, or a raw voice id."
)
resolved.append({"text": str(line.get("text") or ""), "voice_id": voice_id})
check_limits(resolved)
return resolved
def check_limits(inputs: Iterable[dict], *, max_voices: int = MAX_VOICES,
max_chars: int = MAX_CHARS) -> None:
"""Enforce the API's own limits before spending a request on a 422."""
entries = list(inputs)
if not entries:
raise DialogueError("no lines to speak")
distinct = {entry["voice_id"] for entry in entries}
if len(distinct) > max_voices:
raise DialogueError(
f"{len(distinct)} different voices — the limit is {max_voices} per scene"
)
total = sum(len(entry["text"]) for entry in entries)
if total > max_chars:
raise DialogueError(
f"{total} characters — the limit is {max_chars} per scene. "
"Split it into two dialoguectl calls."
)
def spoken_text(action: dict) -> str:
"""The scene as readable text, for the speech bubble and the transcript.
Delivery tags are stripped: `[cheerfully]` is a stage direction for the
model, not something to show (or, via tts.speak's sanitizer, to read out)."""
parts = []
for line in action.get("lines") or []:
text = re.sub(r"\[[^\]]{1,40}\]", " ", str(line.get("text") or ""))
text = " ".join(text.split())
if text:
parts.append(text)
return " ".join(parts)
def describe(action: dict, *, played: bool = True) -> str:
"""The tool-result string handed back to the server."""
lines = action.get("lines") or []
voices = sorted({str(line.get("voice") or "self") for line in lines})
if not played:
return f"[dialogue] not played ({len(lines)} lines)"
return (
f"[dialogue] played {len(lines)} line{'s' if len(lines) != 1 else ''} "
f"in {len(voices)} voice{'s' if len(voices) != 1 else ''}: {', '.join(voices)}"
)
+25 -1
View File
@@ -44,9 +44,17 @@ HELP = (
"petctl emote <" + "|".join(EMOTES) + ">\n"
"petctl say <text>\n"
"petctl wander on|off\n"
"petctl nap on|off"
"petctl nap on|off\n"
"petctl voice reset\n"
"petctl self_restart [why]"
)
# `petctl voice` only ever goes one way: back to the configured voice. Picking
# a *different* one is the server's job (its speak_as reply marker), and it
# already knows how — what it has no way to say is "never mind, be yourself
# again", because it was never told which voice that is.
VOICE_RESETS = ("reset", "default", "normal", "own", "back", "mine", "yours")
class ActionError(Exception):
"""Bad petctl syntax — reported back to the server as command output."""
@@ -129,6 +137,22 @@ def parse(command: str) -> Optional[dict]:
raise ActionError("wander needs on or off")
return {"action": "wander", "enabled": _bool_arg(args[0])}
if verb == "voice":
target = (args[0].lower() if args else "reset").lstrip("-")
if target not in VOICE_RESETS:
raise ActionError(
f"can't set a voice from petctl (got {args[0]!r}); "
"use the speak_as reply marker to pick one. "
"petctl voice reset goes back to the default voice."
)
return {"action": "voice", "voice": "default"}
if verb in ("self_restart", "restart", "reboot"):
# Free text, not a fixed grammar: the argument is a note to the pet's
# *next* process about why it died and what to look at when it comes
# back, so anything the model wants to tell future-itself is valid.
return {"action": "self_restart", "reason": " ".join(args).strip()}
if verb in ("nap", "sleep", "dnd"):
if not args:
raise ActionError("nap needs on or off")
+235
View File
@@ -0,0 +1,235 @@
"""`petctl self_restart` — the pet restarting itself, and remembering why.
Bolt can already edit this repo through `filectl` and run commands through the
shell relay, which means he can change the pet's own code. What he could not
do is *see the result*: the running process keeps the old modules in memory,
so an edit is invisible until somebody restarts the pet by hand, and by then
the conversation that motivated it is over. That makes the edit-test-review
loop a human errand.
This closes the loop. The tricky part is that the thing being asked to report
back is the thing that dies, so the mechanism is built around three problems:
1. **The turn must survive.** A restart mid-turn would kill the HTTP tool
relay before the result was posted, and the server would sit waiting until
it timed out the conversation lost, with no explanation. So the command
only *arms* the restart: it returns immediately, the turn finishes and Bolt
speaks his reply, and the restart happens after (see
`controller._maybe_self_restart`), exactly like the updater's "only between
turns" rule.
2. **A broken edit must not be fatal.** Before anything is armed, the new code
is imported in a *subprocess* (`preflight`) this process still holds the
old modules, so importing here would prove nothing. A syntax error comes
back as the command's output, in the same turn, and nothing restarts. That
is the difference between "Bolt broke the pet and lost his own way to fix
it" and "Bolt got a traceback and tried again".
3. **The reason must outlive the process.** The context (why, what to check,
which version, when) is written to disk before exec and read on the way
back up, so the new process can open with "I'm back — you asked me to check
X" instead of amnesia. That report goes to the server as a normal turn, so
Bolt sees the result of his own change and can carry on.
A loop guard bounds the worst case: `MAX_RESTARTS` inside `WINDOW_SECONDS`
and further self-restarts are refused with a reason, so an edit-restart-crash
cycle stops on its own rather than spinning the process forever.
Pure-ish and injectable throughout (paths, clock, subprocess runner) so the
whole thing is testable without ever restarting anything.
"""
from __future__ import annotations
import json
import os
import subprocess
import sys
import tempfile
import time
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any, Callable, Optional
from . import config
# Lives in the cache dir, not the repo: it is transient state about *this*
# machine's process, and it must never end up in a git diff of the checkout
# Bolt is editing.
DEFAULT_STATE_PATH = Path.home() / ".cache" / "bolt-pet" / "restart_context.json"
# Loop guard. Deliberately small: a healthy edit-check cycle is one restart
# per change, and anything hammering past this is a crash loop, not work.
MAX_RESTARTS = int(os.environ.get("SELF_RESTART_MAX", "5"))
WINDOW_SECONDS = float(os.environ.get("SELF_RESTART_WINDOW_SECONDS", "900"))
# What the preflight subprocess imports. `ui.app` pulls in the widest slice of
# the package (Qt, controller, audio, every helper), so if this imports, a
# restart will at least reach the event loop.
_PREFLIGHT_IMPORT = "import bolt_pet, bolt_pet.controller, bolt_pet.ui.app"
class RestartError(Exception):
"""A refused restart — reported back to the server as command output."""
@dataclass
class RestartContext:
"""What the dying process wants the next one to know."""
reason: str = ""
verify: str = ""
armed_at: float = 0.0
version: str = ""
session: str = ""
recent: list = field(default_factory=list)
restarts: list = field(default_factory=list) # timestamps, for the loop guard
def as_dict(self) -> dict[str, Any]:
return asdict(self)
def _now() -> float:
return time.time()
def load(path: Optional[Path] = None) -> Optional[RestartContext]:
"""Read the context left by a previous process, or None."""
target = Path(path or DEFAULT_STATE_PATH)
try:
data = json.loads(target.read_text(encoding="utf-8"))
except (FileNotFoundError, json.JSONDecodeError, OSError):
return None
if not isinstance(data, dict):
return None
known = {field_name for field_name in RestartContext().as_dict()}
return RestartContext(**{k: v for k, v in data.items() if k in known})
def save(context: RestartContext, path: Optional[Path] = None) -> None:
"""Persist the context atomically — a half-written file on the way out
would make the next process start confused instead of oriented."""
target = Path(path or DEFAULT_STATE_PATH)
target.parent.mkdir(parents=True, exist_ok=True)
descriptor, temp_path = tempfile.mkstemp(dir=target.parent, prefix=".restart_", suffix=".tmp")
try:
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
json.dump(context.as_dict(), handle, ensure_ascii=False, indent=1)
os.replace(temp_path, target)
except BaseException:
try:
os.unlink(temp_path)
except OSError:
pass
raise
def clear(path: Optional[Path] = None) -> None:
"""Consume the context. Called once it has been reported, so the pet
doesn't announce the same restart every time it starts."""
try:
Path(path or DEFAULT_STATE_PATH).unlink()
except (FileNotFoundError, OSError):
pass
def recent_restarts(context: Optional[RestartContext], *, now: Optional[float] = None) -> list:
current = now if now is not None else _now()
stamps = list((context.restarts if context else []) or [])
return [stamp for stamp in stamps if current - float(stamp) <= WINDOW_SECONDS]
def check_loop_guard(context: Optional[RestartContext], *, now: Optional[float] = None) -> None:
"""Refuse to restart if we've already done it too many times recently."""
stamps = recent_restarts(context, now=now)
if len(stamps) >= MAX_RESTARTS:
raise RestartError(
f"refusing: {len(stamps)} self-restarts in the last "
f"{int(WINDOW_SECONDS / 60)} minutes. Something is looping — fix the "
"cause, or wait for the window to clear before trying again."
)
def preflight(
repo: Optional[Path] = None,
run: Optional[Callable[..., Any]] = None,
timeout: float = 120.0,
) -> None:
"""Import the current source in a subprocess; raise if it's broken.
This process has the *old* modules loaded, so importing in-process would
happily succeed on a file that no longer parses. Mirrors
`updater._smoke_test`, and exists for the same reason: never hand the
session to code that can't start."""
runner = run or subprocess.run
root = Path(repo or config.HERE)
try:
completed = runner(
[sys.executable, "-c", _PREFLIGHT_IMPORT],
cwd=str(root), capture_output=True, text=True, timeout=timeout,
env={**os.environ, "QT_QPA_PLATFORM": "offscreen"}, # no display needed to import
)
except Exception as exc: # subprocess itself failed to run
raise RestartError(f"couldn't run the preflight import check: {exc}") from exc
if completed.returncode != 0:
detail = (completed.stderr or completed.stdout or "").strip()
raise RestartError(
"the current code does not import, so restarting would leave you with "
f"nothing running. Fix this first:\n{detail[-800:]}"
)
def arm(
reason: str,
*,
verify: str = "",
version: str = "",
session: str = "",
recent: Optional[list] = None,
path: Optional[Path] = None,
now: Optional[float] = None,
) -> RestartContext:
"""Record why we're about to die, carrying the restart history forward."""
current = now if now is not None else _now()
previous = load(path)
context = RestartContext(
reason=" ".join(str(reason or "").split())[:400],
verify=" ".join(str(verify or "").split())[:400],
armed_at=current,
version=str(version or ""),
session=str(session or ""),
recent=list(recent or [])[-6:],
restarts=recent_restarts(previous, now=current) + [current],
)
save(context, path)
return context
def report(
context: RestartContext,
*,
version: str = "",
now: Optional[float] = None,
) -> str:
"""The message the new process sends the server on the way up.
Phrased as Bolt reporting to himself, because that is what it is: the
server sees it as an ordinary turn, and the reply comes back through the
normal pipeline which is what lets "restart and check X" finish as a
sentence spoken out loud."""
current = now if now is not None else _now()
took = max(0.0, current - float(context.armed_at or current))
lines = [
"[pet self-restart] I restarted myself and I'm back up.",
f"- reason: {context.reason or 'not recorded'}",
f"- took: {took:.1f}s",
f"- version now running: {version or 'unknown'}"
+ (f" (was {context.version})" if context.version and context.version != version else ""),
]
if context.verify:
lines.append(f"- you wanted to check: {context.verify}")
if context.recent:
lines.append("- what we were doing before: " + " | ".join(str(x)[:120] for x in context.recent))
lines.append(
"The new code is loaded and running. If you wanted to verify something, "
"check it now (filectl to read, command to test) and tell the user what you found."
)
return "\n".join(lines)
+23 -3
View File
@@ -9,6 +9,11 @@ memory, tools, and persona as Discord chat and the Linux voice client:
... -> POST /desk/tool_result (repeat until the server sends a reply)
reply <- returned to caller
A reply can also carry a voice (`voice_id`/`voice_name`), which is how the
server's `speak_as` marker reaches us: Bolt searched the ElevenLabs voice
library, picked one, and tagged the reply with it the client is what
actually speaks in it. See `Reply` and controller._apply_voice.
Kept dependency-free beyond `requests` so it's easy to unit test with mocks.
"""
@@ -16,7 +21,7 @@ from __future__ import annotations
import subprocess
from pathlib import Path
from typing import Callable, Optional
from typing import Callable, NamedTuple, Optional
import requests
@@ -29,6 +34,17 @@ class ServerError(Exception):
"""Raised when the server responds with an error payload or unreachable."""
class Reply(NamedTuple):
"""One final reply from the desk API. *voice_id* is set only when the
server tagged this reply with a `speak_as` voice; *voice_name* is the
human-readable name that came with it (may be empty even when the id
isn't). Both empty means "say it in the usual voice"."""
text: str
voice_id: str = ""
voice_name: str = ""
def _headers() -> dict:
return {"X-Desk-Api-Key": config.API_KEY}
@@ -74,7 +90,7 @@ def converse(
text: str,
on_command: Callable[[str], str] = run_local_command,
timeout: float = 120.0,
) -> str:
) -> Reply:
"""Send one turn of conversation to the desk API, relaying any commands
the server sends back until it produces a final reply.
@@ -111,7 +127,11 @@ def converse(
raise ServerError(f"couldn't reach the server during tool relay: {exc}") from exc
if payload.get("type") == "reply":
return str(payload.get("text") or "")
return Reply(
text=str(payload.get("text") or ""),
voice_id=str(payload.get("voice_id") or ""),
voice_name=str(payload.get("voice_name") or ""),
)
raise ServerError(str(payload.get("error") or "unknown server response"))
+6 -1
View File
@@ -26,11 +26,16 @@ class PetState(str, Enum):
# make the pet speak unprompted — a reminder firing, a nudge from the server
# — without the user having said anything first, so there's no preceding
# LISTENING/THINKING leg for that turn.
#
# TALKING -> THINKING is the mirror case: a `dialoguectl` scene is played
# *mid-turn*, while the server is still waiting on the tool result, so the pet
# talks and then goes back to waiting rather than falling to IDLE (which would
# make it look like the turn had ended).
_TRANSITIONS: dict[PetState, set[PetState]] = {
PetState.IDLE: {PetState.LISTENING, PetState.TALKING, PetState.ERROR},
PetState.LISTENING: {PetState.THINKING, PetState.IDLE, PetState.ERROR},
PetState.THINKING: {PetState.TALKING, PetState.IDLE, PetState.ERROR},
PetState.TALKING: {PetState.IDLE, PetState.ERROR},
PetState.TALKING: {PetState.IDLE, PetState.THINKING, PetState.ERROR},
PetState.ERROR: {PetState.IDLE},
}
+4
View File
@@ -80,6 +80,7 @@ def run() -> int:
on_set_nap=_set_nap,
on_show_history=history_window.show_refreshed,
on_show_wake_tuner=tuner_window.show_refreshed,
on_reset_voice=controller.reset_voice,
)
def _handle_napping(napping: bool) -> None:
@@ -87,6 +88,9 @@ def run() -> int:
tray.set_napping(napping)
controller.napping.connect(_handle_napping)
# The server can hand Bolt a different voice mid-conversation (speak_as);
# the tray is where you get his own back.
controller.voice_changed.connect(tray.set_voice)
# Push-to-talk: a global hook, because the pet window never has focus.
# request_talk_now() only sets a threading.Event, so it's safe to call
+23 -1
View File
@@ -1,6 +1,7 @@
"""System tray icon — the pet window is frameless with no taskbar entry, so
this menu is the only always-available way to control or exit it: talk now,
mute, wander, click-through, nap, history, wake-word tuning, quit.
mute, wander, click-through, nap, history, wake-word tuning, voice reset,
quit.
Every entry is a plain callback passed in by ui/app.py; this file knows
nothing about the controller or the pet window.
@@ -46,6 +47,7 @@ class PetTray(QSystemTrayIcon):
on_set_nap: Optional[Callable[[bool], None]] = None,
on_show_history: Optional[Callable[[], None]] = None,
on_show_wake_tuner: Optional[Callable[[], None]] = None,
on_reset_voice: Optional[Callable[[], None]] = None,
parent=None,
):
super().__init__(_make_icon(muted=False), parent)
@@ -102,6 +104,16 @@ class PetTray(QSystemTrayIcon):
tuner_action.triggered.connect(on_show_wake_tuner)
menu.addAction(tuner_action)
# Only ever enabled while a server-picked voice (speak_as) is in use —
# it's the way back from "talk like a pirate", which nothing else
# undoes short of a restart.
self._voice_action = None
if on_reset_voice is not None:
self._voice_action = QAction("Use default voice", menu)
self._voice_action.setEnabled(False)
self._voice_action.triggered.connect(on_reset_voice)
menu.addAction(self._voice_action)
menu.addSeparator()
quit_action = QAction("Quit", menu)
quit_action.triggered.connect(on_quit)
@@ -115,6 +127,16 @@ class PetTray(QSystemTrayIcon):
self._mute_action.setChecked(self._muted)
self._refresh_icon()
def set_voice(self, voice: str) -> None:
"""Reflect the voice the controller is speaking in — a name (or id)
when the server picked one, "" for Bolt's own."""
if self._voice_action is None:
return
self._voice_action.setEnabled(bool(voice))
self._voice_action.setText(
f"Use default voice (now: {voice})" if voice else "Use default voice"
)
def set_napping(self, napping: bool) -> None:
"""Reflect a nap the *controller* decided on (quiet hours, fullscreen,
or a petctl command) not just ones clicked here."""
+3 -3
View File
@@ -41,10 +41,10 @@ def test_full_turn_happy_path(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's the weather")
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: "sunny and 72")
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: controller_mod.server_client.Reply("sunny and 72"))
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
ctrl._handle_conversation_turn()
@@ -150,7 +150,7 @@ def test_heartbeat_speaks_a_pending_announcement_when_idle(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: "don't forget your 3pm")
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
said = _capture(ctrl.said)
ctrl._maybe_heartbeat()
+306 -14
View File
@@ -15,6 +15,7 @@ from PySide6.QtWidgets import QApplication
from bolt_pet import controller as controller_mod
from bolt_pet.notifications import Notification
from bolt_pet.server_client import Reply
from bolt_pet.state import PetState
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
@@ -159,7 +160,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch
ctrl._barge_in = detector
logs = _capture(ctrl.log)
def interrupted_playback(text, on_error=None, should_stop=None):
def interrupted_playback(text, on_error=None, should_stop=None, voice_id=None):
# What really happens: frames get scored during playback, then one
# clears the threshold and playback aborts.
detector._frames, detector._peak, detector._last = 7, 0.81, 0.81
@@ -179,7 +180,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch
def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
logs = _capture(ctrl.log)
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: False) # interrupted
lambda text, on_error=None, should_stop=None, voice_id=None: False) # interrupted
ctrl._speak("a very long explanation")
@@ -189,7 +190,7 @@ def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
def test_uninterrupted_playback_does_not_queue_a_turn(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
ctrl._speak("short answer")
assert not ctrl._talk_now.is_set()
@@ -201,7 +202,7 @@ def spoke(monkeypatch):
"""Playback that always completes, so only the follow-up rule decides
whether another turn is queued."""
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
def test_a_reply_ending_in_a_question_keeps_listening(spoke, ctrl):
@@ -289,7 +290,7 @@ def test_follow_up_can_be_turned_off(spoke, monkeypatch, ctrl):
def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: False)
lambda text, on_error=None, should_stop=None, voice_id=None: False)
ctrl._follow_ups = 3
ctrl._speak("a very long explanation")
@@ -300,7 +301,7 @@ def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
def test_speech_is_recorded_in_the_history(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
ctrl._speak("**bold** reply")
assert ctrl.history.last().text == "**bold** reply" # raw, for copy/paste
@@ -314,10 +315,10 @@ def test_the_active_window_rides_along_with_the_utterance(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.screen_context, "context_for",
lambda text: f"{text}\n\n[on screen right now: app.py]")
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or "that's a KeyError")
lambda text, on_command=None: sent.append(text) or Reply("that's a KeyError"))
ctrl._handle_conversation_turn()
@@ -370,10 +371,10 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "you awake?")
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: "always")
lambda text, on_command=None: Reply("always"))
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
ctrl.set_napping(True)
ctrl._handle_conversation_turn()
@@ -388,9 +389,9 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
sent, spoken = [], []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or "your build is green")
lambda text, on_command=None: sent.append(text) or Reply("your build is green"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
ctrl._drain_notifications()
@@ -506,9 +507,9 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "send me that file")
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: "it's on the way")
lambda text, on_command=None: Reply("it's on the way"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files",
lambda: [{"id": "abc", "name": "notes.txt", "size": 2}])
@@ -518,3 +519,294 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_
ctrl._handle_conversation_turn()
assert (tmp_path / "notes.txt").read_bytes() == b"hi"
# ── server-picked voice (the desk API's speak_as marker) ────────────────────
def _voice_turn(monkeypatch, ctrl, reply, said="talk like a pirate"):
"""Run one full conversation turn whose reply is *reply*, returning the
voice_id each tts.speak() call was given."""
voices = []
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said)
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: reply)
monkeypatch.setattr(
controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None:
voices.append(voice_id) or True,
)
ctrl._handle_conversation_turn()
return voices
def test_a_speak_as_reply_is_spoken_in_that_voice(monkeypatch, ctrl):
voices = _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
assert voices == ["VOICE1"]
assert ctrl.current_voice() == "Terence"
def test_the_picked_voice_sticks_for_later_replies(monkeypatch, ctrl):
"""The server tags one reply and doesn't keep the id in its history, so
it can't re-request the voice when you say "keep talking like that"."""
monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", True)
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
voices = _voice_turn(monkeypatch, ctrl, Reply("Still me."), said="and now?")
assert voices == ["VOICE1"]
def test_voice_stickiness_can_be_turned_off(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", False)
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
voices = _voice_turn(monkeypatch, ctrl, Reply("Back to normal."), said="and now?")
assert voices == [None]
assert ctrl.current_voice() == ""
def test_a_new_pick_replaces_the_old_one(monkeypatch, ctrl):
changes = _capture(ctrl.voice_changed)
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
voices = _voice_turn(monkeypatch, ctrl, Reply("こんにちは。", "VOICE2", "Asahi"),
said="say that in Japanese")
assert voices == ["VOICE2"]
assert changes == ["Terence", "Asahi"]
def test_resetting_the_voice_goes_back_to_the_default(monkeypatch, ctrl):
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
changes = _capture(ctrl.voice_changed)
ctrl.reset_voice()
assert ctrl.current_voice() == ""
assert changes == [""] # the tray's menu entry follows this signal
voices = _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi")
assert voices == [None]
def test_an_unnamed_voice_still_reports_something_resettable(monkeypatch, ctrl):
"""voice_name is optional server-side — falling back to the id keeps the
tray entry from reading "now: " with nothing after it."""
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1"))
assert ctrl.current_voice() == "VOICE1"
def test_petctl_voice_reset_returns_bolt_to_his_own_voice(monkeypatch, ctrl):
"""The server can pick a voice but can't ask for the default back — it
was never told what Bolt's own voice id is. This is how it asks."""
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
output = ctrl._handle_command("petctl voice reset")
assert ran == [] # never reaches a shell, like every other petctl verb
assert "Terence" in output # the server can't see the voice; tell it what changed
assert ctrl.current_voice() == ""
assert _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi") == [None]
def test_petctl_voice_reset_says_so_when_there_was_nothing_to_reset(ctrl):
assert "already" in ctrl._handle_command("petctl voice reset")
# ── dialoguectl (multi-voice scenes) ────────────────────────────────────────
def _dialogue_command(*lines):
import json
return "dialoguectl " + json.dumps({"lines": list(lines)})
def test_dialoguectl_never_reaches_the_shell(monkeypatch, ctrl):
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm",
lambda pcm, rate, should_stop=None: True)
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "[cheerfully] hi"}))
assert ran == []
assert "[dialogue] played 1 line" in output
def test_a_scene_shows_in_the_bubble_with_the_delivery_tags_stripped(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
said = _capture(ctrl.said)
ctrl._handle_command(_dialogue_command(
{"voice": "self", "text": "[cheerfully] Hello there!"},
{"voice": "narrator", "text": "[whispering] He is lying."},
))
assert said == ["Hello there! He is lying."]
assert ctrl.history.last().text == "Hello there! He is lying."
def test_a_mid_turn_scene_returns_to_thinking_not_idle(monkeypatch, ctrl):
"""The server is still waiting on the tool result, so the pet talks and
goes back to waiting dropping to IDLE would look like the turn ended."""
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
ctrl._state.transition(PetState.LISTENING)
ctrl._state.transition(PetState.THINKING)
states = _capture(ctrl.state_changed)
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert states == ["talking", "thinking"]
assert ctrl._state.state == PetState.THINKING
def test_the_scene_uses_a_voice_the_server_picked_with_speak_as(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
seen = {}
def capture(inputs, model_id=None, stability=None):
seen["inputs"] = inputs
return np.zeros(4, dtype=np.int16), 24000
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", capture)
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
ctrl._apply_voice(controller_mod.server_client.Reply("ok", "PICKEDvoice123456789", "Terence"))
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert seen["inputs"][0]["voice_id"] == "PICKEDvoice123456789"
def test_a_synthesis_failure_is_reported_back_for_bolt_to_retry(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
def boom(inputs, model_id=None, stability=None):
raise controller_mod.tts.TtsError("voice_id not found")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", boom)
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert "couldn't synthesize" in output and "voice_id not found" in output
assert ctrl._state.state == PetState.IDLE # nothing left half-transitioned
def test_a_bad_voice_name_comes_back_as_advice_not_an_exception(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
output = ctrl._handle_command(_dialogue_command({"voice": "wizard", "text": "hi"}))
assert "unknown voice" in output and "narrator" in output
def test_dialogue_can_be_switched_off_on_this_device(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "DIALOGUE", False)
called = []
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda *a, **k: called.append(1))
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert "disabled" in output and called == []
def test_talking_over_a_scene_is_reported_up_the_relay(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm",
lambda pcm, rate, should_stop=None: False) # barge-in
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert "interrupted" in output
# ── petctl self_restart ─────────────────────────────────────────────────────
def test_self_restart_arms_after_the_turn_rather_than_dying_mid_relay(monkeypatch, ctrl, tmp_path):
"""Restarting inline would kill the HTTP tool relay before the result was
posted, and the server would wait out its timeout on a turn that can never
finish. So the command returns, the turn completes, *then* the pet dies."""
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None)
restarts = _capture(ctrl.restart_requested)
output = ctrl._handle_command("petctl self_restart check the new dialogue code")
assert "restarting as soon as this turn finishes" in output
assert restarts == [] # nothing has happened yet
assert ctrl._maybe_self_restart() is True
assert restarts and "check the new dialogue code" in restarts[0]
def test_a_broken_edit_is_reported_instead_of_leaving_nothing_running(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
def boom(*args, **kwargs):
raise controller_mod.self_restart.RestartError(
"the current code does not import, so restarting would leave you with "
"nothing running. Fix this first:\nSyntaxError: invalid syntax"
)
monkeypatch.setattr(controller_mod.self_restart, "preflight", boom)
restarts = _capture(ctrl.restart_requested)
output = ctrl._handle_command("petctl self_restart try the new code")
assert "SyntaxError" in output and "refused" in output
assert ctrl._maybe_self_restart() is False
assert restarts == []
def test_a_second_restart_request_in_one_turn_is_a_no_op(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None)
ctrl._handle_command("petctl self_restart first")
assert "already armed" in ctrl._handle_command("petctl self_restart second")
def test_self_restart_can_be_switched_off_on_this_device(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "SELF_RESTART", False)
checked = []
monkeypatch.setattr(controller_mod.self_restart, "preflight",
lambda *a, **k: checked.append(1))
assert "disabled" in ctrl._handle_command("petctl self_restart go")
assert checked == []
def test_coming_back_up_reports_to_the_server_and_speaks_the_reply(monkeypatch, ctrl, tmp_path):
"""The half that makes it a loop: the new process tells Bolt it's back and
why, and his answer is spoken like any other turn."""
state = tmp_path / "ctx.json"
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", state)
controller_mod.self_restart.arm("check the walk cycle", version="0.2.3",
path=state, now=1000.0)
sent, spoken = [], []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or Reply("Good, it's up."))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None:
spoken.append(text) or True)
ctrl._report_self_restart()
assert "[pet self-restart]" in sent[0] and "check the walk cycle" in sent[0]
assert spoken == ["Good, it's up."]
# Consumed, so the next start doesn't announce the same restart again.
assert controller_mod.self_restart.load(state) is None
def test_an_ordinary_start_reports_nothing(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "none.json")
called = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: called.append(text))
ctrl._report_self_restart()
assert called == []
+225
View File
@@ -0,0 +1,225 @@
"""`dialoguectl` — multi-voice scene parsing, voice resolution, API limits,
and the request the ElevenLabs Text to Dialogue endpoint actually gets.
Pure logic plus one mocked HTTP call: no audio device, no network, no display.
"""
import sys
from pathlib import Path
from unittest.mock import MagicMock, patch
import numpy as np
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import dialogue
from bolt_pet.audio import tts
SELF_ID = "aaorr6ZHIL88gEexu7dC"
NARRATOR_ID = "9BWtsMINqrJLrRacOk9x"
VILLAIN_ID = "IKne3meq5aSn9XLyUdCD"
VOICES = {"narrator": NARRATOR_ID, "villain": VILLAIN_ID}
def _scene(*lines):
return '{"lines": [' + ", ".join(lines) + "]}"
# ── parsing ─────────────────────────────────────────────────────────────────
def test_non_dialogue_commands_are_left_alone():
assert dialogue.parse("ls -la") is None
assert dialogue.parse('filectl {"op": "list"}') is None
assert dialogue.parse("") is None
# "dialogues" must not be mistaken for the "dialogue" prefix
assert dialogue.parse("dialogues --list") is None
def test_a_scene_parses_into_lines():
action = dialogue.parse(
'dialoguectl ' + _scene(
'{"voice": "self", "text": "[cheerfully] Hello, how are you?"}',
'{"voice": "villain", "text": "[stuttering] I am... fine."}',
)
)
assert action["action"] == "dialogue"
assert [line["voice"] for line in action["lines"]] == ["self", "villain"]
assert action["lines"][0]["text"].startswith("[cheerfully]")
def test_the_elevenlabs_field_names_are_accepted_too():
"""The model has read that API; copying its shape is the obvious thing to
try, so 'inputs'/'voice_id' work as well as 'lines'/'voice'."""
action = dialogue.parse(
'dialoguectl {"inputs": [{"voice_id": "%s", "text": "hi"}]}' % NARRATOR_ID
)
assert action["lines"] == [{"voice": NARRATOR_ID, "text": "hi"}]
def test_a_line_with_no_voice_defaults_to_the_pet_itself():
action = dialogue.parse('dialoguectl {"lines": [{"text": "just me talking"}]}')
assert action["lines"][0]["voice"] == "self"
def test_truncated_json_explains_the_one_line_rule():
"""The real failure mode: the server's command extractor stops at the
first newline, so a multi-line payload arrives cut in half. The error has
to name the cause, since Bolt is the one who has to fix it."""
with pytest.raises(dialogue.DialogueError, match="one line"):
dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"')
def test_an_empty_or_shapeless_payload_is_rejected():
with pytest.raises(dialogue.DialogueError, match="needs a JSON argument"):
dialogue.parse("dialoguectl")
with pytest.raises(dialogue.DialogueError, match="non-empty"):
dialogue.parse('dialoguectl {"lines": []}')
with pytest.raises(dialogue.DialogueError, match="no text"):
dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": " "}]}')
def test_optional_model_and_stability_ride_along():
action = dialogue.parse(
'dialoguectl {"model_id": "eleven_v3", "stability": 0.8, '
'"lines": [{"text": "hi"}]}'
)
assert action["model"] == "eleven_v3"
assert action["stability"] == 0.8
# ── voice resolution ────────────────────────────────────────────────────────
def test_named_voices_resolve_from_the_configured_cast():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "narrator", "text": "Once upon a time."}',
'{"voice": "villain", "text": "Not this again."}',
))
inputs = dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID)
assert [entry["voice_id"] for entry in inputs] == [NARRATOR_ID, VILLAIN_ID]
def test_self_tracks_the_voice_the_pet_is_currently_using():
"""A scene featuring Bolt should sound like whoever Bolt currently is —
including a voice the server picked mid-conversation with speak_as."""
action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}')
picked = "VOICEfromSPEAKas1234"
assert dialogue.resolve(action, self_voice=picked)[0]["voice_id"] == picked
def test_a_raw_voice_id_passes_straight_through():
action = dialogue.parse('dialoguectl {"lines": [{"voice": "%s", "text": "hi"}]}' % NARRATOR_ID)
assert dialogue.resolve(action, self_voice=SELF_ID)[0]["voice_id"] == NARRATOR_ID
def test_an_unknown_name_lists_what_is_available():
action = dialogue.parse('dialoguectl {"lines": [{"voice": "wizard", "text": "hi"}]}')
with pytest.raises(dialogue.DialogueError) as excinfo:
dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID)
message = str(excinfo.value)
assert "wizard" in message and "narrator" in message and "villain" in message
def test_self_without_a_configured_voice_says_so():
action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}')
with pytest.raises(dialogue.DialogueError, match="ELEVENLABS_VOICE_ID"):
dialogue.resolve(action, self_voice="")
def test_the_voice_map_parser_skips_typos_instead_of_dying():
voices = dialogue.parse_voice_map(f"narrator:{NARRATOR_ID}, broken-entry, villain:{VILLAIN_ID}")
assert voices == {"narrator": NARRATOR_ID, "villain": VILLAIN_ID}
assert dialogue.parse_voice_map("") == {}
# ── API limits, enforced before the request goes out ────────────────────────
def test_too_many_distinct_voices_is_refused_locally():
inputs = [{"text": "hi", "voice_id": f"voice{index:015d}"} for index in range(11)]
with pytest.raises(dialogue.DialogueError, match="limit is 10"):
dialogue.check_limits(inputs)
def test_an_over_long_scene_is_refused_with_advice():
inputs = [{"text": "x" * 1100, "voice_id": SELF_ID} for _ in range(2)]
with pytest.raises(dialogue.DialogueError) as excinfo:
dialogue.check_limits(inputs)
assert "Split it" in str(excinfo.value) # actionable, since Bolt reads this
# ── display / reporting ─────────────────────────────────────────────────────
def test_delivery_tags_are_stripped_from_what_the_bubble_shows():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "self", "text": "[cheerfully] Hello there!"}',
'{"voice": "narrator", "text": "[whispering] He is lying."}',
))
assert dialogue.spoken_text(action) == "Hello there! He is lying."
def test_the_relay_report_names_the_cast():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "self", "text": "one"}', '{"voice": "narrator", "text": "two"}',
))
assert dialogue.describe(action) == "[dialogue] played 2 lines in 2 voices: narrator, self"
# ── the HTTP request ────────────────────────────────────────────────────────
def _pcm_response(samples=(1, 2, 3, 4)):
response = MagicMock()
response.content = np.array(samples, dtype=np.int16).tobytes()
response.raise_for_status = MagicMock()
return response
def test_the_request_matches_the_text_to_dialogue_api(monkeypatch):
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
monkeypatch.setattr(tts.config, "TTS_SAMPLE_RATE", 24000)
monkeypatch.setattr(tts.config, "DIALOGUE_MODEL_ID", "eleven_v3")
inputs = [
{"text": "[cheerfully] Hello", "voice_id": NARRATOR_ID},
{"text": "[stuttering] H-hi", "voice_id": VILLAIN_ID},
]
with patch.object(tts.requests, "post", return_value=_pcm_response()) as post:
pcm, rate = tts.synthesize_dialogue(inputs)
assert rate == 24000 and pcm.tolist() == [1, 2, 3, 4]
args, kwargs = post.call_args
assert args[0] == "https://api.elevenlabs.io/v1/text-to-dialogue"
assert kwargs["params"] == {"output_format": "pcm_24000"}
assert kwargs["headers"] == {"xi-api-key": "test-key"}
assert kwargs["json"]["inputs"] == inputs
assert kwargs["json"]["model_id"] == "eleven_v3"
assert "settings" not in kwargs["json"] # omitted unless asked for
def test_stability_is_only_sent_when_given(monkeypatch):
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
with patch.object(tts.requests, "post", return_value=_pcm_response()) as post:
tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}], stability=0.3)
assert post.call_args.kwargs["json"]["settings"] == {"stability": 0.3}
def test_a_rejected_request_surfaces_what_the_api_said(monkeypatch):
"""The API explains refusals in the body; Bolt reads this through the tool
relay, so it has to reach him rather than being flattened to '422'."""
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
failure = MagicMock()
failure.text = '{"detail": "voice_id not found"}'
error = Exception("422 Client Error")
error.response = failure
response = MagicMock()
response.raise_for_status = MagicMock(side_effect=error)
with patch.object(tts.requests, "post", return_value=response):
with pytest.raises(tts.TtsError, match="voice_id not found"):
tts.synthesize_dialogue([{"text": "hi", "voice_id": "nope"}])
def test_no_api_key_fails_before_the_request(monkeypatch):
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "")
with patch.object(tts.requests, "post") as post:
with pytest.raises(tts.TtsError):
tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}])
post.assert_not_called()
+14
View File
@@ -69,3 +69,17 @@ def test_describe_is_reported_back_to_the_server():
assert "top-left" in pet_actions.describe({"action": "move", "anchor": "top-left"})
assert "wave" in pet_actions.describe({"action": "emote", "emote": "wave"})
assert pet_actions.describe({"action": "help"}) == pet_actions.HELP
def test_voice_reset_parses_with_or_without_the_word_reset():
assert pet_actions.parse("petctl voice reset") == {"action": "voice", "voice": "default"}
assert pet_actions.parse("petctl voice default") == {"action": "voice", "voice": "default"}
assert pet_actions.parse("petctl voice") == {"action": "voice", "voice": "default"}
def test_petctl_cannot_be_used_to_pick_a_voice():
"""Choosing a voice is the server's job (speak_as) — it has the voice
library. petctl only ever undoes one, so an attempt to set a voice here
is pointed back at the marker that works."""
with pytest.raises(pet_actions.ActionError, match="speak_as"):
pet_actions.parse("petctl voice Terence")
+158
View File
@@ -0,0 +1,158 @@
"""`petctl self_restart` — the pet restarting itself and remembering why.
Everything here runs against a temp context file and a fake subprocess runner,
so the tests exercise the arming/preflight/report logic without any process
actually dying.
"""
import subprocess
import sys
from pathlib import Path
from types import SimpleNamespace
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import pet_actions, self_restart
@pytest.fixture
def state(tmp_path):
return tmp_path / "restart_context.json"
def _ok_run(*args, **kwargs):
return SimpleNamespace(returncode=0, stdout="", stderr="")
def _broken_run(*args, **kwargs):
return SimpleNamespace(
returncode=1, stdout="",
stderr=' File "bolt_pet/controller.py", line 42\n def _speak(\nSyntaxError: invalid syntax',
)
# ── parsing ─────────────────────────────────────────────────────────────────
def test_self_restart_parses_with_a_free_text_reason():
action = pet_actions.parse("petctl self_restart check the new walk cycle loads")
assert action == {
"action": "self_restart", "reason": "check the new walk cycle loads",
}
def test_self_restart_needs_no_reason_and_accepts_aliases():
assert pet_actions.parse("petctl self_restart")["reason"] == ""
assert pet_actions.parse("petctl restart")["action"] == "self_restart"
assert pet_actions.parse("petctl reboot")["action"] == "self_restart"
def test_self_restart_is_listed_in_the_help():
assert "self_restart" in pet_actions.HELP
# ── preflight ───────────────────────────────────────────────────────────────
def test_preflight_passes_when_the_code_imports():
self_restart.preflight(run=_ok_run) # no exception
def test_preflight_hands_back_the_traceback_instead_of_dying(state):
"""The whole point: a syntax error Bolt just introduced comes back as
something he can read and fix, in the same turn, with the pet still up."""
with pytest.raises(self_restart.RestartError) as excinfo:
self_restart.preflight(run=_broken_run)
message = str(excinfo.value)
assert "does not import" in message
assert "SyntaxError" in message and "controller.py" in message
def test_preflight_runs_the_import_in_a_subprocess_not_here():
"""This process holds the *old* modules, so an in-process import would
pass on a file that no longer parses."""
seen = {}
def capture(cmd, **kwargs):
seen["cmd"], seen["kwargs"] = cmd, kwargs
return SimpleNamespace(returncode=0, stdout="", stderr="")
self_restart.preflight(run=capture)
assert seen["cmd"][0] == sys.executable
assert "import bolt_pet" in seen["cmd"][2]
assert seen["kwargs"]["env"]["QT_QPA_PLATFORM"] == "offscreen" # imports need no display
def test_a_subprocess_that_cannot_even_run_is_reported(monkeypatch):
def explode(*args, **kwargs):
raise OSError("no python here")
with pytest.raises(self_restart.RestartError, match="couldn't run the preflight"):
self_restart.preflight(run=explode)
# ── context across the restart ──────────────────────────────────────────────
def test_arming_persists_the_reason_for_the_next_process(state):
self_restart.arm("check the sprite frames load", version="0.2.3",
session="pet-desktop", recent=["you: reload the sprites"],
path=state, now=1000.0)
revived = self_restart.load(state)
assert revived.reason == "check the sprite frames load"
assert revived.version == "0.2.3"
assert revived.recent == ["you: reload the sprites"]
assert revived.restarts == [1000.0]
def test_no_context_means_a_normal_start(state):
assert self_restart.load(state) is None
def test_a_corrupt_context_file_is_ignored_not_fatal(state):
state.write_text("{not json at all", encoding="utf-8")
assert self_restart.load(state) is None
def test_clearing_the_context_stops_it_being_re_announced(state):
self_restart.arm("once", path=state, now=1000.0)
self_restart.clear(state)
assert self_restart.load(state) is None
self_restart.clear(state) # clearing twice is not an error
def test_the_report_says_what_happened_and_what_to_check(state):
context = self_restart.arm(
"verify the dialogue command works", verify="verify the dialogue command works",
version="0.2.3", recent=["you: try a scene"], path=state, now=1000.0,
)
text = self_restart.report(context, version="0.2.4", now=1004.5)
assert "I restarted myself" in text
assert "verify the dialogue command works" in text
assert "4.5s" in text
assert "0.2.4" in text and "was 0.2.3" in text
assert "you: try a scene" in text
# ── loop guard ──────────────────────────────────────────────────────────────
def test_restart_history_accumulates_across_restarts(state):
self_restart.arm("one", path=state, now=1000.0)
self_restart.arm("two", path=state, now=1100.0)
assert self_restart.load(state).restarts == [1000.0, 1100.0]
def test_too_many_restarts_in_the_window_is_refused(state):
now = 1000.0
for index in range(self_restart.MAX_RESTARTS):
self_restart.arm(f"attempt {index}", path=state, now=now + index)
with pytest.raises(self_restart.RestartError, match="looping"):
self_restart.check_loop_guard(self_restart.load(state), now=now + 10)
def test_old_restarts_fall_out_of_the_window(state):
now = 1000.0
for index in range(self_restart.MAX_RESTARTS):
self_restart.arm(f"attempt {index}", path=state, now=now + index)
later = now + self_restart.WINDOW_SECONDS + 60
self_restart.check_loop_guard(self_restart.load(state), now=later) # no exception
assert self_restart.recent_restarts(self_restart.load(state), now=later) == []
+16 -2
View File
@@ -27,7 +27,8 @@ def test_converse_returns_reply_directly():
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response({"type": "reply", "text": "hello there"})
result = server_client.converse("hi")
assert result == "hello there"
assert result.text == "hello there"
assert result.voice_id == "" # no speak_as on this reply
post.assert_called_once()
args, kwargs = post.call_args
assert args[0] == "http://test-server:5002/desk/converse"
@@ -43,7 +44,7 @@ def test_converse_relays_a_command_then_returns_reply():
with patch.object(server_client.requests, "post", side_effect=responses) as post:
on_command = MagicMock(return_value="[exit 0]\nhi")
result = server_client.converse("run echo hi", on_command=on_command)
assert result == "done"
assert result.text == "done"
on_command.assert_called_once_with("echo hi")
# second call was to /desk/tool_result with the command's output
second_call = post.call_args_list[1]
@@ -53,6 +54,19 @@ def test_converse_relays_a_command_then_returns_reply():
}
def test_converse_carries_a_speak_as_voice_back_with_the_reply():
"""The server tags a reply with the voice it picked (speak_as); this
client is what actually speaks in it, so the id has to survive the
return trip rather than being dropped with the rest of the payload."""
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response({
"type": "reply", "text": "Ahoy there.",
"voice_id": "hnhGxwvHP8fc469w51rM", "voice_name": "Terence",
})
result = server_client.converse("talk like a pirate")
assert result == server_client.Reply("Ahoy there.", "hnhGxwvHP8fc469w51rM", "Terence")
def test_converse_raises_server_error_on_error_payload():
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response({"type": "error", "error": "unauthorized"})
+26 -1
View File
@@ -8,7 +8,8 @@ import numpy as np
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.audio.tts import chunks_to_int16
from bolt_pet import config as tts_config
from bolt_pet.audio.tts import chunks_to_int16, model_for, voice_for
from bolt_pet.audio.wake_word import NearMissLog
@@ -80,3 +81,27 @@ def test_clear_resets_peak_and_entries():
log.observe(0.45, threshold=0.5, timestamp=1.0)
log.clear()
assert log.entries() == [] and log.peak == 0.0
# ── voice / model selection (server speak_as) ───────────────────────────────
def test_the_override_voice_wins_over_the_configured_one(monkeypatch):
monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "DEFAULT")
assert voice_for("VOICE1") == "VOICE1"
assert voice_for("") == "DEFAULT"
assert voice_for(None) == "DEFAULT"
def test_english_replies_in_the_default_voice_use_the_default_model(monkeypatch):
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
assert model_for("all good here", None) == "eleven_flash_v2"
def test_a_picked_voice_or_non_english_text_uses_the_multilingual_model(monkeypatch):
# eleven_flash_v2 is English-only: it would read either of these as
# mangled phonetic English rather than failing outright.
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
assert model_for("all good here", "VOICE1") == "eleven_flash_v2_5"
assert model_for("こんにちは", None) == "eleven_flash_v2_5"